{"object":"list","data":[{"id":"pixtral-12b","object":"model","canonical_slug":"mistral/pixtral-12b","hugging_face_id":"mistralai/Pixtral-12B-2409","name":"Pixtral 12B","created":1765024302,"description":"Pixtral 12B is Mistral's multimodal model for image understanding, visual question answering, and document analysis.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-09-01","last_updated":"2024-09-01","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/pixtral-12b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Scaleway","slug":"scaleway"}],"pricing":{"prompt":"0.0000002","completion":"0.0000002","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"mistral","slug":"pixtral-12b","tags":["open-source","vision","multimodal","instruction-tuned"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"glm-5.2","object":"model","canonical_slug":"zhipu/glm-5.2","hugging_face_id":"zai-org/GLM-5.2","name":"GLM 5.2","created":1781703617,"description":"GLM-5.2 is Z AI's long-context model for reasoning, coding, tool use, and multilingual chat.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","seed","stop","stream","stream_options","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-06-13","last_updated":"2026-06-13","reasoning":{"mandatory":false,"supported_efforts":["high","max"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.2/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Regolo","slug":"regolo"},{"name":"Nebul","slug":"nebul"},{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"GreenPT","slug":"greenpt"},{"name":"Mistral AI","slug":"mistral"},{"name":"Scaleway","slug":"scaleway"},{"name":"Inceptron","slug":"inceptron"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000011","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"glm-5.2","tags":["long-context","coding","reasoning","function-calling"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"bge-m3","object":"model","canonical_slug":"baai/bge-m3","hugging_face_id":"BAAI/bge-m3","name":"BGE M3","created":1769169016,"description":"BGE M3 is BAAI's multilingual embedding model for dense, sparse, and multi-vector retrieval with long-context inputs.","context_length":8192,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["dimensions","encoding_format"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-01-27","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/baai/bge-m3/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"},{"name":"OVHcloud","slug":"ovhcloud"},{"name":"IONOS Cloud","slug":"ionos"}],"pricing":{"prompt":"0.00000001","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"baai","slug":"bge-m3","tags":["open-source","embedding","multilingual","retrieval"],"author_info":{"slug":"baai","name":"BAAI","display_name":"Beijing Academy of AI","icon_url":null,"gradient_from":"from-red-600","gradient_to":"to-red-800","gradient_via":null,"website_url":"https://www.baai.ac.cn"}},{"id":"ministral-3-14b","object":"model","canonical_slug":"mistral/ministral-3-14b","hugging_face_id":"","name":"Ministral 3 14B","created":1767979257,"description":"Ministral 3 14B is an open Mistral AI model for efficient text and vision workloads.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-12-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/ministral-3-14b/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"},{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.00000022","completion":"0.00000022","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000022","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"ministral-3-14b","tags":["open-source","vision","lightweight","long-context"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"glm-4.7-flash","object":"model","canonical_slug":"zhipu/glm-4.7-flash","hugging_face_id":"","name":"GLM 4.7 Flash","created":1777454590,"description":"GLM 4.7 Flash is Z.AI's faster GLM 4.7 variant available through Amazon Bedrock.","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":202752,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-01-19","last_updated":"2026-01-19","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-4.7-flash/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000008","completion":"0.00000048","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"zhipu","slug":"glm-4.7-flash","tags":["proprietary","instruction-tuned"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"minimax-m3","object":"model","canonical_slug":"minimax/minimax-m3","hugging_face_id":"MiniMaxAI/MiniMax-M3","name":"MiniMax M3","created":1781703617,"description":"MiniMax M3 is a long-context model for chat, coding, reasoning, and agentic workflows.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"MiniMax","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-06-01","last_updated":"2026-06-01","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m3/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000004","completion":"0.000002","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000001","input_cache_write":"0","discount":1,"currency":"USD"},"author":"minimax","slug":"minimax-m3","tags":["long-context","coding","reasoning","agentic"],"author_info":{"slug":"minimax","name":"MiniMax","display_name":"MiniMax AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-indigo-600","gradient_via":null,"website_url":"https://minimaxi.com"}},{"id":"glm-5.2-ponytail","object":"model","canonical_slug":"zhipu/glm-5.2-ponytail","hugging_face_id":"zai-org/GLM-5.2","name":"GLM 5.2 Ponytail","created":1788031623,"description":"GLM-5.2 served by GreenPT with the Ponytail code-compression ruleset built in: the same upstream weights, context and price per token as glm-5.2, with a YAGNI-first \"lazy senior developer\" bias for generated code (standard library and platform natives before custom code, less boilerplate, no speculative abstraction) while keeping requested validation and security. The ruleset is injected as a system prompt ahead of yours and is billed as input tokens, so the saving comes from generating fewer output tokens.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-08-04","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["none","minimal","low","medium","high"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.2-ponytail/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000011","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"glm-5.2-ponytail","tags":["long-context","coding","reasoning","output-compression"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"claude-sonnet-5","object":"model","canonical_slug":"anthropic/claude-sonnet-5","hugging_face_id":"","name":"Claude Sonnet 5","created":1783589719,"description":"Claude Sonnet 5 is Anthropic's near-frontier balanced model with a 1M-token context window and always-on adaptive thinking, delivering strong coding, reasoning, and agentic performance.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","tool_choice","tools","top_k","top_p"],"supported_voices":null,"knowledge_cutoff":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-sonnet-5/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000022","completion":"0.000011","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000022","input_cache_write":"0.00000275","discount":1,"currency":"USD"},"author":"anthropic","slug":"claude-sonnet-5","tags":["proprietary","coding","reasoning","agentic","long-context","vision"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}},{"id":"gpt-4.1-mini","object":"model","canonical_slug":"openai/gpt-4.1-mini","hugging_face_id":"","name":"GPT-4.1 Mini","created":1768138176,"description":"GPT-4.1 Mini is OpenAI's smaller multimodal GPT-4.1 model for cost-efficient long-context, tool, and structured-output workloads.","context_length":1047576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-04-14","last_updated":"2025-04-14","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4.1-mini/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.00000044","completion":"0.00000176","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000011","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-4.1-mini","tags":["multimodal","function-calling","long-context","efficient"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"gpt-5-mini","object":"model","canonical_slug":"openai/gpt-5-mini","hugging_face_id":"","name":"GPT-5 Mini","created":1767619876,"description":"GPT-5 Mini is OpenAI's smaller GPT-5 reasoning model for cost-efficient multimodal, tool, and long-context workloads.","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","reasoning_effort","response_format","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-mini/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.000000275","completion":"0.0000022","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000000275","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-5-mini","tags":["reasoning","long-context","efficient","coding","vision"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"hermes-4-405b","object":"model","canonical_slug":"nousresearch/hermes-4-405b","hugging_face_id":"NousResearch/Hermes-4-405B","name":"Hermes 4 405B","created":1768127608,"description":"NousResearch Hermes 4 405B is an open-weight language model for instruction following, chat, tool use, and reasoning-oriented workflows.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-08-26","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/nousresearch/hermes-4-405b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebius","slug":"nebius"}],"pricing":{"prompt":"0.000001","completion":"0.000003","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"nousresearch","slug":"hermes-4-405b","tags":["open-source","instruction-tuned","function-calling","json-mode"],"author_info":{"slug":"nousresearch","name":"Nous Research","display_name":null,"icon_url":null,"gradient_from":"from-emerald-600","gradient_to":"to-teal-700","gradient_via":null,"website_url":"https://nousresearch.com"}},{"id":"qwen3-coder-next-80b","object":"model","canonical_slug":"alibaba/qwen3-coder-next-80b","hugging_face_id":"Qwen/Qwen3-Coder-Next","name":"Qwen3 Coder Next 80B","created":1777453485,"description":"Qwen3-Coder-Next is Alibaba's open-weight 80B total, 3B active MoE coding model for coding agents, long-horizon tool use, and local development workflows.","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-02-03","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3-coder-next-80b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"IONOS Cloud","slug":"ionos"}],"pricing":{"prompt":"0.00000017","completion":"0.00000089","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"qwen3-coder-next-80b","tags":["open-source","coding","agentic","function-calling","long-context","moe"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"green-embedding","object":"model","canonical_slug":"greenpt/green-embedding","hugging_face_id":"","name":"Green Embedding","created":1768905490,"description":"High-quality text embeddings for semantic search, similarity matching, and vector database operations. EU-hosted.","context_length":32000,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-06-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/greenpt/green-embedding/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000002","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"greenpt","slug":"green-embedding","tags":["eu-native","sustainable","embedding"],"author_info":{"slug":"greenpt","name":"GreenPT","display_name":"GreenPT","icon_url":null,"gradient_from":"from-green-600","gradient_to":"to-green-700","gradient_via":null,"website_url":"https://greenpt.ai"}},{"id":"kimi-k2.6","object":"model","canonical_slug":"moonshot/kimi-k2.6","hugging_face_id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","created":1777453828,"description":"Kimi K2.6 is Moonshot AI's open-weight reasoning model for long-horizon coding, agentic execution, and multimodal reasoning.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"TikTokenTokenizer","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","seed","stop","stream","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-04-21","last_updated":"2026-04-21","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/moonshot/kimi-k2.6/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Inceptron","slug":"inceptron"},{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"GreenPT","slug":"greenpt"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.00000066","completion":"0.000003751","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000022","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"moonshot","slug":"kimi-k2.6","tags":["open-source","moe","long-context","vision","coding","reasoning"],"author_info":{"slug":"moonshot","name":"Moonshot","display_name":"Moonshot AI","icon_url":null,"gradient_from":"from-indigo-500","gradient_to":"to-purple-600","gradient_via":null,"website_url":"https://moonshot.ai"}},{"id":"gpt-5.6-luna","object":"model","canonical_slug":"openai/gpt-5.6-luna","hugging_face_id":"","name":"GPT-5.6 Luna","created":1784716776,"description":"GPT-5.6 Luna is OpenAI's fastest, lowest-cost GPT-5.6 reasoning model for everyday tasks like summarization, drafting, and routine automation.","context_length":1050000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","reasoning_effort","response_format","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-07-09","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.6-luna/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.00000022","completion":"0.00000132","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000022","input_cache_write":"0.000000275","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-5.6-luna","tags":["reasoning","long-context","efficient","high-throughput","vision"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"glm-5.2-honey-lite","object":"model","canonical_slug":"zhipu/glm-5.2-honey-lite","hugging_face_id":"zai-org/GLM-5.2","name":"GLM 5.2 Honey Lite","created":1788029782,"description":"GLM-5.2 served by GreenPT with the Honey-lite output-compression ruleset injected as a system prompt: same upstream weights and same price per token as GLM-5.2, but shorter answers. Honey compresses both code and prose; the -lite tier's rule is to keep the explanation intact, so it sits at the bottom of the family's range (around -36% output tokens on coding work). The ruleset is merged after the caller's own system message, so it wins on conflict.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","stop","stream","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":null,"last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["none","minimal","low","medium","high"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.2-honey-lite/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000011","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"glm-5.2-honey-lite","tags":["open-source","long-context","coding","reasoning","function-calling"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"qwen3-235b-a22b-instruct","object":"model","canonical_slug":"alibaba/qwen3-235b-a22b-instruct","hugging_face_id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct","created":1765183068,"description":"Qwen3 235B A22B Instruct is Alibaba's Mixture-of-Experts model for multilingual chat, reasoning, coding, and tool use.","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":131000,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-07-21","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3-235b-a22b-instruct/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebius","slug":"nebius"},{"name":"AWS Bedrock","slug":"aws-bedrock"},{"name":"Scaleway","slug":"scaleway"},{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"GreenPT","slug":"greenpt"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.000000072","completion":"0.000000464","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000018","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"qwen3-235b-a22b-instruct","tags":["open-source","moe","multilingual","function-calling"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"gemma-3-27b-it","object":"model","canonical_slug":"google/gemma-3-27b-it","hugging_face_id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","created":1765183068,"description":"Gemma 3 27B IT is Google's multimodal instruction-tuned model with text and image input support.","context_length":110000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"top_provider":{"context_length":110000,"max_completion_tokens":110000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-03-12","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-3-27b-it/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebius","slug":"nebius"}],"pricing":{"prompt":"0.0000001","completion":"0.0000003","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"google","slug":"gemma-3-27b-it","tags":["open-source","vision","multimodal","multilingual","instruction-tuned"],"author_info":{"slug":"google","name":"Google","display_name":null,"icon_url":"/images/logos/gemini.webp","gradient_from":"from-blue-500","gradient_to":"to-yellow-500","gradient_via":"via-green-500","website_url":"https://deepmind.google"}},{"id":"deepseek-v4-flash-0731","object":"model","canonical_slug":"deepseek/deepseek-v4-flash-0731","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","created":1785940225,"description":"DeepSeek V4 Flash 0731 is the official release of DeepSeek V4 Flash, superseding the April preview. A sparse mixture-of-experts text model with a 1M-token context window, function calling, and optional reasoning, retrained for agentic and coding work.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":1048576,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","seed","stop","stream","stream_options","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-07-31","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-flash-0731/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Tensorix","slug":"tensorix"},{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"Scaleway","slug":"scaleway"},{"name":"GreenPT","slug":"greenpt"},{"name":"AKI.IO","slug":"aki"},{"name":"Inceptron","slug":"inceptron"}],"pricing":{"prompt":"0.00000013","completion":"0.00000028","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000003","input_cache_write":"0","discount":1,"currency":"USD"},"author":"deepseek","slug":"deepseek-v4-flash-0731","tags":["open-source","moe","long-context","function-calling","reasoning"],"author_info":{"slug":"deepseek","name":"DeepSeek","display_name":null,"icon_url":"/images/logos/deepseek.png","gradient_from":"from-[#0066ff]","gradient_to":"to-[#00ccff]","gradient_via":null,"website_url":"https://deepseek.com"}},{"id":"magistral-small","object":"model","canonical_slug":"mistral/magistral-small","hugging_face_id":"","name":"Magistral Small 1.2","created":1767979257,"description":"Magistral Small 1.2 is Mistral AI's small reasoning model for transparent and multilingual reasoning.","context_length":40000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":40000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-03-17","last_updated":"2025-03-17","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/magistral-small/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.00000055","completion":"0.00000165","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"magistral-small","tags":["open-source","reasoning","multimodal"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"mistral-small-4","object":"model","canonical_slug":"mistral/mistral-small-4","hugging_face_id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral Small 4","created":1777454436,"description":"Mistral Small 4 is Mistral AI's open hybrid MoE model unifying instruct, reasoning, coding, and multimodal workloads.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-03-16","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-small-4/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Regolo","slug":"regolo"},{"name":"AKI.IO","slug":"aki"},{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.000000165","completion":"0.00000066","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000000165","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"mistral-small-4","tags":["open-source","moe","vision","reasoning","coding","function-calling","long-context"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"zhipu/glm-5.3","object":"model","canonical_slug":"zhipu/glm-5.3","hugging_face_id":"zai-org/GLM-5.3","name":"GLM 5.3","created":1788278679,"description":"GLM-5.3 is Z.AI's open-weight, text-only reasoning model for complex coding, long-horizon agentic work, tool use, and multilingual chat. It uses the same base model as GLM-5.2, with its gains coming from post-training.","context_length":524288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":524288,"max_completion_tokens":81920,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":null,"last_updated":null,"reasoning":{"mandatory":true,"supported_efforts":["low","high","max"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/zhipu/glm-5.3/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AKI.IO","slug":"aki"},{"name":"Inceptron","slug":"inceptron"}],"pricing":{"prompt":"0.000001","completion":"0.0000035","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000025","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"zhipu/glm-5.3","tags":[],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"green-r","object":"model","canonical_slug":"greenpt/green-r","hugging_face_id":"","name":"GreenR","created":1768905404,"description":"Reasoning model specialized in logical reasoning and problem-solving. Ideal for complex analytical tasks and mathematical problems. Based on GPT-OSS 120B, EU-hosted.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-08-01","last_updated":null,"reasoning":{"mandatory":true,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/greenpt/green-r/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.00000035","completion":"0.00000095","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"greenpt","slug":"green-r","tags":["eu-native","sustainable","reasoning"],"author_info":{"slug":"greenpt","name":"GreenPT","display_name":"GreenPT","icon_url":null,"gradient_from":"from-green-600","gradient_to":"to-green-700","gradient_via":null,"website_url":"https://greenpt.ai"}},{"id":"nova-pro","object":"model","canonical_slug":"amazon/nova-pro","hugging_face_id":"","name":"Amazon Nova Pro","created":1769967814,"description":"Amazon Nova Pro is a highly capable multimodal model balancing performance and cost for complex tasks.","context_length":300000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":300000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-12-03","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-pro/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000105","completion":"0.0000042","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000002625","input_cache_write":"0","discount":1,"currency":"USD"},"author":"amazon","slug":"nova-pro","tags":["proprietary","multimodal","vision","video","reasoning"],"author_info":{"slug":"amazon","name":"Amazon","display_name":"Amazon Web Services","icon_url":"/images/logos/aws.webp","gradient_from":"from-orange-500","gradient_to":"to-yellow-500","gradient_via":null,"website_url":"https://aws.amazon.com/bedrock"}},{"id":"qwen3-32b","object":"model","canonical_slug":"alibaba/qwen3-32b","hugging_face_id":"Qwen/Qwen3-32B","name":"Qwen3 32B","created":1767979257,"description":"Qwen3 32B is Alibaba's dense multilingual model with tool calling and optional reasoning mode for math, code, and general instruction-following tasks.","context_length":40960,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":40960,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-03-05","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":true},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3-32b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000002","completion":"0.00000079","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"qwen3-32b","tags":["open-source","reasoning","multilingual","tool-calling"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"minimax-m2.5","object":"model","canonical_slug":"minimax/minimax-m2.5","hugging_face_id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax M2.5","created":1773690794,"description":"MiniMax M2.5 is a MiniMax model for agentic coding, tool use, reasoning, and high-throughput chat workloads.","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"MiniMax","instruct_type":null},"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","stop","stream","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-02-12","last_updated":"2026-02-12","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.5/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000003","completion":"0.0000012","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000075","input_cache_write":"0","discount":1,"currency":"USD"},"author":"minimax","slug":"minimax-m2.5","tags":["open-source","moe","coding","reasoning","function-calling"],"author_info":{"slug":"minimax","name":"MiniMax","display_name":"MiniMax AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-indigo-600","gradient_via":null,"website_url":"https://minimaxi.com"}},{"id":"cohere-rerank-4.0-pro","object":"model","canonical_slug":"cohere/cohere-rerank-4.0-pro","hugging_face_id":"","name":"Cohere Rerank 4.0 Pro","created":1784714511,"description":"Cohere Rerank 4.0 Pro is Cohere's highest-quality multilingual cross-encoder reranker. It reorders documents by semantic relevance to a query for search and retrieval-augmented generation, handles complex reasoning-heavy queries and semi-structured data across 100+ languages, and supports a 32K context window. Available through Microsoft Foundry.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-04-06","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/cohere/cohere-rerank-4.0-pro/endpoints"},"supported_api_endpoints":["rerank"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"author":"cohere","slug":"cohere-rerank-4.0-pro","tags":["proprietary","rerank","multilingual"],"author_info":{"slug":"cohere","name":"Cohere","display_name":null,"icon_url":"/images/logos/cohere.webp","gradient_from":"from-purple-600","gradient_to":"to-purple-800","gradient_via":null,"website_url":"https://cohere.com"}},{"id":"alibaba/qwen3-vl-reranker-8b","object":"model","canonical_slug":"alibaba/qwen3-vl-reranker-8b","hugging_face_id":"Qwen/Qwen3-VL-Reranker-8B","name":"Qwen3 VL Reranker 8B","created":1788117786,"description":"","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["top_n"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-01-08","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/alibaba/qwen3-vl-reranker-8b/endpoints"},"supported_api_endpoints":["rerank"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"IONOS Cloud","slug":"ionos"}],"pricing":{"prompt":"0.000000045","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"alibaba/qwen3-vl-reranker-8b","tags":[],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"qwen3-embedding-8b","object":"model","canonical_slug":"alibaba/qwen3-embedding-8b","hugging_face_id":"Qwen/Qwen3-Embedding-8B","name":"Qwen3 Embedding 8B","created":1767963057,"description":"Qwen3 Embedding 8B is Alibaba's embedding model for multilingual semantic search, retrieval, and representation learning.","context_length":40960,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":40960,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["dimensions","encoding_format"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-06-05","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3-embedding-8b/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"Scaleway","slug":"scaleway"},{"name":"Nebul","slug":"nebul"},{"name":"GreenPT","slug":"greenpt"},{"name":"OVHcloud","slug":"ovhcloud"},{"name":"Tensorix","slug":"tensorix"},{"name":"Regolo","slug":"regolo"},{"name":"Nebius","slug":"nebius"}],"pricing":{"prompt":"0.00000001","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"qwen3-embedding-8b","tags":["open-source","embedding","multilingual","retrieval"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"nova-lite","object":"model","canonical_slug":"amazon/nova-lite","hugging_face_id":"","name":"Amazon Nova Lite","created":1769967814,"description":"Amazon Nova Lite is a multimodal model supporting text, images, and video with low-latency responses.","context_length":300000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":300000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-12-03","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-lite/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.000000078","completion":"0.000000312","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000000195","input_cache_write":"0","discount":1,"currency":"USD"},"author":"amazon","slug":"nova-lite","tags":["proprietary","multimodal","vision","video"],"author_info":{"slug":"amazon","name":"Amazon","display_name":"Amazon Web Services","icon_url":"/images/logos/aws.webp","gradient_from":"from-orange-500","gradient_to":"to-yellow-500","gradient_via":null,"website_url":"https://aws.amazon.com/bedrock"}},{"id":"gemma-4","object":"model","canonical_slug":"google/gemma-4","hugging_face_id":"","name":"Gemma 4","created":1777452602,"description":"Gemma 4 is Google DeepMind's open model family built from Gemini 3 research and technology for high intelligence-per-parameter. The 31B variant offers long-context multimodal reasoning, function-calling support, multilingual understanding, and efficient deployment on personal hardware.","context_length":256000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["chat_template_kwargs","max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-04-02","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-4/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Infercom","slug":"infercom"},{"name":"AKI.IO","slug":"aki"},{"name":"GreenPT","slug":"greenpt"},{"name":"Regolo","slug":"regolo"}],"pricing":{"prompt":"0.0000001","completion":"0.0000005","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000001","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"google","slug":"gemma-4","tags":["open-source","multimodal","vision","long-context","function-calling","multilingual","public-preview"],"author_info":{"slug":"google","name":"Google","display_name":null,"icon_url":"/images/logos/gemini.webp","gradient_from":"from-blue-500","gradient_to":"to-yellow-500","gradient_via":"via-green-500","website_url":"https://deepmind.google"}},{"id":"minimax-m2.7","object":"model","canonical_slug":"minimax/minimax-m2.7","hugging_face_id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax M2.7","created":1779711156,"description":"MiniMax M2.7 is MiniMax's 229B parameter reasoning model for agentic coding, software engineering, tool use, professional productivity workflows, and native multi-agent orchestration.","context_length":192000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"MiniMax","instruct_type":null},"top_provider":{"context_length":192000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-03-18","last_updated":"2026-03-18","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.7/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Infercom","slug":"infercom"}],"pricing":{"prompt":"0.0000006","completion":"0.0000024","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"minimax","slug":"minimax-m2.7","tags":["open-source","moe","long-context","coding","reasoning","agentic"],"author_info":{"slug":"minimax","name":"MiniMax","display_name":"MiniMax AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-indigo-600","gradient_via":null,"website_url":"https://minimaxi.com"}},{"id":"devstral-2-123b-instruct-2512","object":"model","canonical_slug":"mistral/devstral-2-123b-instruct-2512","hugging_face_id":"mistralai/Devstral-2-123B-Instruct-2512","name":"Devstral 2 123B Instruct","created":1777452602,"description":"Devstral 2 123B Instruct is Mistral's long-context code model for agentic software engineering and multi-file reasoning tasks.","context_length":200000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-01-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/devstral-2-123b-instruct-2512/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"},{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000048","completion":"0.0000024","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"devstral-2-123b-instruct-2512","tags":["open-source","coding","agentic","long-context","instruction-tuned"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"gpt-5.6-terra","object":"model","canonical_slug":"openai/gpt-5.6-terra","hugging_face_id":"","name":"GPT-5.6 Terra","created":1784716776,"description":"GPT-5.6 Terra is OpenAI's balanced reasoning model for high-volume business workloads such as customer support, internal tools, and document analysis.","context_length":1050000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","reasoning_effort","response_format","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-07-09","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.6-terra/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.0000022","completion":"0.0000132","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000022","input_cache_write":"0.00000275","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-5.6-terra","tags":["reasoning","long-context","general-purpose","efficient","vision"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"kimi-k2.5","object":"model","canonical_slug":"moonshot/kimi-k2.5","hugging_face_id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","created":1771080676,"description":"Kimi K2.5 is Moonshot AI's long-context multimodal model for coding, tool use, and general chat.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Kimi","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-08-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/moonshot/kimi-k2.5/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000005","completion":"0.0000028","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000125","input_cache_write":"0","discount":1,"currency":"USD"},"author":"moonshot","slug":"kimi-k2.5","tags":["open-source","moe","long-context","vision","function-calling"],"author_info":{"slug":"moonshot","name":"Moonshot","display_name":"Moonshot AI","icon_url":null,"gradient_from":"from-indigo-500","gradient_to":"to-purple-600","gradient_via":null,"website_url":"https://moonshot.ai"}},{"id":"qwen-2.5-vl-72b-instruct","object":"model","canonical_slug":"alibaba/qwen-2.5-vl-72b-instruct","hugging_face_id":"Qwen/Qwen2.5-VL-72B-Instruct","name":"Qwen 2.5 VL 72B Instruct","created":1768127608,"description":"Qwen 2.5 VL 72B Instruct is Alibaba's vision-language model for image understanding, OCR, document analysis, and multimodal instruction following.","context_length":32768,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen2","instruct_type":null},"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-01-27","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen-2.5-vl-72b-instruct/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"OVHcloud","slug":"ovhcloud"}],"pricing":{"prompt":"0.00000091","completion":"0.00000091","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"alibaba","slug":"qwen-2.5-vl-72b-instruct","tags":["open-source","vision","multimodal","document-analysis"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"voxtral-small-24b","object":"model","canonical_slug":"mistral/voxtral-small-24b","hugging_face_id":"mistralai/Voxtral-Small-24B-2507","name":"Voxtral Small 24B","created":1765183068,"description":"Voxtral Small 24B is Mistral AI's open audio-capable chat model for speech understanding and text generation.","context_length":32000,"architecture":{"modality":"text+audio->text","input_modalities":["text","audio"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-07-25","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/voxtral-small-24b/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.00000011","completion":"0.00000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0.00007333337","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"voxtral-small-24b","tags":["open-source","audio","function-calling","multilingual"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"cohere-embed-english-v3","object":"model","canonical_slug":"cohere/cohere-embed-english-v3","hugging_face_id":"","name":"Cohere Embed English v3","created":1777454590,"description":"Cohere Embed English v3 is an English text embedding model available through Amazon Bedrock.","context_length":512,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-11-02","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/cohere/cohere-embed-english-v3/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000001","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"cohere","slug":"cohere-embed-english-v3","tags":["proprietary","embedding"],"author_info":{"slug":"cohere","name":"Cohere","display_name":null,"icon_url":"/images/logos/cohere.webp","gradient_from":"from-purple-600","gradient_to":"to-purple-800","gradient_via":null,"website_url":"https://cohere.com"}},{"id":"claude-haiku-4.5","object":"model","canonical_slug":"anthropic/claude-haiku-4.5","hugging_face_id":"","name":"Claude Haiku 4.5","created":1769967813,"description":"Claude Haiku 4.5 is Anthropic's fastest model, optimized for speed and cost-efficiency while maintaining strong performance.","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"claude","instruct_type":"claude"},"top_provider":{"context_length":200000,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-10-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-haiku-4.5/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000011","completion":"0.0000055","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000011","input_cache_write":"0.000001375","discount":1,"currency":"USD"},"author":"anthropic","slug":"claude-haiku-4.5","tags":["proprietary","fast","cost-efficient","vision"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}},{"id":"inf-retriever-v1","object":"model","canonical_slug":"infly/inf-retriever-v1","hugging_face_id":"infly/inf-retriever-v1","name":"INF Retriever v1","created":1783347345,"description":"INF Retriever v1 is INF Tech's LLM-based dense retrieval embedding model, optimized for English and Chinese retrieval.","context_length":32768,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-12-23","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/infly/inf-retriever-v1/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"}],"pricing":{"prompt":"0.00000009","completion":"0.00000045","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"infly","slug":"inf-retriever-v1","tags":["open-source","embedding","retrieval","multilingual"],"author_info":{"slug":"infly","name":"INF Tech","display_name":"INF Tech","icon_url":null,"gradient_from":null,"gradient_to":null,"gradient_via":null,"website_url":"https://huggingface.co/infly"}},{"id":"gpt-5.6-sol","object":"model","canonical_slug":"openai/gpt-5.6-sol","hugging_face_id":"","name":"GPT-5.6 Sol","created":1784716776,"description":"GPT-5.6 Sol is OpenAI's flagship reasoning model for the hardest problems, including complex coding, security research, and long-horizon agentic work.","context_length":1050000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","reasoning_effort","response_format","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-07-09","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.6-sol/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.0000055","completion":"0.000033","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000055","input_cache_write":"0.000006875","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-5.6-sol","tags":["reasoning","long-context","frontier","agentic","coding","vision"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"llama-3.1-8b-instruct","object":"model","canonical_slug":"meta/llama-3.1-8b-instruct","hugging_face_id":"meta-llama/Meta-Llama-3.1-8B-Instruct","name":"Llama 3.1 8B Instruct","created":1765024302,"description":"Meta Llama 3.1 8B Instruct is a compact multilingual instruction-tuned model for chat, tool use, and efficient assistant workloads.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-07-23","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/meta/llama-3.1-8b-instruct/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AKI.IO","slug":"aki"},{"name":"IONOS Cloud","slug":"ionos"},{"name":"Nebius","slug":"nebius"}],"pricing":{"prompt":"0.00000002","completion":"0.00000006","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"meta","slug":"llama-3.1-8b-instruct","tags":["open-source","instruction-tuned","multilingual","function-calling"],"author_info":{"slug":"meta","name":"Meta","display_name":"Meta AI","icon_url":"/images/logos/meta.webp","gradient_from":"from-blue-600","gradient_to":"to-blue-800","gradient_via":null,"website_url":"https://ai.meta.com"}},{"id":"green-l-raw","object":"model","canonical_slug":"greenpt/green-l-raw","hugging_face_id":"","name":"GreenL Raw","created":1768905404,"description":"Large generative model with custom system prompt support. Provides more control over model behavior and responses. EU-hosted.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-06-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/greenpt/green-l-raw/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.00000025","completion":"0.0000008","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"greenpt","slug":"green-l-raw","tags":["eu-native","sustainable","instruction-tuned","raw"],"author_info":{"slug":"greenpt","name":"GreenPT","display_name":"GreenPT","icon_url":null,"gradient_from":"from-green-600","gradient_to":"to-green-700","gradient_via":null,"website_url":"https://greenpt.ai"}},{"id":"gpt-4.1","object":"model","canonical_slug":"openai/gpt-4.1","hugging_face_id":"","name":"GPT-4.1","created":1768138176,"description":"GPT-4.1 is OpenAI's multimodal general-purpose model for instruction following, coding, long-context work, function calling, and structured outputs.","context_length":1047576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-04-14","last_updated":"2025-04-14","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4.1/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.0000022","completion":"0.0000088","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000055","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-4.1","tags":["multimodal","function-calling","long-context","coding"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"green-l","object":"model","canonical_slug":"greenpt/green-l","hugging_face_id":"","name":"GreenL","created":1768905404,"description":"Large generative model optimized for sustainability. Best for general conversations, content creation, and text analysis. Based on Mistral architecture, hosted entirely in the EU.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-01-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/greenpt/green-l/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.00000025","completion":"0.0000008","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"greenpt","slug":"green-l","tags":["eu-native","sustainable","instruction-tuned"],"author_info":{"slug":"greenpt","name":"GreenPT","display_name":"GreenPT","icon_url":null,"gradient_from":"from-green-600","gradient_to":"to-green-700","gradient_via":null,"website_url":"https://greenpt.ai"}},{"id":"deepseek-v3.1","object":"model","canonical_slug":"deepseek/deepseek-v3.1","hugging_face_id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","created":1774895919,"description":"DeepSeek V3.1 is a mixture-of-experts chat model with strong coding and instruction-following capabilities.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-08-21","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v3.1/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000058","completion":"0.00000168","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"deepseek","slug":"deepseek-v3.1","tags":["open-source","long-context","coding"],"author_info":{"slug":"deepseek","name":"DeepSeek","display_name":null,"icon_url":"/images/logos/deepseek.png","gradient_from":"from-[#0066ff]","gradient_to":"to-[#00ccff]","gradient_via":null,"website_url":"https://deepseek.com"}},{"id":"gemma-2-2b-it","object":"model","canonical_slug":"google/gemma-2-2b-it","hugging_face_id":"google/gemma-2-2b-it","name":"Gemma 2 2B IT","created":1768127608,"description":"Gemma 2 2B IT is Google's compact instruction-tuned model from the Gemma 2 family for efficient chat and assistant workloads.","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-06-27","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-2-2b-it/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebius","slug":"nebius"}],"pricing":{"prompt":"0.00000002","completion":"0.00000006","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"google","slug":"gemma-2-2b-it","tags":["open-source","instruction-tuned","efficient"],"author_info":{"slug":"google","name":"Google","display_name":null,"icon_url":"/images/logos/gemini.webp","gradient_from":"from-blue-500","gradient_to":"to-yellow-500","gradient_via":"via-green-500","website_url":"https://deepmind.google"}},{"id":"qwen3.5-9b","object":"model","canonical_slug":"alibaba/qwen3.5-9b","hugging_face_id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","created":1777453993,"description":"Qwen3.5 9B is Alibaba's compact Qwen3.5 model for efficient long-context chat, reasoning, and tool use.","context_length":262000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":262000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-02-23","last_updated":"2026-02-23","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3.5-9b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Regolo","slug":"regolo"},{"name":"OVHcloud","slug":"ovhcloud"},{"name":"IONOS Cloud","slug":"ionos"},{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.00000007","completion":"0.00000035","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"alibaba","slug":"qwen3.5-9b","tags":["open-source","long-context","reasoning","efficient"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"mistral-small-24b","object":"model","canonical_slug":"mistral/mistral-small-24b","hugging_face_id":"mistralai/Mistral-Small-24B-Instruct-2501","name":"Mistral Small 24B Instruct","created":1770318640,"description":"Mistral Small 24B Instruct is a 24B parameter model with 128K context and multimodal (text + image) inputs, optimized for instruction following and long-document understanding.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-06-25","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-small-24b/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"IONOS Cloud","slug":"ionos"}],"pricing":{"prompt":"0.00000011","completion":"0.00000033","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"mistral-small-24b","tags":["open-source","instruction-tuned","vision","long-context","multilingual"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"qwen3-coder-480b-a35b-instruct","object":"model","canonical_slug":"alibaba/qwen3-coder-480b-a35b-instruct","hugging_face_id":"","name":"Qwen3 Coder 480B A35B Instruct","created":1768127608,"description":"Massive code-specialist model optimized for repo-scale coding, long-context understanding, and precise tool calls.","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-07-22","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3-coder-480b-a35b-instruct/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000045","completion":"0.0000018","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"qwen3-coder-480b-a35b-instruct","tags":["coding","moe","long-context","tool-use"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"glm-5-turbo","object":"model","canonical_slug":"zhipu/glm-5-turbo","hugging_face_id":"","name":"GLM 5 Turbo","created":1778850280,"description":"GLM-5-Turbo is Z AI's speed-optimized, API-only GLM-5-series variant for fast, low-latency agentic workloads, with tool calling and toggleable reasoning.","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-03-16","last_updated":"2026-03-16","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5-turbo/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000012","completion":"0.000004","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000003","input_cache_write":"0","discount":1,"currency":"USD"},"author":"zhipu","slug":"glm-5-turbo","tags":["long-context","agentic","reasoning","function-calling"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"glm-5.3","object":"model","canonical_slug":"zhipu/glm-5.3","hugging_face_id":"zai-org/GLM-5.3","name":"GLM 5.3","created":1787948724,"description":"GLM-5.3 is Z.ai's flagship coding and agentic-engineering model. It shares the GLM-5.2 base model, with all gains delivered through post-training, and always operates with reasoning enabled across three effort levels.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"TokenizersBackend","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-08-18","last_updated":"2026-08-25","reasoning":{"mandatory":true,"supported_efforts":["low","high","max"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.3/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Tensorix","slug":"tensorix"},{"name":"Nebul","slug":"nebul"}],"pricing":{"prompt":"0.00000147","completion":"0.00000462","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000035","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"glm-5.3","tags":["open-source","long-context","coding","agentic","reasoning","function-calling"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"claude-opus-4-6","object":"model","canonical_slug":"anthropic/claude-opus-4-6","hugging_face_id":"","name":"Claude Opus 4.6","created":1771234266,"description":"Claude Opus 4.6 is Anthropic's most capable model for complex coding, enterprise agents, and professional reasoning with a 1M context window.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"knowledge_cutoff":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-opus-4-6/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000055","completion":"0.0000275","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000055","input_cache_write":"0.000006875","discount":1,"currency":"USD"},"author":"anthropic","slug":"claude-opus-4-6","tags":["proprietary","coding","reasoning","agentic","long-context","vision"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}},{"id":"minimax-m2.1","object":"model","canonical_slug":"minimax/minimax-m2.1","hugging_face_id":"MiniMaxAI/MiniMax-M2.1","name":"MiniMax M2.1","created":1771080675,"description":"MiniMax M2.1 is a long-context Mixture-of-Experts model from MiniMax for chat, coding, reasoning, and multimodal tasks.","context_length":196608,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"MiniMax","instruct_type":null},"top_provider":{"context_length":196608,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-12-23","last_updated":"2025-12-23","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.1/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000036","completion":"0.00000144","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"minimax","slug":"minimax-m2.1","tags":["open-source","moe","long-context","vision","reasoning"],"author_info":{"slug":"minimax","name":"MiniMax","display_name":"MiniMax AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-indigo-600","gradient_via":null,"website_url":"https://minimaxi.com"}},{"id":"qwen3.8-2.4t-a95b","object":"model","canonical_slug":"alibaba/qwen3.8-2.4t-a95b","hugging_face_id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","created":1788031629,"description":"Qwen3.8 2.4T A95B is Alibaba's open-weight flagship Mixture-of-Experts model (2.4T total parameters, 95B activated per token), the open-weight counterpart of Qwen3.8-Max, for coding, professional work, research and long-horizon agentic tasks. Text-only, with reasoning always on.","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","stop","stream","stream_options","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-08-12","last_updated":null,"reasoning":{"mandatory":true,"supported_efforts":["low","medium","xhigh"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3.8-2.4t-a95b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":0,"repetition_penalty":1,"min_p":0},"providers":[{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000025","completion":"0.000006","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000625","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"qwen3.8-2.4t-a95b","tags":["open-source","moe","reasoning","coding","long-context","agentic","function-calling"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"text-embedding-3-small","object":"model","canonical_slug":"openai/text-embedding-3-small","hugging_face_id":"","name":"Text Embedding 3 Small","created":1767964798,"description":"text-embedding-3-small is OpenAI's efficient text embedding model with 1,536 output dimensions.","context_length":8192,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":"cl100k_base","instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["dimensions","encoding_format"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-01-25","last_updated":"2024-01-25","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/text-embedding-3-small/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"author":"openai","slug":"text-embedding-3-small","tags":["embedding","retrieval","efficient"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"multilingual-e5-large-instruct","object":"model","canonical_slug":"intfloat/multilingual-e5-large-instruct","hugging_face_id":"intfloat/multilingual-e5-large-instruct","name":"Multilingual E5 Large Instruct","created":1783347345,"description":"Multilingual E5 Large Instruct is an instruction-tuned multilingual embedding model for retrieval across ~100 languages.","context_length":8192,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-02-08","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/intfloat/multilingual-e5-large-instruct/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"}],"pricing":{"prompt":"0.00000012","completion":"0.0000006","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"intfloat","slug":"multilingual-e5-large-instruct","tags":["open-source","embedding","multilingual","retrieval"],"author_info":{"slug":"intfloat","name":"intfloat","display_name":null,"icon_url":null,"gradient_from":"from-sky-500","gradient_to":"to-blue-600","gradient_via":null,"website_url":"https://huggingface.co/intfloat"}},{"id":"mistral-7b-instruct","object":"model","canonical_slug":"mistral/mistral-7b-instruct","hugging_face_id":"","name":"Mistral 7B Instruct","created":1777454590,"description":"Mistral 7B Instruct is Mistral AI's 7B-parameter instruction-tuned model.","context_length":32000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":32000,"max_completion_tokens":4000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-09-28","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-7b-instruct/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000016","completion":"0.00000022","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"mistral-7b-instruct","tags":["open-weights","instruction-tuned"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"zhipu/glm-5.2-caveman-lite","object":"model","canonical_slug":"zhipu/glm-5.2-caveman-lite","hugging_face_id":"","name":"GLM 5.2 Caveman Lite","created":1788118165,"description":"","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","stop","stream","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":null,"last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/zhipu/glm-5.2-caveman-lite/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000011","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"zhipu/glm-5.2-caveman-lite","tags":[],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"mistral-small-3.2-24b","object":"model","canonical_slug":"mistral/mistral-small-3.2-24b","hugging_face_id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral Small 3.2 24B","created":1765183068,"description":"Mistral Small 3.2 is a 24B open model with image understanding, function calling, and long-context support.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-06-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-small-3.2-24b/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Scaleway","slug":"scaleway"},{"name":"OVHcloud","slug":"ovhcloud"},{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.00000009","completion":"0.00000028","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"mistral","slug":"mistral-small-3.2-24b","tags":["open-source","vision","multilingual","function-calling"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"mistral-7b-instruct-v0.3","object":"model","canonical_slug":"mistral/mistral-7b-instruct-v0.3","hugging_face_id":"mistralai/Mistral-7B-Instruct-v0.3","name":"Mistral 7B Instruct v0.3","created":1769169016,"description":"Mistral 7B Instruct v0.3 is Mistral AI's 7B instruction-following model for multilingual text generation, coding, extraction, and summarization tasks.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-05-22","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-7b-instruct-v0.3/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"OVHcloud","slug":"ovhcloud"}],"pricing":{"prompt":"0.0000001","completion":"0.0000001","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"mistral","slug":"mistral-7b-instruct-v0.3","tags":["open-source","instruction-tuned","multilingual","coding"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"ministral-3-3b","object":"model","canonical_slug":"mistral/ministral-3-3b","hugging_face_id":"","name":"Ministral 3 3B","created":1767979257,"description":"Ministral 3 3B is an open Mistral AI model for efficient text and vision workloads.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-12-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/ministral-3-3b/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.00000011","completion":"0.00000011","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000011","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"ministral-3-3b","tags":["open-source","vision","lightweight","long-context"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"cohere-embed-multilingual-v3","object":"model","canonical_slug":"cohere/cohere-embed-multilingual-v3","hugging_face_id":"","name":"Cohere Embed Multilingual v3","created":1777454590,"description":"Cohere Embed Multilingual v3 is a multilingual text embedding model available through Amazon Bedrock.","context_length":512,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-11-02","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/cohere/cohere-embed-multilingual-v3/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000001","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"cohere","slug":"cohere-embed-multilingual-v3","tags":["proprietary","embedding","multilingual"],"author_info":{"slug":"cohere","name":"Cohere","display_name":null,"icon_url":"/images/logos/cohere.webp","gradient_from":"from-purple-600","gradient_to":"to-purple-800","gradient_via":null,"website_url":"https://cohere.com"}},{"id":"codestral","object":"model","canonical_slug":"mistral/codestral","hugging_face_id":"","name":"Codestral","created":1767979257,"description":"Codestral is Mistral AI's code model for code completion, code generation, and instruction-following coding tasks.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","response_format","stop","suffix","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-08-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/codestral/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.00000033","completion":"0.00000099","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000033","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"codestral","tags":["code","fim","function-calling"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"alibaba/qwen3.8-27b","object":"model","canonical_slug":"alibaba/qwen3.8-27b","hugging_face_id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","created":1787919100,"description":"Qwen3.8 27B is Alibaba Qwen's open-weight 27B dense multimodal reasoning model for text, image, and video understanding, coding, professional work, research, and long-horizon agentic tasks.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["parallel_tool_calls","reasoning_effort","response_format","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-08-14","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["low","medium","xhigh"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/alibaba/qwen3.8-27b/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"OVHcloud","slug":"ovhcloud"},{"name":"Scaleway","slug":"scaleway"}],"pricing":{"prompt":"0.0000004","completion":"0.0000027","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"alibaba","slug":"alibaba/qwen3.8-27b","tags":["open-source","multimodal","vision","video","reasoning","coding","agentic","function-calling","structured-output","long-context"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"deepseek-r1","object":"model","canonical_slug":"deepseek/deepseek-r1","hugging_face_id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1","created":1765024302,"description":"DeepSeek R1 is DeepSeek's reasoning model for complex problem solving, math, code, and multi-step analysis.","context_length":164000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"top_provider":{"context_length":164000,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-01-20","last_updated":"2025-05-29","reasoning":{"mandatory":true,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-r1/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.00000066","completion":"0.0000026","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000165","input_cache_write":"0","discount":1,"currency":"USD"},"author":"deepseek","slug":"deepseek-r1","tags":["open-source","reasoning","long-context"],"author_info":{"slug":"deepseek","name":"DeepSeek","display_name":null,"icon_url":"/images/logos/deepseek.png","gradient_from":"from-[#0066ff]","gradient_to":"to-[#00ccff]","gradient_via":null,"website_url":"https://deepseek.com"}},{"id":"claude-opus-5","object":"model","canonical_slug":"anthropic/claude-opus-5","hugging_face_id":"","name":"Claude Opus 5","created":1784913939,"description":"Claude Opus 5 is Anthropic's most advanced Opus model, powering long-running, highly capable agents and delivering improvements in coding and professional work, with a 1M-token context window.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","tool_choice","tools","top_k","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-07-23","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-opus-5/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000055","completion":"0.0000275","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000055","input_cache_write":"0.000006875","discount":1,"currency":"USD"},"author":"anthropic","slug":"claude-opus-5","tags":["proprietary","coding","reasoning","agentic","long-context","vision"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}},{"id":"qwen3-30b-a3b-instruct","object":"model","canonical_slug":"alibaba/qwen3-30b-a3b-instruct","hugging_face_id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct","created":1767979257,"description":"Qwen3 30B A3B Instruct is an Alibaba/Qwen compact MoE instruction model for chat, code, tool use, and context-augmented generation.","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-07-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3-30b-a3b-instruct/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"},{"name":"Nebius","slug":"nebius"}],"pricing":{"prompt":"0.0000001","completion":"0.0000003","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"qwen3-30b-a3b-instruct","tags":["open-source","moe","long-context","coding","function-calling"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"nova-micro","object":"model","canonical_slug":"amazon/nova-micro","hugging_face_id":"","name":"Amazon Nova Micro","created":1769967814,"description":"Amazon Nova Micro is the fastest and most cost-effective Nova model, optimized for simple text tasks.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-12-03","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-micro/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.000000046","completion":"0.000000184","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000000115","input_cache_write":"0","discount":1,"currency":"USD"},"author":"amazon","slug":"nova-micro","tags":["proprietary","fast","cost-efficient"],"author_info":{"slug":"amazon","name":"Amazon","display_name":"Amazon Web Services","icon_url":"/images/logos/aws.webp","gradient_from":"from-orange-500","gradient_to":"to-yellow-500","gradient_via":null,"website_url":"https://aws.amazon.com/bedrock"}},{"id":"glm-5.2-caveman","object":"model","canonical_slug":"zhipu/glm-5.2-caveman","hugging_face_id":"zai-org/GLM-5.2","name":"GLM 5.2 Caveman","created":1787955393,"description":"GLM-5.2 served by GreenPT with a built-in prose-compression ruleset: the same upstream model and price per token as glm-5.2, with shorter answers. Code, commands, error strings, numbers and API names are kept verbatim. The injected ruleset is billed as input (~275 extra prompt tokens), so it only pays for itself above roughly 70 output tokens.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["parallel_tool_calls","reasoning_effort","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":null,"last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["none","minimal","low","medium","high"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.2-caveman/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000011","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"glm-5.2-caveman","tags":["long-context","coding","reasoning","function-calling","output-compression"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"glm-5.2-ponytail-lite","object":"model","canonical_slug":"zhipu/glm-5.2-ponytail-lite","hugging_face_id":"zai-org/GLM-5.2","name":"GLM 5.2 Ponytail Lite","created":1788029784,"description":"GLM-5.2 served by GreenPT with the ponytail-lite output-compression ruleset: a lazy-senior-developer bias for code-heavy answers (YAGNI first, stdlib and platform natives before custom code) at its gentlest intensity, building what you asked and naming the lazier alternative in one line. Same weights, context and price per token as glm-5.2.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","presence_penalty","reasoning_effort","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-08-04","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["none","minimal","low","medium","high"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.2-ponytail-lite/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000011","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"glm-5.2-ponytail-lite","tags":["long-context","coding","reasoning","output-compression"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"claude-sonnet-4-6","object":"model","canonical_slug":"anthropic/claude-sonnet-4-6","hugging_face_id":"","name":"Claude Sonnet 4.6","created":1773759863,"description":"Claude Sonnet 4.6 is Anthropic's balanced performance model for coding, reasoning, and agentic workflows with a 1M context window.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"knowledge_cutoff":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-sonnet-4-6/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000033","completion":"0.0000165","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000033","input_cache_write":"0.000004125","discount":1,"currency":"USD"},"author":"anthropic","slug":"claude-sonnet-4-6","tags":["proprietary","coding","reasoning","agentic","long-context","vision"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}},{"id":"green-r-raw","object":"model","canonical_slug":"greenpt/green-r-raw","hugging_face_id":"","name":"GreenR Raw","created":1768905404,"description":"Reasoning model with custom system prompt support. Combines advanced reasoning capabilities with customizable behavior. EU-hosted.","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-08-01","last_updated":null,"reasoning":{"mandatory":true,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/greenpt/green-r-raw/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.00000035","completion":"0.00000095","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"greenpt","slug":"green-r-raw","tags":["eu-native","sustainable","reasoning","raw"],"author_info":{"slug":"greenpt","name":"GreenPT","display_name":"GreenPT","icon_url":null,"gradient_from":"from-green-600","gradient_to":"to-green-700","gradient_via":null,"website_url":"https://greenpt.ai"}},{"id":"cohere-embed-v4","object":"model","canonical_slug":"cohere/cohere-embed-v4","hugging_face_id":"","name":"Cohere Embed v4","created":1769967814,"description":"Cohere Embed v4 is a state-of-the-art embedding model supporting 100+ languages with best-in-class retrieval performance.","context_length":512,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-10-02","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/cohere/cohere-embed-v4/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000012","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"cohere","slug":"cohere-embed-v4","tags":["proprietary","embedding","multilingual","retrieval"],"author_info":{"slug":"cohere","name":"Cohere","display_name":null,"icon_url":"/images/logos/cohere.webp","gradient_from":"from-purple-600","gradient_to":"to-purple-800","gradient_via":null,"website_url":"https://cohere.com"}},{"id":"qwen3-reranker-4b","object":"model","canonical_slug":"alibaba/qwen3-reranker-4b","hugging_face_id":"","name":"Qwen3 Reranker 4B","created":1784714465,"description":"Qwen3-Reranker-4B is a 4B-parameter multilingual cross-encoder reranker from the Qwen3 series. It scores query-document relevance for retrieval-augmented generation and search, with instruction-aware ranking across 100+ languages and a 32K context window.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":null,"last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3-reranker-4b/endpoints"},"supported_api_endpoints":["rerank"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"},{"name":"Regolo","slug":"regolo"}],"pricing":{"prompt":"0.00000012","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"alibaba","slug":"qwen3-reranker-4b","tags":["open-source","rerank","cross-encoder","multilingual"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"cohere-rerank-3.5","object":"model","canonical_slug":"cohere/cohere-rerank-3.5","hugging_face_id":"","name":"Cohere Rerank 3.5","created":1784714557,"description":"Cohere Rerank 3.5 is a multilingual cross-encoder reranker that reorders documents by semantic relevance to a query, improving search and retrieval-augmented generation. Available through Amazon Bedrock.","context_length":4096,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":4096,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":null,"last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/cohere/cohere-rerank-3.5/endpoints"},"supported_api_endpoints":["rerank"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"author":"cohere","slug":"cohere-rerank-3.5","tags":["proprietary","rerank","multilingual"],"author_info":{"slug":"cohere","name":"Cohere","display_name":null,"icon_url":"/images/logos/cohere.webp","gradient_from":"from-purple-600","gradient_to":"to-purple-800","gradient_via":null,"website_url":"https://cohere.com"}},{"id":"codestral-embed","object":"model","canonical_slug":"mistral/codestral-embed","hugging_face_id":"","name":"Codestral Embed","created":1767887445,"description":"Codestral Embed is Mistral AI's code embedding model for semantic code search and retrieval.","context_length":8192,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["dimensions","encoding_format","output_dimension","output_dtype"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-05-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/codestral-embed/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.000000165","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000000165","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"codestral-embed","tags":["embedding","code","retrieval","semantic-search"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"magistral-medium","object":"model","canonical_slug":"mistral/magistral-medium","hugging_face_id":"","name":"Magistral Medium 1.2","created":1767979257,"description":"Magistral Medium 1.2 is Mistral AI's premier reasoning model for domain-specific, transparent, and multilingual reasoning.","context_length":40000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":40000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-09-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/magistral-medium/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.0000022","completion":"0.0000055","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"magistral-medium","tags":["premier","reasoning","multilingual"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"claude-opus-4-5-20251101","object":"model","canonical_slug":"anthropic/claude-opus-4-5-20251101","hugging_face_id":"","name":"Claude Opus 4.5","created":1769967813,"description":"Claude Opus 4.5 is Anthropic's most capable model for complex tasks requiring deep analysis, nuanced understanding, and sophisticated reasoning.","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"claude","instruct_type":"claude"},"top_provider":{"context_length":200000,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-11-01","last_updated":"2025-11-01","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-opus-4-5-20251101/endpoints"},"supported_api_endpoints":["chat.completions","messages","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000055","completion":"0.0000275","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000055","input_cache_write":"0.000006875","discount":1,"currency":"USD"},"author":"anthropic","slug":"claude-opus-4-5-20251101","tags":["proprietary","long-context","vision","reasoning","premium"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}},{"id":"glm-5.2-ponytail-ultra","object":"model","canonical_slug":"zhipu/glm-5.2-ponytail-ultra","hugging_face_id":"zai-org/GLM-5.2","name":"GLM 5.2 Ponytail Ultra","created":1787955330,"description":"GLM-5.2 with GreenPT's \"ponytail\" output-compression ruleset at maximum intensity: a lazy-senior-developer bias for generated code (YAGNI, stdlib and platform natives first, no speculative abstractions) that prefers deletion over addition and pushes back on the requirement itself. Same upstream weights and same price per token as GLM-5.2, with a shorter answer.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","presence_penalty","reasoning_effort","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":null,"last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["minimal","low","medium","high"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.2-ponytail-ultra/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000011","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"glm-5.2-ponytail-ultra","tags":["long-context","coding","reasoning","compression"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"mistral-medium-3.1","object":"model","canonical_slug":"mistral/mistral-medium-3.1","hugging_face_id":"","name":"Mistral Medium 3.1","created":1768905491,"description":"Mistral Medium 3.1 is Mistral AI's frontier-class multimodal model released in August 2025.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-08-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-medium-3.1/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.00000165","completion":"0.00000825","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000165","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"mistral-medium-3.1","tags":["premier","multimodal","vision","function-calling"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"intellect-3","object":"model","canonical_slug":"primeintellect/intellect-3","hugging_face_id":"PrimeIntellect/INTELLECT-3","name":"INTELLECT-3","created":1768127608,"description":"Prime Intellect INTELLECT-3 is an open model for reasoning, code, tool-use, and context-augmented chat workloads.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-11-26","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/primeintellect/intellect-3/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebius","slug":"nebius"}],"pricing":{"prompt":"0.0000002","completion":"0.0000011","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"primeintellect","slug":"intellect-3","tags":["open-source","reasoning","coding","function-calling"],"author_info":{"slug":"primeintellect","name":"Prime Intellect","display_name":"Prime Intellect","icon_url":null,"gradient_from":"from-emerald-500","gradient_to":"to-teal-600","gradient_via":null,"website_url":"https://primeintellect.ai"}},{"id":"glm-5.2-honey","object":"model","canonical_slug":"zhipu/glm-5.2-honey","hugging_face_id":"zai-org/GLM-5.2","name":"GLM 5.2 Honey","created":1788031033,"description":"GLM-5.2 served by GreenPT with its \"honey\" output-compression ruleset injected as a system prompt: minimal code generation plus concise prose, at the same upstream weights, context and price per token as glm-5.2. The injected ruleset is billed as input (~335 extra prompt tokens), so it only pays for itself above roughly 85 output tokens.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","presence_penalty","reasoning_effort","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":null,"last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["none","minimal","low","medium","high"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.2-honey/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000011","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"glm-5.2-honey","tags":["long-context","coding","reasoning","output-compression"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"qwen3.6-35b","object":"model","canonical_slug":"alibaba/qwen3.6-35b","hugging_face_id":"Qwen/Qwen3.6-35B","name":"Qwen3.6 35B","created":1782399397,"description":"Qwen3.6 35B is Alibaba's Qwen3.6-series model for multilingual chat, reasoning, and tool use.","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","stop","stream","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-04-16","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3.6-35b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Scaleway","slug":"scaleway"},{"name":"AKI.IO","slug":"aki"},{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.00000015","completion":"0.0000005","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"alibaba","slug":"qwen3.6-35b","tags":["open-source","multilingual","long-context","function-calling"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"deepseek-v4-pro","object":"model","canonical_slug":"deepseek/deepseek-v4-pro","hugging_face_id":"","name":"DeepSeek V4 Pro","created":1777625927,"description":"DeepSeek V4 Pro is a DeepSeek V4-series text model with a 1M-token context window, function calling, and optional reasoning support.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-04-24","last_updated":"2026-04-24","reasoning":{"mandatory":false,"supported_efforts":["high","max"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-pro/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.00000175","completion":"0.0000035","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000004375","input_cache_write":"0","discount":1,"currency":"USD"},"author":"deepseek","slug":"deepseek-v4-pro","tags":["long-context","function-calling","reasoning"],"author_info":{"slug":"deepseek","name":"DeepSeek","display_name":null,"icon_url":"/images/logos/deepseek.png","gradient_from":"from-[#0066ff]","gradient_to":"to-[#00ccff]","gradient_via":null,"website_url":"https://deepseek.com"}},{"id":"faster-whisper-large-v3","object":"model","canonical_slug":"systran/faster-whisper-large-v3","hugging_face_id":"Systran/faster-whisper-large-v3","name":"Faster Whisper Large V3","created":1777454224,"description":"Faster Whisper Large V3 is Systran's CTranslate2-optimized Whisper Large V3 transcription model.","context_length":1,"architecture":{"modality":"audio->text","input_modalities":["audio"],"output_modalities":["text"],"tokenizer":"Whisper","instruct_type":null},"top_provider":{"context_length":1,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["language","response_format"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-11-23","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/systran/faster-whisper-large-v3/endpoints"},"supported_api_endpoints":["audio.transcriptions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Regolo","slug":"regolo"}],"author":"systran","slug":"faster-whisper-large-v3","tags":["audio","transcription","speech-to-text"],"author_info":{"slug":"systran","name":"Systran","display_name":null,"icon_url":null,"gradient_from":null,"gradient_to":null,"gradient_via":null,"website_url":"https://www.systransoft.com"}},{"id":"nova-2-lite","object":"model","canonical_slug":"amazon/nova-2-lite","hugging_face_id":"","name":"Amazon Nova 2 Lite","created":1777454590,"description":"Amazon Nova 2 Lite is a cost-efficient multimodal model for simple automation, document processing, and customer support across text, images, and video.","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-12-02","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-2-lite/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.000000429","completion":"0.000003597","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000010725","input_cache_write":"0","discount":1,"currency":"USD"},"author":"amazon","slug":"nova-2-lite","tags":["proprietary","multimodal","vision","video","long-context","prompt-caching"],"author_info":{"slug":"amazon","name":"Amazon","display_name":"Amazon Web Services","icon_url":"/images/logos/aws.webp","gradient_from":"from-orange-500","gradient_to":"to-yellow-500","gradient_via":null,"website_url":"https://aws.amazon.com/bedrock"}},{"id":"gpt-4o","object":"model","canonical_slug":"openai/gpt-4o","hugging_face_id":"","name":"GPT-4o","created":1768138176,"description":"GPT-4o is OpenAI's multimodal model for text and image input, text output, tool use, and general-purpose chat workloads.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-11-20","last_updated":"2024-08-06","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.00000275","completion":"0.000011","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000001375","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-4o","tags":["multimodal","function-calling","vision","general-purpose"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"pixtral-large-2502","object":"model","canonical_slug":"mistral/pixtral-large-2502","hugging_face_id":"","name":"Pixtral Large 25.02","created":1769967814,"description":"Pixtral Large is Mistral's flagship multimodal model with exceptional vision and text understanding capabilities.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"mistral","instruct_type":"mistral"},"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-02-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/pixtral-large-2502/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.000002","completion":"0.000006","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"pixtral-large-2502","tags":["proprietary","vision","multimodal","long-context"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"gpt-5.4","object":"model","canonical_slug":"openai/gpt-5.4","hugging_face_id":"","name":"GPT-5.4","created":1777454744,"description":"GPT-5.4 is OpenAI's frontier reasoning model for complex professional work, long-context analysis, coding, and agentic automation.","context_length":1050000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","reasoning_effort","response_format","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.4/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.00000275","completion":"0.0000165","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-5.4","tags":["reasoning","long-context","frontier","agentic","coding","computer-use","vision"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"titan-embed-text-v1","object":"model","canonical_slug":"amazon/titan-embed-text-v1","hugging_face_id":"","name":"Amazon Titan Embeddings G1 - Text","created":1777454590,"description":"Amazon Titan Embeddings G1 - Text is a text embedding model available through Amazon Bedrock.","context_length":8000,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":8000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-09-28","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/amazon/titan-embed-text-v1/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000002","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"amazon","slug":"titan-embed-text-v1","tags":["proprietary","embedding"],"author_info":{"slug":"amazon","name":"Amazon","display_name":"Amazon Web Services","icon_url":"/images/logos/aws.webp","gradient_from":"from-orange-500","gradient_to":"to-yellow-500","gradient_via":null,"website_url":"https://aws.amazon.com/bedrock"}},{"id":"paraphrase-multilingual-mpnet-base-v2","object":"model","canonical_slug":"sentence-transformers/paraphrase-multilingual-mpnet-base-v2","hugging_face_id":"sentence-transformers/paraphrase-multilingual-mpnet-base-v2","name":"Paraphrase Multilingual MPNet v2","created":1770318640,"description":"Paraphrase Multilingual MPNet v2 is a sentence-transformer embedding model producing 768-dimensional vectors for multilingual semantic search and similarity.","context_length":128,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":128,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["encoding_format"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2021-07-23","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/sentence-transformers/paraphrase-multilingual-mpnet-base-v2/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"IONOS Cloud","slug":"ionos"}],"pricing":{"prompt":"0.00000001","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"sentence-transformers","slug":"paraphrase-multilingual-mpnet-base-v2","tags":["embedding","multilingual"],"author_info":{"slug":"sentence-transformers","name":"Sentence Transformers","display_name":"Sentence Transformers","icon_url":null,"gradient_from":null,"gradient_to":null,"gradient_via":null,"website_url":"https://www.sbert.net"}},{"id":"claude-opus-4-7","object":"model","canonical_slug":"anthropic/claude-opus-4-7","hugging_face_id":"","name":"Claude Opus 4.7","created":1777454590,"description":"Claude Opus 4.7 is Anthropic's frontier model for complex coding, reasoning, and agentic workflows.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","tool_choice","tools","top_k","top_p"],"supported_voices":null,"knowledge_cutoff":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-opus-4-7/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000055","completion":"0.0000275","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000055","input_cache_write":"0.000006875","discount":1,"currency":"USD"},"author":"anthropic","slug":"claude-opus-4-7","tags":["proprietary","coding","reasoning","agentic","vision"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}},{"id":"alibaba/qwen3-vl-embedding-8b","object":"model","canonical_slug":"alibaba/qwen3-vl-embedding-8b","hugging_face_id":"Qwen/Qwen3-VL-Embedding-8B","name":"Qwen3 VL Embedding 8B","created":1788171916,"description":"Qwen3 VL Embedding 8B is Alibaba's multimodal embedding model for mapping text, images, document images, and video into a shared vector space for multilingual retrieval and clustering.","context_length":32768,"architecture":{"modality":"text+image+video->embedding","input_modalities":["text","image","video"],"output_modalities":["embedding"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-01-08","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/alibaba/qwen3-vl-embedding-8b/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"IONOS Cloud","slug":"ionos"}],"pricing":{"prompt":"0.00000011","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"alibaba/qwen3-vl-embedding-8b","tags":["open-source","embedding","multimodal","vision","video","multilingual","retrieval"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"bge-large-en-v1.5","object":"model","canonical_slug":"baai/bge-large-en-v1.5","hugging_face_id":"BAAI/bge-large-en-v1.5","name":"BGE Large EN v1.5","created":1770318640,"description":"BGE Large EN v1.5 is an English embedding model with 1024-dimensional vectors, optimized for semantic search and retrieval with long-context inputs.","context_length":8192,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["encoding_format"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-01-28","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/baai/bge-large-en-v1.5/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"IONOS Cloud","slug":"ionos"}],"pricing":{"prompt":"0.000000015","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"baai","slug":"bge-large-en-v1.5","tags":["embedding","english","long-context"],"author_info":{"slug":"baai","name":"BAAI","display_name":"Beijing Academy of AI","icon_url":null,"gradient_from":"from-red-600","gradient_to":"to-red-800","gradient_via":null,"website_url":"https://www.baai.ac.cn"}},{"id":"openai/whisper-large-v3","object":"model","canonical_slug":"openai/whisper-large-v3","hugging_face_id":"openai/whisper-large-v3","name":"Whisper Large v3","created":1788117780,"description":"OpenAI Whisper Large v3 is an open-weight multilingual encoder-decoder model for automatic speech recognition and audio transcription.","context_length":448,"architecture":{"modality":"audio->text","input_modalities":["audio"],"output_modalities":["text"],"tokenizer":"Whisper","instruct_type":null},"top_provider":{"context_length":448,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["language","prompt","response_format","stream","temperature"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-11-06","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/openai/whisper-large-v3/endpoints"},"supported_api_endpoints":["audio.transcriptions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Scaleway","slug":"scaleway"}],"author":"openai","slug":"openai/whisper-large-v3","tags":[],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"mistral-medium-3.5","object":"model","canonical_slug":"mistral/mistral-medium-3.5","hugging_face_id":"mistralai/Mistral-Medium-3.5-128B","name":"Mistral Medium 3.5","created":1788032047,"description":"Mistral Medium 3.5 is Mistral AI's first flagship merged model: a dense 128B multimodal model that handles instruction following, configurable reasoning, and coding in a single set of weights.","context_length":180000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":180000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","seed","stop","stream","stream_options","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-04-28","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["none","high"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-medium-3.5/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Scaleway","slug":"scaleway"},{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000015","completion":"0.0000075","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"mistral","slug":"mistral-medium-3.5","tags":["open-weights","multimodal","vision","reasoning","coding","function-calling","agentic","long-context"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"qwen3.5-397b-a17b","object":"model","canonical_slug":"alibaba/qwen3.5-397b-a17b","hugging_face_id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B A17B","created":1777452602,"description":"Qwen3.5 397B A17B is an Alibaba/Qwen MoE model for long-context reasoning, coding, agentic tasks, tool use, and multimodal understanding.","context_length":250000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":250000,"max_completion_tokens":16000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-02-15","last_updated":"2026-02-15","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3.5-397b-a17b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"},{"name":"Scaleway","slug":"scaleway"},{"name":"GreenPT","slug":"greenpt"},{"name":"IONOS Cloud","slug":"ionos"}],"pricing":{"prompt":"0.0000006","completion":"0.0000036","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"alibaba","slug":"qwen3.5-397b-a17b","tags":["open-source","moe","coding","agentic","reasoning","vision","function-calling"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"text-embedding-3-large","object":"model","canonical_slug":"openai/text-embedding-3-large","hugging_face_id":"","name":"Text Embedding 3 Large","created":1767964798,"description":"text-embedding-3-large is OpenAI's most capable text embedding model with up to 3,072 output dimensions.","context_length":8192,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":"cl100k_base","instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["dimensions","encoding_format"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-01-25","last_updated":"2024-01-25","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/text-embedding-3-large/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"author":"openai","slug":"text-embedding-3-large","tags":["embedding","retrieval","high-quality"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"deepseek-ocr-2","object":"model","canonical_slug":"deepseek/deepseek-ocr-2","hugging_face_id":"deepseek-ai/DeepSeek-OCR-2","name":"DeepSeek-OCR 2","created":1788029792,"description":"DeepSeek-OCR 2 is DeepSeek's open-weight 3B vision-language OCR model. Its DeepEncoder V2 encoder reorders visual tokens (visual causal flow) rather than raster-scanning, converting documents, screenshots, receipts and scene text into text or layout-aware Markdown, including tables, formulas and reading order.","context_length":8000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"top_provider":{"context_length":8000,"max_completion_tokens":4000,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-01-27","last_updated":"2026-02-03","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-ocr-2/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Regolo","slug":"regolo"}],"pricing":{"prompt":"0","completion":"0","request":"0.02","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"deepseek","slug":"deepseek-ocr-2","tags":["open-source","vision","document-analysis","multilingual","moe"],"author_info":{"slug":"deepseek","name":"DeepSeek","display_name":null,"icon_url":"/images/logos/deepseek.png","gradient_from":"from-[#0066ff]","gradient_to":"to-[#00ccff]","gradient_via":null,"website_url":"https://deepseek.com"}},{"id":"glm-5.1","object":"model","canonical_slug":"zhipu/glm-5.1","hugging_face_id":"zai-org/GLM-5.1-FP8","name":"GLM 5.1","created":1777453828,"description":"GLM-5.1 is Z AI's flagship model for long-horizon agentic engineering and advanced coding workflows.","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"TokenizersBackend","instruct_type":null},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-04-07","last_updated":"2026-04-07","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.1/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000014","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000035","input_cache_write":"0","discount":1,"currency":"USD"},"author":"zhipu","slug":"glm-5.1","tags":["open-source","long-context","coding","agentic","reasoning"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"bge-multilingual-gemma2","object":"model","canonical_slug":"baai/bge-multilingual-gemma2","hugging_face_id":"BAAI/bge-multilingual-gemma2","name":"BGE Multilingual Gemma2","created":1767963057,"description":"BGE Multilingual Gemma2 is a BAAI multilingual embedding model based on Gemma2 for dense retrieval and semantic search.","context_length":8192,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":"Gemma","instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["dimensions","encoding_format"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-06-29","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/baai/bge-multilingual-gemma2/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"OVHcloud","slug":"ovhcloud"},{"name":"Scaleway","slug":"scaleway"},{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.00000001","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"baai","slug":"bge-multilingual-gemma2","tags":["embedding","multilingual","retrieval"],"author_info":{"slug":"baai","name":"BAAI","display_name":"Beijing Academy of AI","icon_url":null,"gradient_from":"from-red-600","gradient_to":"to-red-800","gradient_via":null,"website_url":"https://www.baai.ac.cn"}},{"id":"anthropic/claude-opus-5-5","object":"model","canonical_slug":"anthropic/claude-opus-5-5","hugging_face_id":"","name":"Claude Opus 5.5","created":1790267644,"description":"Claude Opus 5.5 is Anthropic's multimodal, long-context model for advanced coding, reasoning, knowledge work, and long-running agentic tasks, with always-on adaptive thinking.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["parallel_tool_calls","reasoning_effort","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-09-22","last_updated":"2026-09-22","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/anthropic/claude-opus-5-5/endpoints"},"supported_api_endpoints":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000044","completion":"0.000022","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000022","input_cache_write":"0.0000055","discount":1,"currency":"USD"},"author":"anthropic","slug":"anthropic/claude-opus-5-5","tags":["proprietary","coding","reasoning","agentic","long-context","vision"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}},{"id":"qwen3.5-122b-a10b","object":"model","canonical_slug":"alibaba/qwen3.5-122b-a10b","hugging_face_id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B A10B","created":1777454224,"description":"Qwen3.5 122B A10B is Alibaba's natively multimodal Mixture-of-Experts model for reasoning, coding, multilingual chat, vision-language understanding, and tool use.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-02-23","last_updated":"2026-02-23","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":true},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3.5-122b-a10b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Regolo","slug":"regolo"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000005","completion":"0.0000035","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000125","input_cache_write":"0","discount":1,"currency":"USD"},"author":"alibaba","slug":"qwen3.5-122b-a10b","tags":["open-source","moe","long-context","coding","reasoning","vision"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"mixtral-8x7b-instruct","object":"model","canonical_slug":"mistral/mixtral-8x7b-instruct","hugging_face_id":"","name":"Mixtral 8x7B Instruct","created":1777454590,"description":"Mixtral 8x7B Instruct is Mistral AI's sparse mixture-of-experts instruction model.","context_length":32000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-12-11","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mixtral-8x7b-instruct/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000049","completion":"0.00000076","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"mixtral-8x7b-instruct","tags":["open-weights","instruction-tuned","moe"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"alibaba/qwen3.6-35b-instruct","object":"model","canonical_slug":"alibaba/qwen3.6-35b-instruct","hugging_face_id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B A3B Instruct","created":1788118233,"description":"","context_length":256000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":null,"last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/alibaba/qwen3.6-35b-instruct/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AKI.IO","slug":"aki"}],"pricing":{"prompt":"0.00000015","completion":"0.00000050","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"alibaba","slug":"alibaba/qwen3.6-35b-instruct","tags":[],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"o4-mini","object":"model","canonical_slug":"openai/o4-mini","hugging_face_id":"","name":"o4-mini","created":1767619877,"description":"o4-mini is OpenAI's compact reasoning model for math, coding, visual tasks, structured outputs, and tool use.","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","reasoning_effort","response_format","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-04-16","last_updated":"2025-04-16","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/o4-mini/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.00000121","completion":"0.00000484","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000303","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"o4-mini","tags":["reasoning","long-context","efficient","coding","math","vision"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"all-minilm-l6-v2","object":"model","canonical_slug":"sentence-transformers/all-minilm-l6-v2","hugging_face_id":"sentence-transformers/all-MiniLM-L6-v2","name":"All MiniLM L6 v2","created":1783347345,"description":"all-MiniLM-L6-v2 is a compact sentence-transformers embedding model for fast English sentence and paragraph retrieval.","context_length":256,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":256,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2021-08-30","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/sentence-transformers/all-minilm-l6-v2/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"}],"pricing":{"prompt":"0.00000009","completion":"0.00000045","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"sentence-transformers","slug":"all-minilm-l6-v2","tags":["open-source","embedding","retrieval"],"author_info":{"slug":"sentence-transformers","name":"Sentence Transformers","display_name":"Sentence Transformers","icon_url":null,"gradient_from":null,"gradient_to":null,"gradient_via":null,"website_url":"https://www.sbert.net"}},{"id":"mistral-large-3","object":"model","canonical_slug":"mistral/mistral-large-3","hugging_face_id":"","name":"Mistral Large 3","created":1767979257,"description":"Mistral Large 3 is Mistral AI's 675B-parameter model for coding, reasoning, multilingual tasks, and vision-enabled chat.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-12-02","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-large-3/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"},{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.00000055","completion":"0.00000165","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000055","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"mistral-large-3","tags":["flagship","instruction-tuned","vision","long-context","coding","agentic","prompt-caching"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"apertus-70b","object":"model","canonical_slug":"swiss-ai/apertus-70b","hugging_face_id":"swiss-ai/Apertus-70B-Instruct-2509","name":"Apertus 70B Instruct","created":1782399397,"description":"Apertus 70B is a fully open multilingual model from the Swiss AI Initiative (ETH Zurich, EPFL, CSCS).","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":65536,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-09-02","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/swiss-ai/apertus-70b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AKI.IO","slug":"aki"},{"name":"Regolo","slug":"regolo"}],"pricing":{"prompt":"0.0000004","completion":"0.0000021","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"swiss-ai","slug":"apertus-70b","tags":["open-source","multilingual","instruction-tuned"],"author_info":{"slug":"swiss-ai","name":"Swiss AI Initiative","display_name":"Swiss AI Initiative","icon_url":null,"gradient_from":null,"gradient_to":null,"gradient_via":null,"website_url":"https://www.swiss-ai.org"}},{"id":"deepseek/deepseek-v4.1-flash","object":"model","canonical_slug":"deepseek/deepseek-v4.1-flash","hugging_face_id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","created":1790250636,"description":"DeepSeek V4.1 Flash is an open-weight multimodal Mixture-of-Experts reasoning model with native text and image input, text output, a one-million-token context window, and a focus on agentic and coding workloads.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","reasoning_effort","stream","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-09-10","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["none","minimal","low","medium","high"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek/deepseek-v4.1-flash/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"},{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.00000022","completion":"0.00000110","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000011","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"deepseek","slug":"deepseek/deepseek-v4.1-flash","tags":["open-source","multimodal","vision","reasoning","coding","agentic","long-context","moe"],"author_info":{"slug":"deepseek","name":"DeepSeek","display_name":null,"icon_url":"/images/logos/deepseek.png","gradient_from":"from-[#0066ff]","gradient_to":"to-[#00ccff]","gradient_via":null,"website_url":"https://deepseek.com"}},{"id":"gpt-5-nano","object":"model","canonical_slug":"openai/gpt-5-nano","hugging_face_id":"","name":"GPT-5 Nano","created":1777454744,"description":"GPT-5 Nano is OpenAI's smallest GPT-5 reasoning model for high-volume multimodal and tool-assisted workloads.","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","reasoning_effort","response_format","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5-nano/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.000000055","completion":"0.00000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000000055","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-5-nano","tags":["reasoning","long-context","efficient","high-throughput","vision"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"qwen3.8-27b","object":"model","canonical_slug":"alibaba/qwen3.8-27b","hugging_face_id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","created":1788034833,"description":"Qwen3.8 27B is Alibaba's dense native vision-language model of the Qwen3.8 generation, for coding, professional work, research and long-horizon agentic tasks. It understands images and video, with flexible thinking controls (reasoning effort low/medium/xhigh) that tune reasoning depth and preserve thinking across turns.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"chatml"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-08-14","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["low","medium","xhigh"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3.8-27b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AKI.IO","slug":"aki"}],"pricing":{"prompt":"0.0000003","completion":"0.0000022","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000001","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"alibaba","slug":"qwen3.8-27b","tags":["open-source","multimodal","vision","video","reasoning","coding","agentic","long-context","multilingual","function-calling"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"titan-embed-text-v2","object":"model","canonical_slug":"amazon/titan-embed-text-v2","hugging_face_id":"","name":"Amazon Titan Embeddings v2","created":1769967814,"description":"Amazon Titan Embeddings v2 offers configurable embedding dimensions (256-1024) and improved multilingual support.","context_length":8000,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":8000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-04-30","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/amazon/titan-embed-text-v2/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000002","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"amazon","slug":"titan-embed-text-v2","tags":["proprietary","embedding","multilingual","configurable-dimensions"],"author_info":{"slug":"amazon","name":"Amazon","display_name":"Amazon Web Services","icon_url":"/images/logos/aws.webp","gradient_from":"from-orange-500","gradient_to":"to-yellow-500","gradient_via":null,"website_url":"https://aws.amazon.com/bedrock"}},{"id":"claude-sonnet-4-5","object":"model","canonical_slug":"anthropic/claude-sonnet-4-5","hugging_face_id":"","name":"Claude Sonnet 4.5","created":1769967813,"description":"Claude Sonnet 4.5 is Anthropic's most advanced model, offering exceptional performance on complex reasoning tasks with a 200K context window.","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"claude","instruct_type":"claude"},"top_provider":{"context_length":200000,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-sonnet-4-5/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000033","completion":"0.0000165","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000033","input_cache_write":"0.000004125","discount":1,"currency":"USD"},"author":"anthropic","slug":"claude-sonnet-4-5","tags":["proprietary","long-context","vision","reasoning"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}},{"id":"deepseek-v3.2","object":"model","canonical_slug":"deepseek/deepseek-v3.2","hugging_face_id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek V3.2","created":1774895903,"description":"DeepSeek V3.2 is a mixture-of-experts model with improved reasoning, coding, and instruction-following capabilities.","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"top_provider":{"context_length":163840,"max_completion_tokens":163840,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-12-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v3.2/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000003","completion":"0.0000005","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000075","input_cache_write":"0","discount":1,"currency":"USD"},"author":"deepseek","slug":"deepseek-v3.2","tags":["open-source","long-context","reasoning","coding"],"author_info":{"slug":"deepseek","name":"DeepSeek","display_name":null,"icon_url":"/images/logos/deepseek.png","gradient_from":"from-[#0066ff]","gradient_to":"to-[#00ccff]","gradient_via":null,"website_url":"https://deepseek.com"}},{"id":"mistral-nemo-12b","object":"model","canonical_slug":"mistral/mistral-nemo-12b","hugging_face_id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo 12B","created":1765024302,"description":"Mistral Nemo 12B is Mistral AI's open multilingual model built with NVIDIA for efficient long-context text generation.","context_length":120832,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":120832,"max_completion_tokens":120832,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-07-17","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-nemo-12b/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"OVHcloud","slug":"ovhcloud"},{"name":"IONOS Cloud","slug":"ionos"}],"pricing":{"prompt":"0.00000013","completion":"0.00000013","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"mistral","slug":"mistral-nemo-12b","tags":["open-source","multilingual","long-context"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"mistral-embed","object":"model","canonical_slug":"mistral/mistral-embed","hugging_face_id":"","name":"Mistral Embed","created":1767964798,"description":"Mistral Embed is Mistral AI's text embedding model for semantic search, retrieval, clustering, and classification.","context_length":8192,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["dimensions","encoding_format","output_dimension","output_dtype"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-12-11","last_updated":"2023-12-11","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-embed/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.00000011","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"mistral-embed","tags":["embedding","retrieval","semantic-search"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"nemotron-3-nano-omni","object":"model","canonical_slug":"nvidia/nemotron-3-nano-omni","hugging_face_id":"","name":"Nemotron 3 Nano Omni","created":1777453939,"description":"NVIDIA Nemotron 3 Nano Omni is an efficient omni-modal reasoning model for agentic AI, coding, tool use, and long-context workflows.","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nemotron","instruct_type":null},"top_provider":{"context_length":65536,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-04-24","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-nano-omni/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"}],"pricing":{"prompt":"0.00000016","completion":"0.00000105","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000005","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"nvidia","slug":"nemotron-3-nano-omni","tags":["open-source","reasoning","coding","function-calling","json-mode"],"author_info":{"slug":"nvidia","name":"NVIDIA","display_name":null,"icon_url":"/images/logos/nvidia.webp","gradient_from":"from-[#76b900]","gradient_to":"to-[#5a8c00]","gradient_via":null,"website_url":"https://nvidia.com"}},{"id":"text-embedding-ada-002","object":"model","canonical_slug":"openai/text-embedding-ada-002","hugging_face_id":"","name":"Text Embedding Ada 002","created":1767964798,"description":"text-embedding-ada-002 is OpenAI's older general-purpose text embedding model with 1,536 output dimensions.","context_length":8192,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":"cl100k_base","instruct_type":null},"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["dimensions","encoding_format"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2022-12-15","last_updated":"2022-12-15","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/text-embedding-ada-002/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"author":"openai","slug":"text-embedding-ada-002","tags":["embedding","retrieval","legacy"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"kimi-k3","object":"model","canonical_slug":"moonshot/kimi-k3","hugging_face_id":"moonshotai/Kimi-K3","name":"Kimi K3","created":1785184294,"description":"Moonshot AI's Kimi K3 frontier open-weights MoE model (MXFP4, 1M context) with MTP speculative decoding, strong agentic tool use, reasoning, and coding.","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Kimi","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","stop","stream","stream_options","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-07-27","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/moonshot/kimi-k3/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebius","slug":"nebius"},{"name":"Tensorix","slug":"tensorix"},{"name":"GreenPT","slug":"greenpt"},{"name":"Lyceum Technology GmbH","slug":"lyceum"}],"pricing":{"prompt":"0.000003","completion":"0.000015","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"moonshot","slug":"kimi-k3","tags":["open-source","moe","long-context","vision","coding","reasoning","function-calling"],"author_info":{"slug":"moonshot","name":"Moonshot","display_name":"Moonshot AI","icon_url":null,"gradient_from":"from-indigo-500","gradient_to":"to-purple-600","gradient_via":null,"website_url":"https://moonshot.ai"}},{"id":"glm-5","object":"model","canonical_slug":"zhipu/glm-5","hugging_face_id":"THUDM/GLM-5","name":"GLM 5","created":1774895903,"description":"GLM-5 is Z AI's flagship language model for coding, reasoning, tool use, and multilingual chat.","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":202752,"max_completion_tokens":202752,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-02-12","last_updated":"2026-02-12","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.000001","completion":"0.0000032","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000025","input_cache_write":"0","discount":1,"currency":"USD"},"author":"zhipu","slug":"glm-5","tags":["open-source","multilingual","long-context","coding","reasoning"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"gpt-oss-20b","object":"model","canonical_slug":"openai/gpt-oss-20b","hugging_face_id":"openai/gpt-oss-20b","name":"GPT-OSS 20B","created":1767979257,"description":"GPT-OSS 20B is OpenAI's smaller open-weight reasoning model for efficient chat, coding, and agentic workloads.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"o200k_base","instruct_type":null},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-08-05","last_updated":"2025-08-05","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-20b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Regolo","slug":"regolo"},{"name":"OVHcloud","slug":"ovhcloud"},{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000004","completion":"0.00000015","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"openai","slug":"gpt-oss-20b","tags":["open-source","reasoning","coding","function-calling"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"llama-3.3-70b-instruct","object":"model","canonical_slug":"meta/llama-3.3-70b-instruct","hugging_face_id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","created":1765024302,"description":"Llama 3.3 70B Instruct is Meta's multilingual instruction-tuned model for chat, reasoning, and tool-use workloads.","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"top_provider":{"context_length":131000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-12-06","last_updated":"2024-12-06","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/meta/llama-3.3-70b-instruct/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Scaleway","slug":"scaleway"},{"name":"OVHcloud","slug":"ovhcloud"},{"name":"AKI.IO","slug":"aki"},{"name":"IONOS Cloud","slug":"ionos"},{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.00000065","completion":"0.00000065","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"meta","slug":"llama-3.3-70b-instruct","tags":["open-source","instruction-tuned","function-calling"],"author_info":{"slug":"meta","name":"Meta","display_name":"Meta AI","icon_url":"/images/logos/meta.webp","gradient_from":"from-blue-600","gradient_to":"to-blue-800","gradient_via":null,"website_url":"https://ai.meta.com"}},{"id":"gpt-4o-mini","object":"model","canonical_slug":"openai/gpt-4o-mini","hugging_face_id":"","name":"GPT-4o Mini","created":1768138176,"description":"GPT-4o Mini is OpenAI's compact multimodal model for low-cost chat, vision, tool use, and structured-output workloads.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-07-18","last_updated":"2024-07-18","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-mini/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.000000165","completion":"0.00000066","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000083","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-4o-mini","tags":["multimodal","function-calling","vision","efficient"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"ministral-3-8b","object":"model","canonical_slug":"mistral/ministral-3-8b","hugging_face_id":"","name":"Ministral 3 8B","created":1767979257,"description":"Ministral 3 8B is an open Mistral AI model for efficient text and vision workloads.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Tekken","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","random_seed","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-12-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/ministral-3-8b/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"pricing":{"prompt":"0.000000165","completion":"0.000000165","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000000165","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"ministral-3-8b","tags":["open-source","vision","lightweight","long-context"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"qwen3-coder-30b-a3b","object":"model","canonical_slug":"alibaba/qwen3-coder-30b-a3b","hugging_face_id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3 Coder 30B A3B","created":1765183068,"description":"Qwen3 Coder 30B A3B is Alibaba's Mixture-of-Experts coding model for software engineering and agentic code tasks.","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-07-01","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/qwen3-coder-30b-a3b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Scaleway","slug":"scaleway"},{"name":"OVHcloud","slug":"ovhcloud"},{"name":"AWS Bedrock","slug":"aws-bedrock"},{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.00000006","completion":"0.00000022","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"alibaba","slug":"qwen3-coder-30b-a3b","tags":["open-source","moe","coding","agentic"],"author_info":{"slug":"alibaba","name":"Alibaba","display_name":"Alibaba (Qwen)","icon_url":"/images/logos/qwen.webp","gradient_from":"from-purple-600","gradient_to":"to-indigo-700","gradient_via":null,"website_url":"https://qwenlm.github.io"}},{"id":"mistral-large-2402","object":"model","canonical_slug":"mistral/mistral-large-2402","hugging_face_id":"","name":"Mistral Large 2402","created":1777454590,"description":"Mistral Large 2402 is Mistral AI's high-capability text model available through Amazon Bedrock.","context_length":32000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2024-02-26","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral-large-2402/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000043","completion":"0.000013","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"mistral","slug":"mistral-large-2402","tags":["proprietary","instruction-tuned","multilingual"],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"gpt-oss-120b","object":"model","canonical_slug":"openai/gpt-oss-120b","hugging_face_id":"openai/gpt-oss-120b","name":"GPT-OSS 120B","created":1765183068,"description":"GPT-OSS 120B is OpenAI's open-weight reasoning model for agentic, coding, and general chat workloads.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"o200k_base","instruct_type":null},"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-08-05","last_updated":"2025-08-05","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-120b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AKI.IO","slug":"aki"},{"name":"OVHcloud","slug":"ovhcloud"},{"name":"Scaleway","slug":"scaleway"},{"name":"Regolo","slug":"regolo"},{"name":"AWS Bedrock","slug":"aws-bedrock"},{"name":"IONOS Cloud","slug":"ionos"},{"name":"Nebius","slug":"nebius"},{"name":"GreenPT","slug":"greenpt"},{"name":"Infercom","slug":"infercom"},{"name":"Nebul","slug":"nebul"}],"pricing":{"prompt":"0.00000008","completion":"0.0000004","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"openai","slug":"gpt-oss-120b","tags":["open-source","reasoning","coding","function-calling"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"glm-4.7","object":"model","canonical_slug":"zhipu/glm-4.7","hugging_face_id":"THUDM/GLM-4.7","name":"GLM 4.7","created":1774895903,"description":"GLM-4.7 is Z AI's model for high-throughput chat, coding, and function-calling workloads.","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":202752,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-12-22","last_updated":"2025-12-22","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-4.7/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.00000072","completion":"0.00000264","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"zhipu","slug":"glm-4.7","tags":["open-source","multilingual","long-context","function-calling","reasoning"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"glm-5.2-caveman-ultra","object":"model","canonical_slug":"zhipu/glm-5.2-caveman-ultra","hugging_face_id":"zai-org/GLM-5.2","name":"GLM 5.2 Caveman Ultra","created":1787955392,"description":"GLM-5.2 served by GreenPT with its most aggressive caveman output-compression ruleset injected as a system prompt: the same upstream weights, context and price per token as glm-5.2, answering near answer-only. Caveman compresses prose style only - code, commands, error strings, numbers and API names are kept verbatim.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","stop","stream","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-06-13","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":["none","minimal","low","medium","high"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.2-caveman-ultra/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"GreenPT","slug":"greenpt"}],"pricing":{"prompt":"0.0000011","completion":"0.0000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000275","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"zhipu","slug":"glm-5.2-caveman-ultra","tags":["long-context","coding","reasoning","function-calling"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"gpt-4.1-nano","object":"model","canonical_slug":"openai/gpt-4.1-nano","hugging_face_id":"","name":"GPT-4.1 Nano","created":1769687270,"description":"GPT-4.1 Nano is OpenAI's lowest-latency GPT-4.1 model for high-volume multimodal, long-context, and tool-assisted tasks.","context_length":1047576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-04-14","last_updated":"2025-04-14","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4.1-nano/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.00000011","completion":"0.00000044","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000028","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"gpt-4.1-nano","tags":["multimodal","function-calling","long-context","efficient","high-throughput"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"glm-5v-turbo","object":"model","canonical_slug":"zhipu/glm-5v-turbo","hugging_face_id":"","name":"GLM 5V Turbo","created":1778850280,"description":"GLM-5V-Turbo is Z AI's API-only multimodal GLM-5-series vision model that accepts text and image input and produces text, optimized for fast agentic and vision-coding workflows with tool calling and toggleable reasoning.","context_length":202752,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-04-01","last_updated":"2026-04-01","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5v-turbo/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.0000012","completion":"0.000004","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000003","input_cache_write":"0","discount":1,"currency":"USD"},"author":"zhipu","slug":"glm-5v-turbo","tags":["vision","multimodal","long-context","agentic","reasoning"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"glm-5.3-flash","object":"model","canonical_slug":"zhipu/glm-5.3-flash","hugging_face_id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","created":1787987010,"description":"GLM-5.3-Flash is Z.ai's natively multimodal 320B-A18B mixture-of-experts model for reasoning, coding, agentic work, long-context tasks, images, and video.","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"ChatGLM","instruct_type":null},"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["parallel_tool_calls","reasoning_effort","stream","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-08-26","last_updated":null,"reasoning":{"mandatory":true,"supported_efforts":["low","high","max"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/zhipu/glm-5.3-flash/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"reasoning_effort":"max"},"providers":[{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"Inceptron","slug":"inceptron"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.00000015","completion":"0.00000050","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000007","input_cache_write":"0","discount":1,"currency":"USD"},"author":"zhipu","slug":"glm-5.3-flash","tags":["open-source","multimodal","vision","video","long-context","reasoning","agentic","function-calling"],"author_info":{"slug":"zhipu","name":"Zhipu","display_name":"Zhipu AI","icon_url":null,"gradient_from":"from-blue-500","gradient_to":"to-cyan-500","gradient_via":null,"website_url":"https://www.zhipuai.cn"}},{"id":"kimi-k2.7-code","object":"model","canonical_slug":"moonshot/kimi-k2.7-code","hugging_face_id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","created":1781703617,"description":"Kimi K2.7 Code is Moonshot AI's open-weight Mixture-of-Experts coding model for long-horizon agentic engineering.","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Kimi","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","n","parallel_tool_calls","presence_penalty","reasoning_effort","response_format","seed","stop","stream","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-06-12","last_updated":"2026-06-12","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/moonshot/kimi-k2.7-code/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Lyceum Technology GmbH","slug":"lyceum"},{"name":"GreenPT","slug":"greenpt"},{"name":"Inceptron","slug":"inceptron"},{"name":"Tensorix","slug":"tensorix"}],"pricing":{"prompt":"0.00000075","completion":"0.0000035","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.0000002","input_cache_write":"0","discount":1,"currency":"USD"},"author":"moonshot","slug":"kimi-k2.7-code","tags":["open-source","moe","long-context","coding","reasoning"],"author_info":{"slug":"moonshot","name":"Moonshot","display_name":"Moonshot AI","icon_url":null,"gradient_from":"from-indigo-500","gradient_to":"to-purple-600","gradient_via":null,"website_url":"https://moonshot.ai"}},{"id":"mistral/mistral-embed","object":"model","canonical_slug":"mistral/mistral-embed","hugging_face_id":"","name":"Mistral Embed","created":1788185018,"description":"Mistral AI embedding model for converting text and code into 1024-dimensional vectors for retrieval and similarity search.","context_length":8000,"architecture":{"modality":"text->embedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":8000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["encoding_format","metadata","output_dtype"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-12-11","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/mistral/mistral/mistral-embed/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Mistral AI","slug":"mistral"}],"author":"mistral","slug":"mistral/mistral-embed","tags":[],"author_info":{"slug":"mistral","name":"Mistral","display_name":"Mistral AI","icon_url":"/images/logos/mistral.jpeg","gradient_from":"from-[#ff7000]","gradient_to":"to-[#ff9a00]","gradient_via":null,"website_url":"https://mistral.ai"}},{"id":"titan-multimodal-embed","object":"model","canonical_slug":"amazon/titan-multimodal-embed","hugging_face_id":"","name":"Amazon Titan Multimodal Embeddings","created":1769967814,"description":"Amazon Titan Multimodal Embeddings generates embeddings for both text and images for cross-modal search.","context_length":8000,"architecture":{"modality":"text+image->embedding","input_modalities":["text","image"],"output_modalities":["embedding"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":8000,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-11-29","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/amazon/titan-multimodal-embed/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.000001","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"amazon","slug":"titan-multimodal-embed","tags":["proprietary","embedding","multimodal","vision"],"author_info":{"slug":"amazon","name":"Amazon","display_name":"Amazon Web Services","icon_url":"/images/logos/aws.webp","gradient_from":"from-orange-500","gradient_to":"to-yellow-500","gradient_via":null,"website_url":"https://aws.amazon.com/bedrock"}},{"id":"nemotron-3-super-120b-a12b","object":"model","canonical_slug":"nvidia/nemotron-3-super-120b-a12b","hugging_face_id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","created":1777454224,"description":"NVIDIA Nemotron 3 Super 120B A12B is a Mixture-of-Experts model for chat, coding, reasoning, and tool-use workloads.","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nemotron","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-03-11","last_updated":"2026-03-11","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-super-120b-a12b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebul","slug":"nebul"},{"name":"Nebius","slug":"nebius"}],"pricing":{"prompt":"0.00000030","completion":"0.00000090","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"nvidia","slug":"nemotron-3-super-120b-a12b","tags":["open-source","moe","coding","reasoning","function-calling"],"author_info":{"slug":"nvidia","name":"NVIDIA","display_name":null,"icon_url":"/images/logos/nvidia.webp","gradient_from":"from-[#76b900]","gradient_to":"to-[#5a8c00]","gradient_via":null,"website_url":"https://nvidia.com"}},{"id":"intfloat/e5-mistral-7b-instruct","object":"model","canonical_slug":"intfloat/e5-mistral-7b-instruct","hugging_face_id":"intfloat/e5-mistral-7b-instruct","name":"E5 Mistral 7B Instruct","created":1788171914,"description":"E5-Mistral-7B-Instruct is intfloat's instruction-tuned text embedding model built on Mistral-7B-v0.1. It produces 4096-dimensional normalized embeddings for retrieval, semantic search, similarity, classification, and RAG. Queries should include a one-sentence task instruction; documents do not require one. The author recommends English-only use.","context_length":4096,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":null,"instruct_type":null},"top_provider":{"context_length":4096,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2023-12-20","last_updated":null,"reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/intfloat/intfloat/e5-mistral-7b-instruct/endpoints"},"supported_api_endpoints":["embeddings"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Infercom","slug":"infercom"}],"pricing":{"prompt":"0.00000013","completion":"0","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"EUR"},"author":"intfloat","slug":"intfloat/e5-mistral-7b-instruct","tags":["open-source","embedding","instruction-tuned","retrieval","english"],"author_info":{"slug":"intfloat","name":"intfloat","display_name":null,"icon_url":null,"gradient_from":"from-sky-500","gradient_to":"to-blue-600","gradient_via":null,"website_url":"https://huggingface.co/intfloat"}},{"id":"o3-mini","object":"model","canonical_slug":"openai/o3-mini","hugging_face_id":"","name":"o3-mini","created":1767619877,"description":"o3-mini is OpenAI's compact text-only reasoning model optimized for cost-efficient coding, math, science, and structured-output tasks.","context_length":200000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","parallel_tool_calls","reasoning_effort","response_format","tool_choice","tools"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-01-31","last_updated":"2025-01-29","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/openai/o3-mini/endpoints"},"supported_api_endpoints":["chat.completions","responses"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Microsoft Foundry","slug":"microsoft-foundry"}],"pricing":{"prompt":"0.00000121","completion":"0.00000484","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.000000605","input_cache_write":"0","discount":1,"currency":"USD"},"author":"openai","slug":"o3-mini","tags":["reasoning","long-context","efficient","coding","math"],"author_info":{"slug":"openai","name":"OpenAI","display_name":null,"icon_url":"/images/logos/openai.webp","gradient_from":"from-zinc-800","gradient_to":"to-zinc-900","gradient_via":null,"website_url":"https://openai.com"}},{"id":"nemotron-3-nano-30b-a3b","object":"model","canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B","created":1768127608,"description":"NVIDIA Nemotron 3 Nano 30B A3B is an efficient MoE model for chat, reasoning, coding, and tool use.","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nemotron","instruct_type":null},"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2025-12-15","last_updated":"2025-12-15","reasoning":{"mandatory":false,"supported_efforts":[],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-nano-30b-a3b/endpoints"},"supported_api_endpoints":["chat.completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"Nebius","slug":"nebius"}],"pricing":{"prompt":"0.00000006","completion":"0.00000024","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0","discount":1,"currency":"USD"},"author":"nvidia","slug":"nemotron-3-nano-30b-a3b","tags":["open-source","moe","reasoning","coding","function-calling"],"author_info":{"slug":"nvidia","name":"NVIDIA","display_name":null,"icon_url":"/images/logos/nvidia.webp","gradient_from":"from-[#76b900]","gradient_to":"to-[#5a8c00]","gradient_via":null,"website_url":"https://nvidia.com"}},{"id":"claude-opus-4-8","object":"model","canonical_slug":"anthropic/claude-opus-4-8","hugging_face_id":"","name":"Claude Opus 4.8","created":1782845149,"description":"Claude Opus 4.8 is Anthropic's most capable model for complex coding, autonomous agents, and deep reasoning with a 1M context window.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","parallel_tool_calls","reasoning_effort","response_format","stop","tool_choice","tools","top_k","top_p"],"supported_voices":null,"knowledge_cutoff":null,"release_date":"2026-05-28","last_updated":"2026-05-28","reasoning":{"mandatory":false,"supported_efforts":["low","medium","high","xhigh","max"],"supports_max_tokens":false},"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-opus-4-8/endpoints"},"supported_api_endpoints":["chat.completions","completions"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"providers":[{"name":"AWS Bedrock","slug":"aws-bedrock"}],"pricing":{"prompt":"0.0000055","completion":"0.0000275","request":"0","image":"0","image_token":"0","image_output":"0","audio":"0","input_audio_cache":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0.00000055","input_cache_write":"0.000006875","discount":1,"currency":"USD"},"author":"anthropic","slug":"claude-opus-4-8","tags":["proprietary","coding","reasoning","agentic","long-context","vision"],"author_info":{"slug":"anthropic","name":"Anthropic","display_name":null,"icon_url":"/images/logos/anthropic.jpeg","gradient_from":"from-[#cc785c]","gradient_to":"to-[#d4a574]","gradient_via":null,"website_url":"https://anthropic.com"}}]}