{"object":"list","data":[{"id":"qwen3.8-max","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.8 Max","description":"Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...","context_length":1000000,"aliases":["qwen/qwen3.8-max"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"xhigh","default_enabled":true,"supported_efforts":["xhigh","high","medium","low","minimal"]}},{"id":"deepseek-v4-flash","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V4 Flash","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1000000,"aliases":["deepseek/deepseek-v4-flash"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","supported_efforts":["xhigh","high"]}},{"id":"glm-5v-turbo","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 5v Turbo","description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","context_length":202800,"aliases":["z-ai/glm-5v-turbo"],"pricing":{"prompt":"0.0000012","completion":"0.000004"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"deepseek-v4-pro","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V4 Pro","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1000000,"aliases":["deepseek/deepseek-v4-pro"],"pricing":{"prompt":"0.0000024","completion":"0.0000048"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","supported_efforts":["xhigh","high"]}},{"id":"kimi-k3","object":"model","created":1790425692,"owned_by":"moonshotai","name":"Kimi K3","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"aliases":["moonshotai/kimi-k3"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"max","default_enabled":true,"supported_efforts":["max","high","low"]}},{"id":"deepseek-v3.2","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V3.2","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"aliases":["deepseek/deepseek-v3.2"],"pricing":{"prompt":"2.8e-7","completion":"4.2e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"minimax-m3","object":"model","created":1790425692,"owned_by":"minimax","name":"Minimax M3","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":1048576,"aliases":["minimax/minimax-m3"],"pricing":{"prompt":"3e-7","completion":"0.0000012"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-2.5-flash-lite","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 2.5 Flash Lite","description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"aliases":["google/gemini-2.5-flash-lite"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3.7-plus","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.7 Plus","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","context_length":1000000,"aliases":["qwen/qwen3.7-plus"],"pricing":{"prompt":"3.2e-7","completion":"0.00000128"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"gemini-3.1-flash-lite-preview","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3.1 Flash Lite Preview","description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","context_length":1048576,"aliases":["google/gemini-3.1-flash-lite-preview"],"pricing":{"prompt":"2.5e-7","completion":"0.0000015"},"architecture":{"modality":"text+image+video+audio->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]}},{"id":"qwen3.7-max","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.7 Max","description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","context_length":1000000,"aliases":["qwen/qwen3.7-max"],"pricing":{"prompt":"0.000001475","completion":"0.000004425"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"gpt-5.3-codex","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.3 Codex","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","context_length":400000,"aliases":["openai/gpt-5.3-codex"],"pricing":{"prompt":"0.00000175","completion":"0.000014"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"gemini-3.1-pro-preview","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3.1 Pro Preview","description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"aliases":["google/gemini-3.1-pro-preview"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"pricing_tiers":[{"above_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000018"}],"architecture":{"modality":"audio+image+text+video->text","input_modalities":["audio","image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]}},{"id":"claude-sonnet-4.6","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Sonnet 4.6","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"aliases":["anthropic/claude-sonnet-4.6"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["max","high","medium","low"]}},{"id":"claude-opus-4.6","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Opus 4.6","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"aliases":["anthropic/claude-opus-4.6"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":false,"supported_efforts":["max","high","medium","low"],"supports_max_tokens":true}},{"id":"alibaba/wan3.0-video-edit","object":"model","created":1790425692,"owned_by":"alibaba","name":"Wan3.0 Video Edit","description":null,"context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+video->video","input_modalities":["text","video"],"output_modalities":["video"]}},{"id":"alibaba/wan3.0-image-to-video","object":"model","created":1790425692,"owned_by":"alibaba","name":"Wan3.0 Image To Video","description":null,"context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"]}},{"id":"alibaba/wan3.0-reference-to-video","object":"model","created":1790425692,"owned_by":"alibaba","name":"Wan3.0 Reference To Video","description":null,"context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image+video->video","input_modalities":["text","image","video"],"output_modalities":["video"]}},{"id":"alibaba/wan3.0-text-to-video","object":"model","created":1790425692,"owned_by":"alibaba","name":"Wan3.0 Text To Video","description":null,"context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text->video","input_modalities":["text"],"output_modalities":["video"]}},{"id":"qwen3.7-flash","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.7 Flash","description":null,"context_length":1000000,"aliases":["qwen/qwen3.7-flash"],"pricing":{"prompt":"3e-8","completion":"1.3e-7"},"pricing_tiers":[{"above_prompt_tokens":32768,"prompt":"1.0000000000000001e-7","completion":"4.0000000000000003e-7"},{"above_prompt_tokens":262144,"prompt":"2.0000000000000002e-7","completion":"8.000000000000001e-7"}],"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true}},{"id":"qwen3.5-35b-a3b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.5 35b A3b","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"aliases":["qwen/qwen3.5-35b-a3b"],"pricing":{"prompt":"3.125e-7","completion":"0.00000125"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3.5-9b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.5 9b","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"aliases":["qwen/qwen3.5-9b"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"1.5e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3.5-27b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.5 27b","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"aliases":["qwen/qwen3.5-27b"],"pricing":{"prompt":"1.95e-7","completion":"0.00000156"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemma-3-4b-it","object":"model","created":1790425692,"owned_by":"google","name":"Gemma 3 4b It","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"aliases":["google/gemma-3-4b-it"],"pricing":{"prompt":"5.0000000000000004e-8","completion":"1.0000000000000001e-7"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"qwen3-next-80b-a3b-instruct","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Next 80b A3b Instruct","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262140,"aliases":["qwen/qwen3-next-80b-a3b-instruct"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"0.0000011"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"gpt-6-luna","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 6 Luna","description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic...","context_length":1050000,"aliases":["openai/gpt-6-luna"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"5e-7"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"2.0000000000000002e-7","completion":"7.5e-7"}],"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"deepseek-v4.1-flash","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V4.1 Flash","description":null,"context_length":1000000,"aliases":["deepseek/deepseek-v4.1-flash"],"pricing":{"prompt":"3e-7","completion":"0.0000012"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","high","low"]}},{"id":"qwen3.5-397b-a17b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.5 397b A17b","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":262144,"aliases":["qwen/qwen3.5-397b-a17b"],"pricing":{"prompt":"5.5e-7","completion":"0.0000035"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3.6-plus","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.6 Plus","description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","context_length":1000000,"aliases":["qwen/qwen3.6-plus"],"pricing":{"prompt":"3.25e-7","completion":"0.00000195"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"mimo-v2.6-flash","object":"model","created":1790425692,"owned_by":"xiaomi","name":"Mimo V2.6 Flash","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","context_length":1048576,"aliases":["xiaomi/mimo-v2.6-flash"],"pricing":{"prompt":"1.4e-7","completion":"2.8e-7"},"architecture":{"modality":"text+image+video+audio->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-coder-flash","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Coder Flash","description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","context_length":128000,"aliases":["qwen/qwen3-coder-flash"],"pricing":{"prompt":"1.95e-7","completion":"9.75e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen3.5-122b-a10b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.5 122b A10b","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","context_length":262144,"aliases":["qwen/qwen3.5-122b-a10b"],"pricing":{"prompt":"2.9e-7","completion":"0.0000024"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-vl-235b-a22b-instruct","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Vl 235b A22b Instruct","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"aliases":["qwen/qwen3-vl-235b-a22b-instruct"],"pricing":{"prompt":"2.1e-7","completion":"0.0000018999999999999998"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]}},{"id":"deepseek-chat-v3.1","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek Chat V3.1","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":131072,"aliases":["deepseek/deepseek-chat-v3.1"],"pricing":{"prompt":"2.5e-7","completion":"9.499999999999999e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"glm-4.6","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 4.6","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":202752,"aliases":["z-ai/glm-4.6"],"pricing":{"prompt":"4.3e-7","completion":"0.00000175"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-3.5-flash-lite","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3.5 Flash Lite","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1000000,"aliases":["google/gemini-3.5-flash-lite"],"pricing":{"prompt":"3e-7","completion":"0.0000025"},"architecture":{"modality":"text+image+video+audio->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]}},{"id":"gemini-2.5-flash","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 2.5 Flash","description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"aliases":["google/gemini-2.5-flash"],"pricing":{"prompt":"3e-7","completion":"0.0000025"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3.8-27b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.8 27b","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","context_length":262144,"aliases":["qwen/qwen3.8-27b"],"pricing":{"prompt":"4.5000000000000003e-7","completion":"0.0000032000000000000003"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"xhigh","default_enabled":true,"supported_efforts":["xhigh","medium","low"]}},{"id":"deepseek-chat","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek Chat","description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":163840,"aliases":["deepseek/deepseek-chat"],"pricing":{"prompt":"4.0000000000000003e-7","completion":"0.0000013"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"minimax-m2.1","object":"model","created":1790425692,"owned_by":"minimax","name":"Minimax M2.1","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","context_length":204800,"aliases":["minimax/minimax-m2.1"],"pricing":{"prompt":"3e-7","completion":"0.0000012"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"glm-5.3-flashx","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 5.3 Flashx","description":null,"context_length":1000000,"aliases":["z-ai/glm-5.3-flashx"],"pricing":{"prompt":"3.7e-7","completion":"0.00000125"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"max","default_enabled":true,"supported_efforts":["max","high","low"]}},{"id":"gemini-3-flash-preview","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3 Flash Preview","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1000000,"aliases":["google/gemini-3-flash-preview"],"pricing":{"prompt":"5e-7","completion":"0.000003"},"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","image","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]}},{"id":"glm-5.3","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 5.3","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","context_length":1024000,"aliases":["z-ai/glm-5.3"],"pricing":{"prompt":"0.0000014","completion":"0.0000044"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"max","default_enabled":true,"supported_efforts":["max","high","low"]}},{"id":"mimo-v2.6-pro","object":"model","created":1790425692,"owned_by":"xiaomi","name":"Mimo V2.6 Pro","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","context_length":1048576,"aliases":["xiaomi/mimo-v2.6-pro"],"pricing":{"prompt":"4.35e-7","completion":"8.7e-7"},"architecture":{"modality":"text+image+video+audio->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"deepseek-r1","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek R1","description":"DeepSeek R1 is here: Performance on par with OpenAI o1, but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","context_length":163840,"aliases":["deepseek/deepseek-r1"],"pricing":{"prompt":"7e-7","completion":"0.0000025"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"qwen3-coder-plus","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Coder Plus","description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","context_length":128000,"aliases":["qwen/qwen3-coder-plus"],"pricing":{"prompt":"6.5e-7","completion":"0.00000325"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-3.8-flash","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3.8 Flash","description":"Gemini 3.8 Flash is Google's most intelligent Flash model with significant gains from 3.7 Flash across software engineering, agentic tasks, and multi-step reasoning.","context_length":1000000,"aliases":["google/gemini-3.8-flash"],"pricing":{"prompt":"7.5e-7","completion":"0.00000375"},"architecture":{"modality":"text+image+video+audio->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low"]}},{"id":"gemini-3.6-flash","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3.6 Flash","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1000000,"aliases":["google/gemini-3.6-flash"],"pricing":{"prompt":"0.0000015","completion":"0.0000075"},"architecture":{"modality":"text+image+video+audio->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]}},{"id":"gpt-5.4-mini","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.4 Mini","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"aliases":["openai/gpt-5.4-mini"],"pricing":{"prompt":"7.5e-7","completion":"0.0000045"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"gemini-3.7-flash","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3.7 Flash","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","context_length":1000000,"aliases":["google/gemini-3.7-flash"],"pricing":{"prompt":"7.5e-7","completion":"0.00000375"},"architecture":{"modality":"text+image+video+audio->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low"]}},{"id":"qwen3-max","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Max","description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","context_length":262144,"aliases":["qwen/qwen3-max"],"pricing":{"prompt":"7.8e-7","completion":"0.0000039"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"grok-build-0.1","object":"model","created":1790425692,"owned_by":"x-ai","name":"Grok Build 0.1","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","context_length":256000,"aliases":["x-ai/grok-build-0.1"],"pricing":{"prompt":"0.000001","completion":"0.000002"},"pricing_tiers":[{"above_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004"}],"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"claude-haiku-4.5","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Haiku 4.5","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"aliases":["anthropic/claude-haiku-4.5"],"pricing":{"prompt":"0.000001","completion":"0.000005"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-2.5-pro-preview-05-06","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 2.5 Pro Preview 05 06","description":null,"context_length":1048576,"aliases":["google/gemini-2.5-pro-preview-05-06"],"pricing":{"prompt":"0.00000125","completion":"0.00001"},"pricing_tiers":[{"above_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000015"}],"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"gemini-2.5-pro","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 2.5 Pro","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"aliases":["google/gemini-2.5-pro"],"pricing":{"prompt":"0.00000125","completion":"0.00001"},"pricing_tiers":[{"above_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000015"}],"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"qwen3.8-2.4t-a95b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.8 2.4t A95b","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of Qwen3.8 Max, with 95 billion active parameters out of 2.4 trillion total. It is...","context_length":262144,"aliases":["qwen/qwen3.8-2.4t-a95b"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"xhigh","default_enabled":true,"supported_efforts":["xhigh","medium","low"]}},{"id":"grok-4.3","object":"model","created":1790425692,"owned_by":"x-ai","name":"Grok 4.3","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"aliases":["x-ai/grok-4.3"],"pricing":{"prompt":"0.00000125","completion":"0.0000025"},"pricing_tiers":[{"above_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005"}],"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"low","default_enabled":true,"supported_efforts":["high","medium","low","none"]}},{"id":"gemini-3.5-flash","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3.5 Flash","description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1000000,"aliases":["google/gemini-3.5-flash"],"pricing":{"prompt":"0.0000015","completion":"0.000009"},"architecture":{"modality":"text+image+video+audio->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]}},{"id":"gpt-6-sol","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 6 Sol","description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional...","context_length":1050000,"aliases":["openai/gpt-6-sol"],"pricing":{"prompt":"0.000002","completion":"0.00001"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015"}],"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"grok-4.6","object":"model","created":1790425692,"owned_by":"x-ai","name":"Grok 4.6","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by Grok 4.7.","context_length":500000,"aliases":["x-ai/grok-4.6"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"pricing_tiers":[{"above_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012"}],"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","default_enabled":true,"supported_efforts":["xhigh","high","medium","low"]}},{"id":"grok-4.5","object":"model","created":1790425692,"owned_by":"x-ai","name":"Grok 4.5","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","context_length":500000,"aliases":["x-ai/grok-4.5"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"pricing_tiers":[{"above_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012"}],"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","default_enabled":true,"supported_efforts":["high","medium","low"]}},{"id":"claude-sonnet-5","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Sonnet 5","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"aliases":["anthropic/claude-sonnet-5"],"pricing":{"prompt":"0.000002","completion":"0.00001"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"gpt-5.4","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.4","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"aliases":["openai/gpt-5.4"],"pricing":{"prompt":"0.0000025","completion":"0.000015"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"claude-sonnet-4.5","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Sonnet 4.5","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"aliases":["anthropic/claude-sonnet-4.5"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gpt-5.5","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.5","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"aliases":["openai/gpt-5.5"],"pricing":{"prompt":"0.000005","completion":"0.00003"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"claude-opus-5.5","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Opus 5.5","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","context_length":1000000,"aliases":["anthropic/claude-opus-5.5"],"pricing":{"prompt":"0.000004","completion":"0.00002"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"claude-opus-4.7","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Opus 4.7","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"aliases":["anthropic/claude-opus-4.7"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":false,"supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"claude-opus-4.8","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Opus 4.8","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"aliases":["anthropic/claude-opus-4.8"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":false,"supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"claude-opus-5","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Opus 5","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"aliases":["anthropic/claude-opus-5"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"claude-opus-4.5","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Opus 4.5","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"aliases":["anthropic/claude-opus-4.5"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"claude-fable-5","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Fable 5","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"aliases":["anthropic/claude-fable-5"],"pricing":{"prompt":"0.00001","completion":"0.00005"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"claude-fable-5.1","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Fable 5.1","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","context_length":1000000,"aliases":["anthropic/claude-fable-5.1"],"pricing":{"prompt":"0.00001","completion":"0.00005"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"gpt-5.6-luna","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.6 Luna","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"aliases":["openai/gpt-5.6-luna"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"0.0000012000000000000002"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"4.0000000000000003e-7","completion":"0.0000024000000000000003"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"gpt-5.6-terra","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.6 Terra","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"aliases":["openai/gpt-5.6-terra"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000024"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"bytedance/seedance-2.5","object":"model","created":1790425692,"owned_by":"bytedance","name":"Seedance 2.5","description":"Seedance 2.5 is a video generation model from ByteDance. It is suited for long-form storytelling, multimodal reference-based generation, video editing, and video extension. It supports first-frame and first-and-last-frame control, up to 50 image, video, and audio reference assets, optional generated audio, and multilingual audiovisual generation.","context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"]}},{"id":"x-ai/grok-imagine-image-2.0","object":"model","created":1790425692,"owned_by":"x-ai","name":"Grok Imagine Image 2.0","description":null,"context_length":null,"aliases":[],"pricing":{"prompt":"0","completion":"0","image":"0.01"},"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"]}},{"id":"qwen3-embedding-8b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Embedding 8b","description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"aliases":["qwen/qwen3-embedding-8b"],"pricing":{"prompt":"1e-8","completion":"0"},"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"]}},{"id":"qwen3-embedding-4b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Embedding 4b","description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"aliases":["qwen/qwen3-embedding-4b"],"pricing":{"prompt":"2e-8","completion":"0"},"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"]}},{"id":"typesafe-ai/jev","object":"model","created":1790425692,"owned_by":"typesafe-ai","name":"Jev","description":null,"context_length":32000,"aliases":[],"pricing":{"prompt":"4.2000000000000006e-8","completion":"0"},"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"]}},{"id":"glm-4.7-flash","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 4.7 Flash","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","context_length":200000,"aliases":["z-ai/glm-4.7-flash"],"pricing":{"prompt":"6.049999999999999e-8","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"qwen3.5-flash-02-23","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.5 Flash 02 23","description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","context_length":1000000,"aliases":["qwen/qwen3.5-flash-02-23"],"pricing":{"prompt":"6.5e-8","completion":"2.6e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemma-4-26b-a4b-it","object":"model","created":1790425692,"owned_by":"google","name":"Gemma 4 26b A4b It","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"aliases":["google/gemma-4-26b-a4b-it"],"pricing":{"prompt":"9e-8","completion":"3.4000000000000003e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"qwen3-coder-30b-a3b-instruct","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Coder 30b A3b Instruct","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","context_length":262144,"aliases":["qwen/qwen3-coder-30b-a3b-instruct"],"pricing":{"prompt":"7e-8","completion":"2.8e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"seed-1.6-flash","object":"model","created":1790425692,"owned_by":"bytedance-seed","name":"Seed 1.6 Flash","description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","context_length":262144,"aliases":["bytedance-seed/seed-1.6-flash"],"pricing":{"prompt":"7.5e-8","completion":"3e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemma-4-31b-it","object":"model","created":1790425692,"owned_by":"google","name":"Gemma 4 31b It","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"aliases":["google/gemma-4-31b-it"],"pricing":{"prompt":"9e-8","completion":"3.4000000000000003e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"seed-2.0-mini","object":"model","created":1790425692,"owned_by":"bytedance-seed","name":"Seed 2.0 Mini","description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","context_length":262144,"aliases":["bytedance-seed/seed-2.0-mini"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]}},{"id":"qwen3-vl-32b-instruct","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Vl 32b Instruct","description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_length":131072,"aliases":["qwen/qwen3-vl-32b-instruct"],"pricing":{"prompt":"1.0399999999999999e-7","completion":"4.1599999999999997e-7"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"qwen3-vl-8b-instruct","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Vl 8b Instruct","description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","context_length":262144,"aliases":["qwen/qwen3-vl-8b-instruct"],"pricing":{"prompt":"1.17e-7","completion":"4.5500000000000004e-7"},"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"]}},{"id":"qwen3-coder-next","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Coder Next","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","context_length":262144,"aliases":["qwen/qwen3-coder-next"],"pricing":{"prompt":"1.2e-7","completion":"8.000000000000001e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"glm-4.5-air","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 4.5 Air","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"aliases":["z-ai/glm-4.5-air"],"pricing":{"prompt":"1.3e-7","completion":"8.5e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"deepseek-v4-flash-0731","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V4 Flash 0731","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","context_length":1048576,"aliases":["deepseek/deepseek-v4-flash-0731"],"pricing":{"prompt":"4.4e-7","completion":"0.00000132"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","high","low"]}},{"id":"mimo-v2.5","object":"model","created":1790425692,"owned_by":"xiaomi","name":"Mimo V2.5","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","context_length":1050000,"aliases":["xiaomi/mimo-v2.5"],"pricing":{"prompt":"1.4e-7","completion":"2.8e-7"},"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-vl-30b-a3b-instruct","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Vl 30b A3b Instruct","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":262144,"aliases":["qwen/qwen3-vl-30b-a3b-instruct"],"pricing":{"prompt":"1.5e-7","completion":"6e-7"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"gemini-embedding-001","object":"model","created":1790425692,"owned_by":"google","name":"Gemini Embedding","description":"gemini-embedding-001 provides a unified cutting edge experience across domains, including science, legal, finance, and coding. This embedding model has consistently held a top spot on the Massive Text Embedding Benchmark...","context_length":20000,"aliases":["google/gemini-embedding-001"],"pricing":{"prompt":"1.5e-7","completion":"0"},"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"]}},{"id":"glm-5.3-flash","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 5.3 Flash","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","context_length":1310720,"aliases":["z-ai/glm-5.3-flash"],"pricing":{"prompt":"1.5e-7","completion":"5e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"max","default_enabled":true,"supported_efforts":["max","high","low"]}},{"id":"qwen3.6-35b-a3b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.6 35b A3b","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"aliases":["qwen/qwen3.6-35b-a3b"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"9.000000000000001e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"gemini-embedding-2-preview","object":"model","created":1790425692,"owned_by":"google","name":"Gemini Embedding 2 Preview","description":"Gemini Embedding 2 Preview is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It...","context_length":8192,"aliases":["google/gemini-embedding-2-preview"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"0"},"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"]}},{"id":"claude-3-haiku","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude 3 Haiku","description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","context_length":200000,"aliases":["anthropic/claude-3-haiku"],"pricing":{"prompt":"2.5e-7","completion":"0.00000125"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"seed-1.6","object":"model","created":1790425692,"owned_by":"bytedance-seed","name":"Seed 1.6","description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","context_length":262144,"aliases":["bytedance-seed/seed-1.6"],"pricing":{"prompt":"2.5e-7","completion":"0.000002"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"seed-2.0-lite","object":"model","created":1790425692,"owned_by":"bytedance-seed","name":"Seed 2.0 Lite","description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","context_length":262144,"aliases":["bytedance-seed/seed-2.0-lite"],"pricing":{"prompt":"2.5e-7","completion":"0.000002"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]}},{"id":"qwen3.5-plus-02-15","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.5 Plus 02 15","description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","context_length":1000000,"aliases":["qwen/qwen3.5-plus-02-15"],"pricing":{"prompt":"2.6e-7","completion":"0.00000156"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"deepseek-v3.2-exp","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V3.2 Exp","description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"aliases":["deepseek/deepseek-v3.2-exp"],"pricing":{"prompt":"2.7e-7","completion":"4.1e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"deepseek-v3.1-terminus","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V3.1 Terminus","description":"DeepSeek-V3.1 Terminus is an update to DeepSeek V3.1 that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","context_length":163840,"aliases":["deepseek/deepseek-v3.1-terminus"],"pricing":{"prompt":"2.7e-7","completion":"0.000001"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3.6-27b","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.6 27b","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"aliases":["qwen/qwen3.6-27b"],"pricing":{"prompt":"2.8899999999999995e-7","completion":"0.0000024"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"qwen3-coder","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3 Coder","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262144,"aliases":["qwen/qwen3-coder"],"pricing":{"prompt":"3e-7","completion":"0.000001"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen3.5-plus-20260420","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.5 Plus 20260420","description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","context_length":1000000,"aliases":["qwen/qwen3.5-plus-20260420"],"pricing":{"prompt":"3e-7","completion":"0.0000018000000000000001"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-2.5-flash-image","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 2.5 Flash Image","description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","context_length":32768,"aliases":["google/gemini-2.5-flash-image"],"pricing":{"prompt":"3e-7","completion":"0.0000025"},"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["text","image"]}},{"id":"glm-4.6v","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 4.6v","description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","context_length":131072,"aliases":["z-ai/glm-4.6v"],"pricing":{"prompt":"3e-7","completion":"9.000000000000001e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen-2.5-72b-instruct","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen 2.5 72b Instruct","description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"aliases":["qwen/qwen-2.5-72b-instruct"],"pricing":{"prompt":"3.6e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"deepseek-v4-pro-0813","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V4 Pro 0813","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","context_length":1048576,"aliases":["deepseek/deepseek-v4-pro-0813"],"pricing":{"prompt":"0.00000132","completion":"0.00000396"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","supported_efforts":["max","high","low"]}},{"id":"mimo-v2.5-pro","object":"model","created":1790425692,"owned_by":"xiaomi","name":"Mimo V2.5 Pro","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","context_length":1050000,"aliases":["xiaomi/mimo-v2.5-pro"],"pricing":{"prompt":"4.35e-7","completion":"8.7e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"deepseek-r1-0528","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek R1 0528","description":"May 28th update to the original DeepSeek R1 Performance on par with OpenAI o1, but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"aliases":["deepseek/deepseek-r1-0528"],"pricing":{"prompt":"5e-7","completion":"0.0000021499999999999997"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"gemini-3.1-flash-image","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3.1 Flash Image","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","context_length":131072,"aliases":["google/gemini-3.1-flash-image"],"pricing":{"prompt":"5e-7","completion":"0.000003"},"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["text","image"]},"reasoning":{"mandatory":false,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","minimal"]}},{"id":"gemini-3.1-flash-image-preview","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3.1 Flash Image Preview","description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","context_length":65536,"aliases":["google/gemini-3.1-flash-image-preview"],"pricing":{"prompt":"5e-7","completion":"0.000003"},"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["text","image"]},"reasoning":{"mandatory":false,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","minimal"]}},{"id":"glm-4.5v","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 4.5v","description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","context_length":65536,"aliases":["z-ai/glm-4.5v"],"pricing":{"prompt":"6e-7","completion":"0.0000018000000000000001"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"kimi-k2-0905","object":"model","created":1790425692,"owned_by":"moonshotai","name":"Kimi K2 0905","description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","context_length":262144,"aliases":["moonshotai/kimi-k2-0905"],"pricing":{"prompt":"6e-7","completion":"0.0000025"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"glm-4.5","object":"model","created":1790425692,"owned_by":"z-ai","name":"Glm 4.5","description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","context_length":131072,"aliases":["z-ai/glm-4.5"],"pricing":{"prompt":"6e-7","completion":"0.0000022"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"grok-4.7","object":"model","created":1790425692,"owned_by":"x-ai","name":"Grok 4.7","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","context_length":500000,"aliases":["x-ai/grok-4.7"],"pricing":{"prompt":"0.0000016000000000000001","completion":"0.0000048"},"pricing_tiers":[{"above_prompt_tokens":200000,"prompt":"0.0000032000000000000003","completion":"0.0000096"}],"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","default_enabled":true,"supported_efforts":["xhigh","high","medium","low"]}},{"id":"grok-4.20-multi-agent","object":"model","created":1790425692,"owned_by":"x-ai","name":"Grok 4.20 Multi Agent","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","context_length":2000000,"aliases":["x-ai/grok-4.20-multi-agent"],"pricing":{"prompt":"0.00000125","completion":"0.0000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["xhigh","high","medium","low"]}},{"id":"grok-4.20","object":"model","created":1790425692,"owned_by":"x-ai","name":"Grok 4.20","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","context_length":2000000,"aliases":["x-ai/grok-4.20"],"pricing":{"prompt":"0.00000125","completion":"0.0000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"gemini-2.5-pro-preview","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 2.5 Pro Preview","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"aliases":["google/gemini-2.5-pro-preview"],"pricing":{"prompt":"0.00000125","completion":"0.00001"},"pricing_tiers":[{"above_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000015"}],"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"gemini-3-pro-image","object":"model","created":1790425692,"owned_by":"google","name":"Gemini 3 Pro Image","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":131072,"aliases":["google/gemini-3-pro-image"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["text","image"]},"reasoning":{"mandatory":true}},{"id":"claude-sonnet-4","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Sonnet 4","description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","context_length":200000,"aliases":["anthropic/claude-sonnet-4"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"claude-opus-4.1","object":"model","created":1790425692,"owned_by":"anthropic","name":"Claude Opus 4.1","description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"aliases":["anthropic/claude-opus-4.1"],"pricing":{"prompt":"0.000015","completion":"0.000075"},"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"bytedance/seedance-1-5-pro","object":"model","created":1790425692,"owned_by":"bytedance","name":"Seedance 1 5 Pro","description":"ByteDance's next-generation audio-visual generation model with a 4.5B parameter Dual-Branch Diffusion Transformer architecture. Seedance 1.5 Pro generates video and audio simultaneously in a single unified pass — eliminating the timing issues of sequential audio dubbing. Supports multi-language lip-sync (English, Mandarin, Japanese, Korean, Spanish, and more), cinematic camera control (pan, tilt, zoom, orbit), multi-character dialogue, and character consistency across shots. Produces clips from 4–12 seconds at up to 1080p. The number of tokens is given by (height of output video * width of output video * duration * 24) / 1024","context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"]}},{"id":"google/veo-3.1","object":"model","created":1790425692,"owned_by":"google","name":"Veo 3.1","description":"Google's state-of-the-art video generation model, built for maximum visual fidelity in final production cuts. Veo 3.1 generates high-quality 1080p video from text or image prompts with native synchronized audio — including dialogue, ambient effects, and background sound. Supports scene extension (up to 20 chained clips for 140+ second narratives), frames-to-video transitions between two images, vertical video for Shorts, and 4K upscaling.","context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"]}},{"id":"minimax/hailuo-3-max","object":"model","created":1790425692,"owned_by":"minimax","name":"Hailuo 3 Max","description":"MiniMax H3 Max is a video-generation model from MiniMax, jointly released with fal.ai. Derived through additional training from MiniMax H3, it is designed for faster text-to-video and image-to-video generation with...","context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"]}},{"id":"qwen/qwen-image-3","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen Image 3","description":null,"context_length":null,"aliases":[],"pricing":{"prompt":"0","completion":"0","image":"0.03"},"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"]}},{"id":"bytedance/seedance-2.0","object":"model","created":1790425692,"owned_by":"bytedance","name":"Seedance 2.0","description":"Seedance 2.0 is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It is particularly strong at preserving character consistency, visual style, and camera movement from reference material. The number of tokens is given by (height of output video * width of output video * duration * 24) / 1024","context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"]}},{"id":"openai/sora-2-pro","object":"model","created":1790425692,"owned_by":"openai","name":"Sora 2 Pro","description":"OpenAI's flagship video generation model, delivering production-quality video with physics-accurate motion, synchronized audio, and world-state persistence across shots. Sora 2 Pro follows intricate multi-shot instructions while maintaining consistent spatial relationships — objects don't disappear or change shape between cuts. Supports text-to-video and image-to-video, with synchronized background soundscapes, speech, and sound effects. Includes advanced content safety with C2PA metadata provenance and SynthID-style watermarking.","context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"]}},{"id":"qwen/qwen-image-3-pro","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen Image 3 Pro","description":null,"context_length":null,"aliases":[],"pricing":{"prompt":"0","completion":"0","image":"0.075"},"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"]}},{"id":"bytedance/seedance-2.0-fast","object":"model","created":1790425692,"owned_by":"bytedance","name":"Seedance 2.0 Fast","description":"Seedance 2.0 Fast is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It prioritizes generation speed and lower cost over maximum output quality. The number of tokens is given by (height of output video * width of output video * duration * 24) / 1024","context_length":0,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"]}},{"id":"text-embedding-3-small","object":"model","created":1790425692,"owned_by":"openai","name":"Text Embedding 3 Small","description":"text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...","context_length":8192,"aliases":["openai/text-embedding-3-small"],"pricing":{"prompt":"2.1000000000000003e-8","completion":"0"},"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"]}},{"id":"gpt-oss-20b","object":"model","created":1790425692,"owned_by":"openai","name":"GPT Oss 20b","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"aliases":["openai/gpt-oss-20b"],"pricing":{"prompt":"3e-8","completion":"1.3e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]}},{"id":"gpt-oss-safeguard-20b","object":"model","created":1790425692,"owned_by":"openai","name":"GPT Oss Safeguard 20b","description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","context_length":131072,"aliases":["openai/gpt-oss-safeguard-20b"],"pricing":{"prompt":"7.5e-8","completion":"3e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"text-embedding-ada-002","object":"model","created":1790425692,"owned_by":"openai","name":"Text Embedding Ada 002","description":"text-embedding-ada-002 is OpenAI's legacy text embedding model.","context_length":8192,"aliases":["openai/text-embedding-ada-002"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"0"},"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"]}},{"id":"text-embedding-3-large","object":"model","created":1790425692,"owned_by":"openai","name":"Text Embedding 3 Large","description":"text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...","context_length":8192,"aliases":["openai/text-embedding-3-large"],"pricing":{"prompt":"1.3e-7","completion":"0"},"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"]}},{"id":"gpt-oss-120b","object":"model","created":1790425692,"owned_by":"openai","name":"GPT Oss 120b","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"aliases":["openai/gpt-oss-120b"],"pricing":{"prompt":"1.5e-7","completion":"6e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]}},{"id":"gpt-5.6-luna-pro","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.6 Luna Pro","description":"GPT-5.6 Luna Pro is the same underlying model as GPT-5.6 Luna, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"aliases":["openai/gpt-5.6-luna-pro"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"0.0000012000000000000002"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"4.0000000000000003e-7","completion":"0.0000024000000000000003"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"gpt-3.5-turbo","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 3.5 Turbo","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"aliases":["openai/gpt-3.5-turbo"],"pricing":{"prompt":"5e-7","completion":"0.0000015"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"gpt-5.6-sol-pro","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.6 Sol Pro","description":"GPT-5.6 Sol Pro is the same underlying model as GPT-5.6 Sol, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"aliases":["openai/gpt-5.6-sol-pro"],"pricing":{"prompt":"0.000004","completion":"0.00002"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.000008","completion":"0.00003"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"gpt-5.6-sol","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.6 Sol","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"aliases":["openai/gpt-5.6-sol"],"pricing":{"prompt":"0.000004","completion":"0.00002"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.000008","completion":"0.00003"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"knowledge_cutoff":"2026-02-16","max_completion_tokens":128000,"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"gpt-5.6-terra-pro","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.6 Terra Pro","description":"GPT-5.6 Terra Pro is the same underlying model as GPT-5.6 Terra, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"aliases":["openai/gpt-5.6-terra-pro"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000024"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"gpt-image-1-mini","object":"model","created":1790425692,"owned_by":"openai","name":"GPT Image 1 Mini","description":null,"context_length":400000,"aliases":["openai/gpt-image-1-mini"],"pricing":{"prompt":"0.0000025","completion":"0.0000025"},"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"]}},{"id":"gpt-4o","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 4o","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of GPT-4 Turbo while being twice as...","context_length":128000,"aliases":["openai/gpt-4o"],"pricing":{"prompt":"0.0000025","completion":"0.00001"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]}},{"id":"gpt-image-2.5-sunburst","object":"model","created":1790425692,"owned_by":"openai","name":"GPT Image 2.5 Sunburst","description":null,"context_length":400000,"aliases":["openai/gpt-image-2.5-sunburst"],"pricing":{"prompt":"0.000008","completion":"0.000008"},"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"]}},{"id":"gpt-image-2.5-flare","object":"model","created":1790425692,"owned_by":"openai","name":"GPT Image 2.5 Flare","description":null,"context_length":400000,"aliases":["openai/gpt-image-2.5-flare"],"pricing":{"prompt":"0.000008","completion":"0.000008"},"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"]}},{"id":"gpt-5.4-image-2","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 5.4 Image 2","description":"GPT-5.4 Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","context_length":272000,"aliases":["openai/gpt-5.4-image-2"],"pricing":{"prompt":"0.000008","completion":"0.000015"},"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["text","image"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"gpt-6-astra","object":"model","created":1790425692,"owned_by":"openai","name":"GPT 6 Astra","description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","context_length":1050000,"aliases":["openai/gpt-6-astra"],"pricing":{"prompt":"0.00001","completion":"0.00005"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.00002","completion":"0.000075"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"gpt-image-1","object":"model","created":1790425692,"owned_by":"openai","name":"GPT Image 1","description":null,"context_length":400000,"aliases":["openai/gpt-image-1"],"pricing":{"prompt":"0.00001","completion":"0.00001"},"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"]}},{"id":"deepseek-v4-flash-0731free","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V4 Flash 0731free","description":null,"context_length":1048576,"aliases":["deepseek/deepseek-v4-flash-0731free"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen/qwen3.7-flash:free","object":"model","created":1790425692,"owned_by":"qwen","name":"Qwen3.7 Flash (free)","description":"Rate-limited free tier.","context_length":1000000,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true}},{"id":"deepseek/deepseek-v4-flash-0731free:free","object":"model","created":1790425692,"owned_by":"deepseek","name":"Deepseek V4 Flash 0731free (free)","description":"Rate-limited free tier.","context_length":1048576,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"auto","object":"model","created":1790425692,"owned_by":"bazaarlink","name":"Auto Router","description":"Automatically selects the best model for the request.","aliases":[],"pricing":{"prompt":"-1","completion":"-1"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"auto:free","object":"model","created":1790425692,"owned_by":"bazaarlink","name":"Auto Router (free)","description":"Automatically routes to a free model. Always free, rate-limited.","aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}}]}