{"models":[{"id":"qwen/qwen3.8-max","label":"Qwen3.8 Max","provider":"Qwen","category":["long-context"],"inputPrice":2,"outputPrice":6,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-08-03T04:33:32.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"xhigh","default_enabled":true,"supported_efforts":["xhigh","high","medium","low","minimal"]},"cacheReadPrice":0.25,"cacheWritePrice":2.5,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...","supportedParameters":"frequency_penalty,include_reasoning,logprobs,max_tokens,presence_penalty,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"deepseek/deepseek-v4-flash","label":"Deepseek V4 Flash","provider":"DeepSeek","category":["fast","cheap","long-context"],"inputPrice":0.2,"outputPrice":0.4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-04-24T03:17:46.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","supported_efforts":["xhigh","high"]},"cacheReadPrice":0.03,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":34.3,"rankIntelligence":70,"totalIntelligence":664,"rankCoding":55,"totalCoding":259,"rankSpeed":26,"totalSpeed":333,"rankInputPrice":215,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_completion_tokens,max_tokens,min_p,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_a,top_k,top_logprobs,top_p"},{"id":"z-ai/glm-5v-turbo","label":"Glm 5v Turbo","provider":"Zhipu AI","category":["long-context"],"inputPrice":1.2,"outputPrice":4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":202800,"releasedAt":"2026-04-01T16:37:38.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":true},"cacheReadPrice":0.252,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":23.5,"rankIntelligence":163,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":null,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","supportedParameters":"include_reasoning,max_tokens,reasoning,response_format,temperature,tool_choice,tools,top_k,top_p"},{"id":"deepseek/deepseek-v4-pro","label":"Deepseek V4 Pro","provider":"DeepSeek","category":["long-context"],"inputPrice":2.4,"outputPrice":4.8,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-04-24T03:17:59.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","supported_efforts":["xhigh","high"]},"cacheReadPrice":0.135,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":36,"rankIntelligence":64,"totalIntelligence":664,"rankCoding":56,"totalCoding":259,"rankSpeed":168,"totalSpeed":333,"rankInputPrice":315,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_completion_tokens,max_tokens,min_p,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"moonshotai/kimi-k3","label":"Kimi K3","provider":"Moonshot AI","category":["long-context"],"inputPrice":3,"outputPrice":15,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2026-07-16T15:30:58.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"max","default_enabled":true,"supported_efforts":["max","high","low"]},"cacheReadPrice":0.30264435,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"deepseek/deepseek-v3.2","label":"Deepseek V3.2","provider":"DeepSeek","category":["coding","cheap"],"inputPrice":0.28,"outputPrice":0.42,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":163840,"releasedAt":"2025-12-01T13:10:42.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":false},"cacheReadPrice":0.1345,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":"2026-09-28T00:00:00.000Z","aa":{"intelligenceIndex":16,"rankIntelligence":256,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":158,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"minimax/minimax-m3","label":"Minimax M3","provider":"MiniMax","category":["fast","cheap","long-context"],"inputPrice":0.3,"outputPrice":1.2,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2026-05-31T16:36:14.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.06,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":29.2,"rankIntelligence":103,"totalIntelligence":664,"rankCoding":85,"totalCoding":259,"rankSpeed":49,"totalSpeed":333,"rankInputPrice":163,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"google/gemini-2.5-flash-lite","label":"Gemini 2.5 Flash Lite","provider":"Google","category":["fast","cheap","long-context"],"inputPrice":0.1,"outputPrice":0.4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2025-07-22T16:04:36.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.01,"cacheWritePrice":0.0833333333333333,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":6.7,"rankIntelligence":522,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":13,"totalSpeed":333,"rankInputPrice":58,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","supportedParameters":"include_reasoning,max_tokens,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"qwen/qwen3.7-plus","label":"Qwen3.7 Plus","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.32,"outputPrice":1.28,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-06-03T13:03:03.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":true},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","supportedParameters":"frequency_penalty,include_reasoning,logprobs,max_tokens,presence_penalty,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"google/gemini-3.1-flash-lite-preview","label":"Gemini 3.1 Flash Lite Preview","provider":"Google","category":["fast","cheap","long-context"],"inputPrice":0.25,"outputPrice":1.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2026-03-03T04:37:53.000Z","modality":"text+image+video+audio->text","inputModalities":"text,image,video,audio","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]},"cacheReadPrice":0.025,"cacheWritePrice":0.0833333333333333,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":15.6,"rankIntelligence":258,"totalIntelligence":664,"rankCoding":154,"totalCoding":259,"rankSpeed":8,"totalSpeed":333,"rankInputPrice":143,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"qwen/qwen3.7-max","label":"Qwen3.7 Max","provider":"Qwen","category":["long-context"],"inputPrice":1.475,"outputPrice":4.425,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-05-21T15:21:01.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":true},"cacheReadPrice":0.295,"cacheWritePrice":1.84375,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","supportedParameters":"frequency_penalty,include_reasoning,logprobs,max_tokens,presence_penalty,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"openai/gpt-5.3-codex","label":"GPT 5.3 Codex","provider":"OpenAI","category":["long-context"],"inputPrice":1.75,"outputPrice":14,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":400000,"releasedAt":"2026-02-24T18:52:44.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","supported_efforts":["xhigh","high","medium","low","none"]},"cacheReadPrice":0.175,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":32.5,"rankIntelligence":85,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":113,"totalSpeed":333,"rankInputPrice":328,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"google/gemini-3.1-pro-preview","label":"Gemini 3.1 Pro Preview","provider":"Google","category":["fast","long-context"],"inputPrice":2,"outputPrice":12,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2026-02-19T14:00:27.000Z","modality":"audio+image+text+video->text","inputModalities":"audio,image,text,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]},"cacheReadPrice":0.2,"cacheWritePrice":0.375,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":200000,"input":4,"output":18}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":29.7,"rankIntelligence":100,"totalIntelligence":664,"rankCoding":56,"totalCoding":259,"rankSpeed":112,"totalSpeed":333,"rankInputPrice":333,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"anthropic/claude-sonnet-4.6","label":"Claude Sonnet 4.6","provider":"Anthropic","category":["flagship","coding","long-context"],"inputPrice":3,"outputPrice":15,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-02-17T15:43:10.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","supported_efforts":["max","high","medium","low"]},"cacheReadPrice":0.3,"cacheWritePrice":3.75,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":24.7,"rankIntelligence":148,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":294,"totalSpeed":333,"rankInputPrice":378,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p,verbosity"},{"id":"anthropic/claude-opus-4.6","label":"Claude Opus 4.6","provider":"Anthropic","category":["flagship","coding","creative","long-context"],"inputPrice":5,"outputPrice":25,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-02-04T15:30:50.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","default_enabled":false,"supported_efforts":["max","high","medium","low"],"supports_max_tokens":true},"cacheReadPrice":0.5,"cacheWritePrice":6.25,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":26.4,"rankIntelligence":126,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":316,"totalSpeed":333,"rankInputPrice":402,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p,verbosity"},{"id":"alibaba/wan3.0-video-edit","label":"Wan3.0 Video Edit","provider":"Alibaba","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":null,"modality":"text+video->video","inputModalities":"text,video","outputModalities":"video","videoModes":null,"pricePerImage":null,"billingModel":"request","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":0.03,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":null},{"id":"alibaba/wan3.0-image-to-video","label":"Wan3.0 Image To Video","provider":"Alibaba","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":null,"modality":"text+image->video","inputModalities":"text,image","outputModalities":"video","videoModes":null,"pricePerImage":null,"billingModel":"request","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":0.03,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":null},{"id":"alibaba/wan3.0-reference-to-video","label":"Wan3.0 Reference To Video","provider":"Alibaba","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":null,"modality":"text+image+video->video","inputModalities":"text,image,video","outputModalities":"video","videoModes":null,"pricePerImage":null,"billingModel":"request","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":0.03,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":null},{"id":"alibaba/wan3.0-text-to-video","label":"Wan3.0 Text To Video","provider":"Alibaba","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":null,"modality":"text->video","inputModalities":"text","outputModalities":"video","videoModes":null,"pricePerImage":null,"billingModel":"request","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":0.03,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":null},{"id":"qwen/qwen3.7-flash","label":"Qwen3.7 Flash","provider":"Qwen","category":["fast","cheap","long-context"],"inputPrice":0.03,"outputPrice":0.13,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":null,"modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true},"cacheReadPrice":0.006,"cacheWritePrice":0.038,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":32768,"input":0.1,"output":0.4},{"abovePromptTokens":262144,"input":0.2,"output":0.8}],"effectiveDeprecationDate":null,"aa":null,"isFree":true,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":null},{"id":"qwen/qwen3.5-35b-a3b","label":"Qwen3.5 35b A3b","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.3125,"outputPrice":1.25,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-02-25T21:10:22.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.25,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3.5-9b","label":"Qwen3.5 9b","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.1,"outputPrice":0.15,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-03-10T14:19:56.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3.5-27b","label":"Qwen3.5 27b","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.195,"outputPrice":1.56,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-02-25T21:10:10.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"google/gemma-3-4b-it","label":"Gemma 3 4b It","provider":"Google","category":["cheap"],"inputPrice":0.05,"outputPrice":0.1,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2025-03-13T22:38:30.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","supportedParameters":"frequency_penalty,logit_bias,max_tokens,min_p,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,top_k,top_p"},{"id":"qwen/qwen3-next-80b-a3b-instruct","label":"Qwen3 Next 80b A3b Instruct","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.1,"outputPrice":1.1,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262140,"releasedAt":"2025-09-11T17:36:53.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":0.07,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,min_p,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"openai/gpt-6-luna","label":"GPT 6 Luna","provider":"OpenAI","category":["cheap","long-context"],"inputPrice":0.1,"outputPrice":0.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1050000,"releasedAt":"2026-09-22T18:13:06.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]},"cacheReadPrice":0.01,"cacheWritePrice":0.125,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":272000,"input":0.2,"output":0.75}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":37.3,"rankIntelligence":60,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":84,"totalSpeed":333,"rankInputPrice":58,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"deepseek/deepseek-v4.1-flash","label":"Deepseek V4.1 Flash","provider":"DeepSeek","category":["fast","cheap","long-context"],"inputPrice":0.3,"outputPrice":1.2,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":null,"modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","high","low"]},"cacheReadPrice":0.03,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":39.5,"rankIntelligence":51,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":21,"totalSpeed":333,"rankInputPrice":163,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3.5-397b-a17b","label":"Qwen3.5 397b A17b","provider":"Qwen","category":["long-context"],"inputPrice":0.55,"outputPrice":3.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-02-16T06:23:38.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.3,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3.6-plus","label":"Qwen3.6 Plus","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.325,"outputPrice":1.95,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-04-02T12:39:17.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":0.40625,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","supportedParameters":"frequency_penalty,include_reasoning,logprobs,max_tokens,presence_penalty,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"xiaomi/mimo-v2.6-flash","label":"Mimo V2.6 Flash","provider":"xiaomi","category":["fast","cheap","long-context"],"inputPrice":0.14,"outputPrice":0.28,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2026-09-21T20:07:44.000Z","modality":"text+image+video+audio->text","inputModalities":"text,image,video,audio","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.0028,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"qwen/qwen3-coder-flash","label":"Qwen3 Coder Flash","provider":"Qwen","category":["fast","cheap"],"inputPrice":0.195,"outputPrice":0.975,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":128000,"releasedAt":"2025-09-17T13:25:36.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":0.039,"cacheWritePrice":0.24375,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","supportedParameters":"frequency_penalty,logprobs,max_tokens,presence_penalty,response_format,seed,stop,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3.5-122b-a10b","label":"Qwen3.5 122b A10b","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.29,"outputPrice":2.4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-02-25T21:09:49.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3-vl-235b-a22b-instruct","label":"Qwen3 Vl 235b A22b Instruct","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.21,"outputPrice":1.9,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2025-09-23T23:04:47.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":0.1,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,min_p,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"deepseek/deepseek-chat-v3.1","label":"Deepseek Chat V3.1","provider":"DeepSeek","category":["coding","cheap"],"inputPrice":0.25,"outputPrice":0.95,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2025-08-21T12:33:48.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.13,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"z-ai/glm-4.6","label":"Glm 4.6","provider":"Zhipu AI","category":["cheap","long-context"],"inputPrice":0.43,"outputPrice":1.75,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":202752,"releasedAt":"2025-09-30T12:32:56.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.08,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":14.9,"rankIntelligence":272,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":319,"totalSpeed":333,"rankInputPrice":233,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"google/gemini-3.5-flash-lite","label":"Gemini 3.5 Flash Lite","provider":"Google","category":["fast","cheap","long-context"],"inputPrice":0.3,"outputPrice":2.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-07-21T15:12:06.000Z","modality":"text+image+video+audio->text","inputModalities":"text,image,video,audio","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]},"cacheReadPrice":0.03,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":22.2,"rankIntelligence":183,"totalIntelligence":664,"rankCoding":114,"totalCoding":259,"rankSpeed":6,"totalSpeed":333,"rankInputPrice":163,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"google/gemini-2.5-flash","label":"Gemini 2.5 Flash","provider":"Google","category":["fast","cheap","long-context"],"inputPrice":0.3,"outputPrice":2.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2025-06-17T15:01:28.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.03,"cacheWritePrice":0.0833333333333333,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":9.9,"rankIntelligence":375,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":39,"totalSpeed":333,"rankInputPrice":163,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","supportedParameters":"include_reasoning,max_tokens,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"qwen/qwen3.8-27b","label":"Qwen3.8 27b","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.45,"outputPrice":3.2,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-08-14T15:55:10.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"xhigh","default_enabled":true,"supported_efforts":["xhigh","medium","low"]},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_a,top_k,top_logprobs,top_p"},{"id":"deepseek/deepseek-chat","label":"Deepseek Chat","provider":"DeepSeek","category":["coding","cheap"],"inputPrice":0.4,"outputPrice":1.3,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":163840,"releasedAt":"2024-12-26T19:28:40.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":5.3,"rankIntelligence":612,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":null,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","supportedParameters":"frequency_penalty,logit_bias,max_tokens,min_p,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"minimax/minimax-m2.1","label":"Minimax M2.1","provider":"MiniMax","category":["fast","cheap","long-context"],"inputPrice":0.3,"outputPrice":1.2,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":204800,"releasedAt":"2025-12-23T01:56:37.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true},"cacheReadPrice":0.03,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":20.9,"rankIntelligence":195,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":159,"totalSpeed":333,"rankInputPrice":163,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,temperature,tool_choice,tools,top_k,top_p"},{"id":"z-ai/glm-5.3-flashx","label":"Glm 5.3 Flashx","provider":"Zhipu AI","category":["fast","cheap","long-context"],"inputPrice":0.37,"outputPrice":1.25,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":null,"modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"max","default_enabled":true,"supported_efforts":["max","high","low"]},"cacheReadPrice":0.075,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":null},{"id":"google/gemini-3-flash-preview","label":"Gemini 3 Flash Preview","provider":"Google","category":["fast","cheap","long-context"],"inputPrice":0.5,"outputPrice":3,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2025-12-17T15:57:58.000Z","modality":"text+image+audio+video->text","inputModalities":"text,image,audio,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]},"cacheReadPrice":0.05,"cacheWritePrice":0.0833333333333333,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":15.6,"rankIntelligence":258,"totalIntelligence":664,"rankCoding":154,"totalCoding":259,"rankSpeed":8,"totalSpeed":333,"rankInputPrice":143,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"z-ai/glm-5.3","label":"Glm 5.3","provider":"Zhipu AI","category":["long-context"],"inputPrice":1.4,"outputPrice":4.4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1024000,"releasedAt":"2026-08-18T20:57:35.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"max","default_enabled":true,"supported_efforts":["max","high","low"]},"cacheReadPrice":0.26,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":44.8,"rankIntelligence":27,"totalIntelligence":664,"rankCoding":29,"totalCoding":259,"rankSpeed":228,"totalSpeed":333,"rankInputPrice":318,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,parallel_tool_calls,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"xiaomi/mimo-v2.6-pro","label":"Mimo V2.6 Pro","provider":"xiaomi","category":["cheap","long-context"],"inputPrice":0.435,"outputPrice":0.87,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2026-09-21T20:07:39.000Z","modality":"text+image+video+audio->text","inputModalities":"text,image,video,audio","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.0036,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":46.3,"rankIntelligence":22,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":298,"totalSpeed":333,"rankInputPrice":209,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"deepseek/deepseek-r1","label":"Deepseek R1","provider":"DeepSeek","category":["reasoning"],"inputPrice":0.7,"outputPrice":2.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":163840,"releasedAt":"2025-01-20T13:51:35.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":13.1,"rankIntelligence":303,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":316,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek R1 is here: Performance on par with OpenAI o1, but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,temperature,tool_choice,tools,top_k,top_p"},{"id":"qwen/qwen3-coder-plus","label":"Qwen3 Coder Plus","provider":"Qwen","category":["flagship"],"inputPrice":0.65,"outputPrice":3.25,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":128000,"releasedAt":"2025-09-23T21:25:07.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.13,"cacheWritePrice":0.8125,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","supportedParameters":"frequency_penalty,logprobs,max_tokens,presence_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"google/gemini-3.8-flash","label":"Gemini 3.8 Flash","provider":"Google","category":["fast","long-context"],"inputPrice":0.75,"outputPrice":3.75,"originalInputPrice":1.5,"originalOutputPrice":7.5,"discount":50,"contextLength":1000000,"releasedAt":"2026-09-02T15:14:16.000Z","modality":"text+image+video+audio->text","inputModalities":"text,image,video,audio","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low"]},"cacheReadPrice":0.075,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":40.9,"rankIntelligence":42,"totalIntelligence":664,"rankCoding":18,"totalCoding":259,"rankSpeed":9,"totalSpeed":333,"rankInputPrice":258,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3.8 Flash is Google's most intelligent Flash model with significant gains from 3.7 Flash across software engineering, agentic tasks, and multi-step reasoning.","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"google/gemini-3.6-flash","label":"Gemini 3.6 Flash","provider":"Google","category":["fast","long-context"],"inputPrice":1.5,"outputPrice":7.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-07-21T15:12:13.000Z","modality":"text+image+video+audio->text","inputModalities":"text,image,video,audio","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]},"cacheReadPrice":0.075,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":34,"rankIntelligence":73,"totalIntelligence":664,"rankCoding":54,"totalCoding":259,"rankSpeed":32,"totalSpeed":333,"rankInputPrice":258,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"openai/gpt-5.4-mini","label":"GPT 5.4 Mini","provider":"OpenAI","category":["fast","long-context"],"inputPrice":0.75,"outputPrice":4.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":400000,"releasedAt":"2026-03-17T11:49:38.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"]},"cacheReadPrice":0.075,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":24.1,"rankIntelligence":157,"totalIntelligence":664,"rankCoding":93,"totalCoding":259,"rankSpeed":47,"totalSpeed":333,"rankInputPrice":258,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"google/gemini-3.7-flash","label":"Gemini 3.7 Flash","provider":"Google","category":["fast","long-context"],"inputPrice":0.75,"outputPrice":3.75,"originalInputPrice":1.5,"originalOutputPrice":7.5,"discount":50,"contextLength":1000000,"releasedAt":"2026-08-13T17:03:01.000Z","modality":"text+image+video+audio->text","inputModalities":"text,image,video,audio","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low"]},"cacheReadPrice":0.075,"cacheWritePrice":0.05,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":39.1,"rankIntelligence":54,"totalIntelligence":664,"rankCoding":22,"totalCoding":259,"rankSpeed":7,"totalSpeed":333,"rankInputPrice":258,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"qwen/qwen3-max","label":"Qwen3 Max","provider":"Qwen","category":["long-context"],"inputPrice":0.78,"outputPrice":3.9,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2025-09-23T21:26:48.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.156,"cacheWritePrice":0.975,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","supportedParameters":"frequency_penalty,logprobs,max_tokens,presence_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"x-ai/grok-build-0.1","label":"Grok Build 0.1","provider":"xAI","category":["creative","long-context"],"inputPrice":1,"outputPrice":2,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":256000,"releasedAt":"2026-05-20T17:28:43.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true},"cacheReadPrice":0.2,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":200000,"input":2,"output":4}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":27.2,"rankIntelligence":120,"totalIntelligence":664,"rankCoding":109,"totalCoding":259,"rankSpeed":287,"totalSpeed":333,"rankInputPrice":274,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","supportedParameters":"include_reasoning,logprobs,max_tokens,reasoning,response_format,seed,structured_outputs,temperature,tool_choice,tools,top_logprobs,top_p"},{"id":"anthropic/claude-haiku-4.5","label":"Claude Haiku 4.5","provider":"Anthropic","category":["fast","long-context"],"inputPrice":1,"outputPrice":5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":200000,"releasedAt":"2025-10-15T17:00:38.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.1,"cacheWritePrice":1.25,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":15.4,"rankIntelligence":264,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":194,"totalSpeed":333,"rankInputPrice":274,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,response_format,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"google/gemini-2.5-pro-preview-05-06","label":"Gemini 2.5 Pro Preview 05 06","provider":"Google","category":["flagship","reasoning","fast","long-context"],"inputPrice":1.25,"outputPrice":10,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":null,"modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true},"cacheReadPrice":0.125,"cacheWritePrice":0.375,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":200000,"input":2.5,"output":15}],"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":null},{"id":"google/gemini-2.5-pro","label":"Gemini 2.5 Pro","provider":"Google","category":["flagship","reasoning","fast","long-context"],"inputPrice":1.25,"outputPrice":10,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2025-06-17T14:12:24.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true},"cacheReadPrice":0.125,"cacheWritePrice":0.375,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":200000,"input":2.5,"output":15}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":16.1,"rankIntelligence":255,"totalIntelligence":664,"rankCoding":157,"totalCoding":259,"rankSpeed":92,"totalSpeed":333,"rankInputPrice":289,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","supportedParameters":"include_reasoning,max_tokens,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"qwen/qwen3.8-2.4t-a95b","label":"Qwen3.8 2.4t A95b","provider":"Qwen","category":["long-context"],"inputPrice":2,"outputPrice":6,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-08-12T16:21:42.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"xhigh","default_enabled":true,"supported_efforts":["xhigh","medium","low"]},"cacheReadPrice":0.2,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of Qwen3.8 Max, with 95 billion active parameters out of 2.4 trillion total. It is...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"x-ai/grok-4.3","label":"Grok 4.3","provider":"xAI","category":["creative","long-context"],"inputPrice":1.25,"outputPrice":2.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-04-30T23:30:21.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"low","default_enabled":true,"supported_efforts":["high","medium","low","none"]},"cacheReadPrice":0.2,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":200000,"input":2.5,"output":5}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":24.9,"rankIntelligence":144,"totalIntelligence":664,"rankCoding":136,"totalCoding":259,"rankSpeed":98,"totalSpeed":333,"rankInputPrice":289,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","supportedParameters":"include_reasoning,logprobs,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,temperature,tool_choice,tools,top_logprobs,top_p"},{"id":"google/gemini-3.5-flash","label":"Gemini 3.5 Flash","provider":"Google","category":["fast","long-context"],"inputPrice":1.5,"outputPrice":9,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-05-19T12:30:00.000Z","modality":"text+image+video+audio->text","inputModalities":"text,image,video,audio","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]},"cacheReadPrice":0.15,"cacheWritePrice":0.0833333333333333,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":32.6,"rankIntelligence":84,"totalIntelligence":664,"rankCoding":52,"totalCoding":259,"rankSpeed":28,"totalSpeed":333,"rankInputPrice":322,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"openai/gpt-6-sol","label":"GPT 6 Sol","provider":"OpenAI","category":["long-context"],"inputPrice":2,"outputPrice":10,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1050000,"releasedAt":"2026-09-22T18:12:55.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]},"cacheReadPrice":0.2,"cacheWritePrice":2.5,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":272000,"input":4,"output":15}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":47.5,"rankIntelligence":18,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":184,"totalSpeed":333,"rankInputPrice":333,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"x-ai/grok-4.6","label":"Grok 4.6","provider":"xAI","category":["creative","long-context"],"inputPrice":2,"outputPrice":6,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":500000,"releasedAt":"2026-08-12T15:35:57.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"high","default_enabled":true,"supported_efforts":["xhigh","high","medium","low"]},"cacheReadPrice":0.5,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":200000,"input":4,"output":12}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":44.3,"rankIntelligence":29,"totalIntelligence":664,"rankCoding":12,"totalCoding":259,"rankSpeed":232,"totalSpeed":333,"rankInputPrice":333,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by Grok 4.7.","supportedParameters":"include_reasoning,logprobs,max_tokens,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"x-ai/grok-4.5","label":"Grok 4.5","provider":"xAI","category":["creative","long-context"],"inputPrice":2,"outputPrice":6,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":500000,"releasedAt":"2026-07-08T15:05:54.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"high","default_enabled":true,"supported_efforts":["high","medium","low"]},"cacheReadPrice":0.5,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":200000,"input":4,"output":12}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":38.8,"rankIntelligence":56,"totalIntelligence":664,"rankCoding":37,"totalCoding":259,"rankSpeed":240,"totalSpeed":333,"rankInputPrice":333,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","supportedParameters":"include_reasoning,logprobs,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,temperature,tool_choice,tools,top_logprobs,top_p"},{"id":"anthropic/claude-sonnet-5","label":"Claude Sonnet 5","provider":"Anthropic","category":["flagship","coding","long-context"],"inputPrice":2,"outputPrice":10,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-06-30T18:11:23.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"]},"cacheReadPrice":0.2,"cacheWritePrice":2.5,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":38.2,"rankIntelligence":58,"totalIntelligence":664,"rankCoding":43,"totalCoding":259,"rankSpeed":206,"totalSpeed":333,"rankInputPrice":333,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,tool_choice,tools,verbosity"},{"id":"openai/gpt-5.4","label":"GPT 5.4","provider":"OpenAI","category":["long-context"],"inputPrice":2.5,"outputPrice":15,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1050000,"releasedAt":"2026-03-05T18:12:32.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"]},"cacheReadPrice":0.25,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":9.6,"rankIntelligence":384,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":null,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"anthropic/claude-sonnet-4.5","label":"Claude Sonnet 4.5","provider":"Anthropic","category":["flagship","coding","long-context"],"inputPrice":3,"outputPrice":15,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2025-09-29T16:01:16.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.3,"cacheWritePrice":3.75,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":19.3,"rankIntelligence":218,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":307,"totalSpeed":333,"rankInputPrice":378,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,response_format,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"openai/gpt-5.5","label":"GPT 5.5","provider":"OpenAI","category":["long-context"],"inputPrice":5,"outputPrice":30,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1050000,"releasedAt":"2026-04-24T17:31:33.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["xhigh","high","medium","low","none"]},"cacheReadPrice":0.5,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":38.4,"rankIntelligence":57,"totalIntelligence":664,"rankCoding":28,"totalCoding":259,"rankSpeed":139,"totalSpeed":333,"rankInputPrice":402,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"anthropic/claude-opus-5.5","label":"Claude Opus 5.5","provider":"Anthropic","category":["flagship","coding","creative","long-context"],"inputPrice":4,"outputPrice":20,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-09-22T16:32:12.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"high","supported_efforts":["max","xhigh","high","medium","low"]},"cacheReadPrice":0.2,"cacheWritePrice":5,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":57.6,"rankIntelligence":1,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":389,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,temperature,tool_choice,tools,verbosity"},{"id":"anthropic/claude-opus-4.7","label":"Claude Opus 4.7","provider":"Anthropic","category":["flagship","coding","creative","long-context"],"inputPrice":5,"outputPrice":25,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-04-16T14:51:40.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","default_enabled":false,"supported_efforts":["max","xhigh","high","medium","low"]},"cacheReadPrice":0.5,"cacheWritePrice":6.25,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":40.7,"rankIntelligence":43,"totalIntelligence":664,"rankCoding":34,"totalCoding":259,"rankSpeed":281,"totalSpeed":333,"rankInputPrice":402,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,tool_choice,tools,verbosity"},{"id":"anthropic/claude-opus-4.8","label":"Claude Opus 4.8","provider":"Anthropic","category":["flagship","coding","creative","long-context"],"inputPrice":5,"outputPrice":25,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-05-27T18:04:51.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","default_enabled":false,"supported_efforts":["max","xhigh","high","medium","low"]},"cacheReadPrice":0.5,"cacheWritePrice":6.25,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":41.8,"rankIntelligence":40,"totalIntelligence":664,"rankCoding":31,"totalCoding":259,"rankSpeed":269,"totalSpeed":333,"rankInputPrice":402,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,temperature,tool_choice,tools,verbosity"},{"id":"anthropic/claude-opus-5","label":"Claude Opus 5","provider":"Anthropic","category":["flagship","coding","creative","long-context"],"inputPrice":5,"outputPrice":25,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-07-24T17:02:24.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"]},"cacheReadPrice":0.5,"cacheWritePrice":6.25,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":50.8,"rankIntelligence":11,"totalIntelligence":664,"rankCoding":5,"totalCoding":259,"rankSpeed":255,"totalSpeed":333,"rankInputPrice":402,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,temperature,tool_choice,tools,verbosity"},{"id":"anthropic/claude-opus-4.5","label":"Claude Opus 4.5","provider":"Anthropic","category":["flagship","coding","creative","long-context"],"inputPrice":5,"outputPrice":25,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":200000,"releasedAt":"2025-11-24T18:56:20.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.5,"cacheWritePrice":6.25,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":23.7,"rankIntelligence":160,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":291,"totalSpeed":333,"rankInputPrice":402,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,response_format,stop,structured_outputs,temperature,tool_choice,tools,top_k,verbosity"},{"id":"anthropic/claude-fable-5","label":"Claude Fable 5","provider":"Anthropic","category":["long-context"],"inputPrice":10,"outputPrice":50,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-06-09T12:18:35.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"high","supported_efforts":["max","xhigh","high","medium","low"]},"cacheReadPrice":1,"cacheWritePrice":12.5,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":49.6,"rankIntelligence":13,"totalIntelligence":664,"rankCoding":15,"totalCoding":259,"rankSpeed":217,"totalSpeed":333,"rankInputPrice":422,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,tool_choice,tools,verbosity"},{"id":"anthropic/claude-fable-5.1","label":"Claude Fable 5.1","provider":"Anthropic","category":["long-context"],"inputPrice":10,"outputPrice":50,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-09-01T18:03:58.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"high","supported_efforts":["max","xhigh","high","medium","low"]},"cacheReadPrice":0.25,"cacheWritePrice":12.5,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":53.4,"rankIntelligence":4,"totalIntelligence":664,"rankCoding":1,"totalCoding":259,"rankSpeed":226,"totalSpeed":333,"rankInputPrice":422,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,tool_choice,tools,verbosity"},{"id":"openai/gpt-5.6-luna","label":"GPT 5.6 Luna","provider":"OpenAI","category":["cheap","long-context"],"inputPrice":0.2,"outputPrice":1.2000000000000002,"originalInputPrice":1,"originalOutputPrice":6,"discount":80,"contextLength":1050000,"releasedAt":"2026-07-09T09:54:24.000Z","modality":"text+image+file->text","inputModalities":"file,image,text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]},"cacheReadPrice":0.020000000000000004,"cacheWritePrice":0.25,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":272000,"input":0.4,"output":2.4000000000000004}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":37.3,"rankIntelligence":60,"totalIntelligence":664,"rankCoding":47,"totalCoding":259,"rankSpeed":99,"totalSpeed":333,"rankInputPrice":115,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"openai/gpt-5.6-terra","label":"GPT 5.6 Terra","provider":"OpenAI","category":["long-context"],"inputPrice":2,"outputPrice":12,"originalInputPrice":2.5,"originalOutputPrice":15,"discount":20,"contextLength":1050000,"releasedAt":"2026-07-09T09:54:17.000Z","modality":"text+image+file->text","inputModalities":"file,image,text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]},"cacheReadPrice":0.2,"cacheWritePrice":2.5,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":272000,"input":4,"output":24}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":42.1,"rankIntelligence":39,"totalIntelligence":664,"rankCoding":13,"totalCoding":259,"rankSpeed":147,"totalSpeed":333,"rankInputPrice":333,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"bytedance/seedance-2.5","label":"Seedance 2.5","provider":"ByteDance","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":null,"modality":"text+image+audio+video->video","inputModalities":"text,image,video,audio","outputModalities":"video","videoModes":null,"pricePerImage":null,"billingModel":"second","pricePerMegapixel":null,"pricePerSecondJson":{"withAudio":{"480p":0.1028056,"720p":0.23112,"1080p":0.52},"withoutAudio":{"480p":0.1028056,"720p":0.23112,"1080p":0.52}},"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Seedance 2.5 is a video generation model from ByteDance. It is suited for long-form storytelling, multimodal reference-based generation, video editing, and video extension. It supports first-frame and first-and-last-frame control, up to 50 image, video, and audio reference assets, optional generated audio, and multilingual audiovisual generation.","supportedParameters":"frequency_penalty"},{"id":"x-ai/grok-imagine-image-2.0","label":"Grok Imagine Image 2.0","provider":"xAI","category":["creative","cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":null,"releasedAt":null,"modality":"text+image->image","inputModalities":"text,image","outputModalities":"image","videoModes":null,"pricePerImage":0.01,"billingModel":"image","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":"logprobs,max_tokens,response_format,seed,temperature,top_logprobs,top_p"},{"id":"qwen/qwen3-embedding-8b","label":"Qwen3 Embedding 8b","provider":"Qwen","category":["cheap"],"inputPrice":0.01,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":32768,"releasedAt":"2025-10-28T19:43:42.000Z","modality":"text->embeddings","inputModalities":"text","outputModalities":"embeddings","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","supportedParameters":""},{"id":"qwen/qwen3-embedding-4b","label":"Qwen3 Embedding 4b","provider":"Qwen","category":["cheap"],"inputPrice":0.02,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":32768,"releasedAt":"2025-10-28T14:48:42.000Z","modality":"text->embeddings","inputModalities":"text","outputModalities":"embeddings","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","supportedParameters":""},{"id":"typesafe-ai/jev","label":"Jev","provider":"typesafe-ai","category":["cheap"],"inputPrice":0.042,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":32000,"releasedAt":null,"modality":"text->decisions","inputModalities":"text","outputModalities":"decisions","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":null},{"id":"z-ai/glm-4.7-flash","label":"Glm 4.7 Flash","provider":"Zhipu AI","category":["fast","cheap","long-context"],"inputPrice":0.0605,"outputPrice":0.4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":200000,"releasedAt":"2026-01-19T14:45:13.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":true},"cacheReadPrice":0.01,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":14.9,"rankIntelligence":272,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":230,"totalSpeed":333,"rankInputPrice":48,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3.5-flash-02-23","label":"Qwen3.5 Flash 02 23","provider":"Qwen","category":["fast","cheap","long-context"],"inputPrice":0.065,"outputPrice":0.26,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-02-25T21:09:36.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":0.08125,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,presence_penalty,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"google/gemma-4-26b-a4b-it","label":"Gemma 4 26b A4b It","provider":"Google","category":["cheap","long-context"],"inputPrice":0.09,"outputPrice":0.34,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-04-03T14:53:09.000Z","modality":"text+image+video->text","inputModalities":"image,text,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":false},"cacheReadPrice":0.05,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3-coder-30b-a3b-instruct","label":"Qwen3 Coder 30b A3b Instruct","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.07,"outputPrice":0.28,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2025-07-31T14:32:59.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","supportedParameters":"frequency_penalty,logprobs,max_tokens,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"bytedance-seed/seed-1.6-flash","label":"Seed 1.6 Flash","provider":"bytedance-seed","category":["fast","cheap","long-context"],"inputPrice":0.075,"outputPrice":0.3,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2025-12-23T15:50:11.000Z","modality":"text+image+video->text","inputModalities":"image,text,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,reasoning,response_format,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"google/gemma-4-31b-it","label":"Gemma 4 31b It","provider":"Google","category":["cheap","long-context"],"inputPrice":0.09,"outputPrice":0.34,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-04-02T16:48:06.000Z","modality":"text+image+video->text","inputModalities":"image,text,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":false},"cacheReadPrice":0.05,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"bytedance-seed/seed-2.0-mini","label":"Seed 2.0 Mini","provider":"bytedance-seed","category":["fast","cheap","long-context"],"inputPrice":0.1,"outputPrice":0.4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-02-26T18:38:27.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"qwen/qwen3-vl-32b-instruct","label":"Qwen3 Vl 32b Instruct","provider":"Qwen","category":["cheap"],"inputPrice":0.104,"outputPrice":0.416,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2025-10-23T14:55:32.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":"2026-10-09T00:00:00.000Z","aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","supportedParameters":"frequency_penalty,logprobs,max_tokens,presence_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3-vl-8b-instruct","label":"Qwen3 Vl 8b Instruct","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.117,"outputPrice":0.455,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2025-10-14T17:35:08.000Z","modality":"text+image->text","inputModalities":"image,text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":"2026-10-09T00:00:00.000Z","aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3-coder-next","label":"Qwen3 Coder Next","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.12,"outputPrice":0.8,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-02-04T00:15:01.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":0.07,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"z-ai/glm-4.5-air","label":"Glm 4.5 Air","provider":"Zhipu AI","category":["cheap"],"inputPrice":0.13,"outputPrice":0.85,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2025-07-25T19:20:58.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.025,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":11.1,"rankIntelligence":351,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":322,"totalSpeed":333,"rankInputPrice":108,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,presence_penalty,reasoning,repetition_penalty,seed,stop,temperature,tool_choice,tools,top_k,top_p"},{"id":"deepseek/deepseek-v4-flash-0731","label":"Deepseek V4 Flash 0731","provider":"DeepSeek","category":["fast","cheap","long-context"],"inputPrice":0.44,"outputPrice":1.32,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2026-07-31T06:21:48.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","high","low"]},"cacheReadPrice":0.044,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,parallel_tool_calls,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_a,top_k,top_logprobs,top_p"},{"id":"xiaomi/mimo-v2.5","label":"Mimo V2.5","provider":"xiaomi","category":["cheap","long-context"],"inputPrice":0.14,"outputPrice":0.28,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1050000,"releasedAt":"2026-04-22T16:11:09.000Z","modality":"text+image+audio+video->text","inputModalities":"text,audio,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.0028,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":26,"rankIntelligence":131,"totalIntelligence":664,"rankCoding":79,"totalCoding":259,"rankSpeed":293,"totalSpeed":333,"rankInputPrice":209,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"qwen/qwen3-vl-30b-a3b-instruct","label":"Qwen3 Vl 30b A3b Instruct","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.15,"outputPrice":0.6,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2025-10-06T23:47:56.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,min_p,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"google/gemini-embedding-001","label":"Gemini Embedding","provider":"Google","category":["fast","cheap"],"inputPrice":0.15,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":20000,"releasedAt":"2025-10-31T20:43:30.000Z","modality":"text->embeddings","inputModalities":"text","outputModalities":"embeddings","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"gemini-embedding-001 provides a unified cutting edge experience across domains, including science, legal, finance, and coding. This embedding model has consistently held a top spot on the Massive Text Embedding Benchmark...","supportedParameters":""},{"id":"z-ai/glm-5.3-flash","label":"Glm 5.3 Flash","provider":"Zhipu AI","category":["fast","cheap","long-context"],"inputPrice":0.15,"outputPrice":0.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1310720,"releasedAt":"2026-08-26T13:59:01.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"max","default_enabled":true,"supported_efforts":["max","high","low"]},"cacheReadPrice":0.03,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":41.8,"rankIntelligence":40,"totalIntelligence":664,"rankCoding":43,"totalCoding":259,"rankSpeed":300,"totalSpeed":333,"rankInputPrice":90,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,parallel_tool_calls,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3.6-35b-a3b","label":"Qwen3.6 35b A3b","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.1,"outputPrice":0.9,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-04-27T03:24:15.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":true},"cacheReadPrice":0.05,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"google/gemini-embedding-2-preview","label":"Gemini Embedding 2 Preview","provider":"Google","category":["fast","cheap"],"inputPrice":0.2,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":8192,"releasedAt":"2026-04-17T14:34:25.000Z","modality":"text+image+file+audio+video->embeddings","inputModalities":"text,image,file,audio,video","outputModalities":"embeddings","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini Embedding 2 Preview is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It...","supportedParameters":""},{"id":"anthropic/claude-3-haiku","label":"Claude 3 Haiku","provider":"Anthropic","category":["fast","cheap","long-context"],"inputPrice":0.25,"outputPrice":1.25,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":200000,"releasedAt":"2024-03-13T00:00:00.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":0.03,"cacheWritePrice":0.3,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":"2026-09-26T00:00:00.000Z","aa":{"intelligenceIndex":5.6,"rankIntelligence":591,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":143,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","supportedParameters":"max_tokens,stop,temperature,tool_choice,tools,top_k,top_p"},{"id":"bytedance-seed/seed-1.6","label":"Seed 1.6","provider":"bytedance-seed","category":["cheap","long-context"],"inputPrice":0.25,"outputPrice":2,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2025-12-23T15:49:57.000Z","modality":"text+image+video->text","inputModalities":"image,text,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,reasoning,response_format,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"bytedance-seed/seed-2.0-lite","label":"Seed 2.0 Lite","provider":"bytedance-seed","category":["fast","cheap","long-context"],"inputPrice":0.25,"outputPrice":2,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-03-10T15:40:31.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"qwen/qwen3.5-plus-02-15","label":"Qwen3.5 Plus 02 15","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.26,"outputPrice":1.56,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-02-16T08:10:16.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":0.325,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","supportedParameters":"frequency_penalty,include_reasoning,logprobs,max_tokens,presence_penalty,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"deepseek/deepseek-v3.2-exp","label":"Deepseek V3.2 Exp","provider":"DeepSeek","category":["coding","cheap"],"inputPrice":0.27,"outputPrice":0.41,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":163840,"releasedAt":"2025-09-29T12:54:41.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":"2026-09-28T00:00:00.000Z","aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"deepseek/deepseek-v3.1-terminus","label":"Deepseek V3.1 Terminus","provider":"DeepSeek","category":["coding","cheap"],"inputPrice":0.27,"outputPrice":1,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":163840,"releasedAt":"2025-09-22T13:37:55.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.135,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":"2026-09-28T00:00:00.000Z","aa":{"intelligenceIndex":13.9,"rankIntelligence":288,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":157,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek-V3.1 Terminus is an update to DeepSeek V3.1 that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"qwen/qwen3.6-27b","label":"Qwen3.6 27b","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.289,"outputPrice":2.4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2026-04-27T01:57:44.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":true},"cacheReadPrice":0.15,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3-coder","label":"Qwen3 Coder","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.3,"outputPrice":1,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2025-07-23T00:29:06.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":0.1,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,min_p,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"qwen/qwen3.5-plus-20260420","label":"Qwen3.5 Plus 20260420","provider":"Qwen","category":["cheap","long-context"],"inputPrice":0.3,"outputPrice":1.8,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1000000,"releasedAt":"2026-04-27T03:42:48.000Z","modality":"text+image+video->text","inputModalities":"text,image,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":null,"cacheWritePrice":0.375,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","supportedParameters":"frequency_penalty,include_reasoning,logprobs,max_tokens,presence_penalty,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"google/gemini-2.5-flash-image","label":"Gemini 2.5 Flash Image","provider":"Google","category":["fast","cheap"],"inputPrice":0.3,"outputPrice":2.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":32768,"releasedAt":"2025-10-07T20:53:51.000Z","modality":"text+image->text+image","inputModalities":"image,text","outputModalities":"image,text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","supportedParameters":"max_tokens,response_format,seed,stop,structured_outputs,temperature,top_p"},{"id":"z-ai/glm-4.6v","label":"Glm 4.6v","provider":"Zhipu AI","category":["cheap"],"inputPrice":0.3,"outputPrice":0.9,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2025-12-08T15:24:22.000Z","modality":"text+image+video->text","inputModalities":"image,text,video","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.055,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":8.4,"rankIntelligence":432,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":223,"totalSpeed":333,"rankInputPrice":163,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,temperature,tool_choice,tools,top_k,top_p"},{"id":"qwen/qwen-2.5-72b-instruct","label":"Qwen 2.5 72b Instruct","provider":"Qwen","category":["flagship","cheap"],"inputPrice":0.36,"outputPrice":0.4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":32768,"releasedAt":"2024-09-19T00:00:00.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","supportedParameters":"frequency_penalty,logit_bias,max_tokens,min_p,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"deepseek/deepseek-v4-pro-0813","label":"Deepseek V4 Pro 0813","provider":"DeepSeek","category":["long-context"],"inputPrice":1.32,"outputPrice":3.96,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2026-08-12T15:42:44.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"high","supported_efforts":["max","high","low"]},"cacheReadPrice":0.132,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":36,"rankIntelligence":64,"totalIntelligence":664,"rankCoding":56,"totalCoding":259,"rankSpeed":168,"totalSpeed":333,"rankInputPrice":315,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"xiaomi/mimo-v2.5-pro","label":"Mimo V2.5 Pro","provider":"xiaomi","category":["cheap","long-context"],"inputPrice":0.435,"outputPrice":0.87,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1050000,"releasedAt":"2026-04-22T16:11:13.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.0036,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":26,"rankIntelligence":131,"totalIntelligence":664,"rankCoding":79,"totalCoding":259,"rankSpeed":293,"totalSpeed":333,"rankInputPrice":209,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"deepseek/deepseek-r1-0528","label":"Deepseek R1 0528","provider":"DeepSeek","category":["reasoning","cheap"],"inputPrice":0.5,"outputPrice":2.15,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":163840,"releasedAt":"2025-05-28T17:59:30.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true},"cacheReadPrice":0.35,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"May 28th update to the original DeepSeek R1 Performance on par with OpenAI o1, but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"google/gemini-3.1-flash-image","label":"Gemini 3.1 Flash Image","provider":"Google","category":["fast","cheap"],"inputPrice":0.5,"outputPrice":3,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2026-06-18T03:41:05.000Z","modality":"text+image->text+image","inputModalities":"image,text","outputModalities":"image,text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","minimal"]},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,temperature,top_p"},{"id":"google/gemini-3.1-flash-image-preview","label":"Gemini 3.1 Flash Image Preview","provider":"Google","category":["fast","cheap"],"inputPrice":0.5,"outputPrice":3,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":65536,"releasedAt":"2026-02-26T15:25:58.000Z","modality":"text+image->text+image","inputModalities":"image,text","outputModalities":"image,text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","minimal"]},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","supportedParameters":"include_reasoning,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,temperature,top_p"},{"id":"z-ai/glm-4.5v","label":"Glm 4.5v","provider":"Zhipu AI","category":["flagship"],"inputPrice":0.6,"outputPrice":1.8,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":65536,"releasedAt":"2025-08-11T14:24:48.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.11,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":6.7,"rankIntelligence":522,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":302,"totalSpeed":333,"rankInputPrice":237,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","supportedParameters":"frequency_penalty,include_reasoning,max_tokens,presence_penalty,reasoning,repetition_penalty,response_format,seed,stop,temperature,tool_choice,tools,top_k,top_p"},{"id":"moonshotai/kimi-k2-0905","label":"Kimi K2 0905","provider":"Moonshot AI","category":["long-context"],"inputPrice":0.6,"outputPrice":2.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":262144,"releasedAt":"2025-09-04T21:25:47.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","supportedParameters":"frequency_penalty,max_tokens,presence_penalty,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_p"},{"id":"z-ai/glm-4.5","label":"Glm 4.5","provider":"Zhipu AI","category":["flagship"],"inputPrice":0.6,"outputPrice":2.2,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2025-07-25T19:22:27.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.11,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":12.8,"rankIntelligence":310,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":null,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","supportedParameters":"include_reasoning,max_tokens,reasoning,response_format,temperature,tool_choice,tools,top_k,top_p"},{"id":"x-ai/grok-4.7","label":"Grok 4.7","provider":"xAI","category":["creative","long-context"],"inputPrice":1.6,"outputPrice":4.8,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":500000,"releasedAt":"2026-09-21T16:19:01.000Z","modality":"text+image->text","inputModalities":"text,image","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"high","default_enabled":true,"supported_efforts":["xhigh","high","medium","low"]},"cacheReadPrice":0.4,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":200000,"input":3.2,"output":9.6}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":46.4,"rankIntelligence":21,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":288,"totalSpeed":333,"rankInputPrice":333,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","supportedParameters":"include_reasoning,logprobs,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,temperature,tool_choice,tools,top_logprobs,top_p"},{"id":"x-ai/grok-4.20-multi-agent","label":"Grok 4.20 Multi Agent","provider":"xAI","category":["creative","long-context"],"inputPrice":1.25,"outputPrice":2.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":2000000,"releasedAt":"2026-03-31T17:45:58.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["xhigh","high","medium","low"]},"cacheReadPrice":0.2,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","supportedParameters":"include_reasoning,logprobs,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,temperature,top_logprobs,top_p"},{"id":"x-ai/grok-4.20","label":"Grok 4.20","provider":"xAI","category":["creative","long-context"],"inputPrice":1.25,"outputPrice":2.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":2000000,"releasedAt":"2026-03-31T17:43:39.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_enabled":false},"cacheReadPrice":0.2,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":25.7,"rankIntelligence":135,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":126,"totalSpeed":333,"rankInputPrice":289,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","supportedParameters":"include_reasoning,logprobs,max_tokens,reasoning,response_format,seed,structured_outputs,temperature,tool_choice,tools,top_logprobs,top_p"},{"id":"google/gemini-2.5-pro-preview","label":"Gemini 2.5 Pro Preview","provider":"Google","category":["flagship","reasoning","fast","long-context"],"inputPrice":1.25,"outputPrice":10,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":"2025-06-05T15:27:37.000Z","modality":"text+image+file+audio->text","inputModalities":"file,image,text,audio","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true},"cacheReadPrice":0.125,"cacheWritePrice":0.375,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":200000,"input":2.5,"output":15}],"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","supportedParameters":"include_reasoning,max_tokens,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"google/gemini-3-pro-image","label":"Gemini 3 Pro Image","provider":"Google","category":["fast"],"inputPrice":2,"outputPrice":12,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2026-06-18T03:40:54.000Z","modality":"text+image->text+image","inputModalities":"image,text","outputModalities":"image,text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","supportedParameters":"include_reasoning,max_tokens,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"anthropic/claude-sonnet-4","label":"Claude Sonnet 4","provider":"Anthropic","category":["flagship","coding","long-context"],"inputPrice":3,"outputPrice":15,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":200000,"releasedAt":"2025-05-22T16:12:51.000Z","modality":"text+image+file->text","inputModalities":"image,text,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":0.3,"cacheWritePrice":3.75,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":16.6,"rankIntelligence":249,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":null,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","supportedParameters":"include_reasoning,max_tokens,reasoning,stop,temperature,tool_choice,tools,top_k,top_p"},{"id":"anthropic/claude-opus-4.1","label":"Claude Opus 4.1","provider":"Anthropic","category":["flagship","coding","creative","long-context"],"inputPrice":15,"outputPrice":75,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":200000,"releasedAt":"2025-08-05T16:33:11.000Z","modality":"text+image+file->text","inputModalities":"image,text,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false},"cacheReadPrice":1.5,"cacheWritePrice":18.75,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":18.6,"rankIntelligence":225,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":434,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","supportedParameters":"include_reasoning,max_tokens,reasoning,stop,temperature,tool_choice,tools,top_k,top_p"},{"id":"bytedance/seedance-1-5-pro","label":"Seedance 1 5 Pro","provider":"ByteDance","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":"2026-03-23T14:53:28.339Z","modality":"text+image->video","inputModalities":"text,image","outputModalities":"video","videoModes":"t2v,i2v","pricePerImage":null,"billingModel":"second","pricePerMegapixel":null,"pricePerSecondJson":{"withAudio":{"480p":0.0230592,"720p":0.05184,"1080p":0.11664},"withoutAudio":{"480p":0.0115296,"720p":0.02592,"1080p":0.05832}},"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"ByteDance's next-generation audio-visual generation model with a 4.5B parameter Dual-Branch Diffusion Transformer architecture. Seedance 1.5 Pro generates video and audio simultaneously in a single unified pass — eliminating the timing issues of sequential audio dubbing. Supports multi-language lip-sync (English, Mandarin, Japanese, Korean, Spanish, and more), cinematic camera control (pan, tilt, zoom, orbit), multi-character dialogue, and character consistency across shots. Produces clips from 4–12 seconds at up to 1080p. The number of tokens is given by (height of output video * width of output video * duration * 24) / 1024","supportedParameters":"frequency_penalty"},{"id":"google/veo-3.1","label":"Veo 3.1","provider":"Google","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":"2026-03-23T14:45:48.494Z","modality":"text+image->video","inputModalities":"text,image","outputModalities":"video","videoModes":"t2v,i2v","pricePerImage":null,"billingModel":"second","pricePerMegapixel":null,"pricePerSecondJson":{"withAudio":{"4k":0.6,"1080p":0.4},"withoutAudio":{"4k":0.4,"1080p":0.2}},"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Google's state-of-the-art video generation model, built for maximum visual fidelity in final production cuts. Veo 3.1 generates high-quality 1080p video from text or image prompts with native synchronized audio — including dialogue, ambient effects, and background sound. Supports scene extension (up to 20 chained clips for 140+ second narratives), frames-to-video transitions between two images, vertical video for Shorts, and 4K upscaling.","supportedParameters":"max_tokens,response_format,seed,temperature,top_p"},{"id":"minimax/hailuo-3-max","label":"Hailuo 3 Max","provider":"MiniMax","category":["fast","cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":"2026-09-02T01:41:50.000Z","modality":"text+image->video","inputModalities":"text,image","outputModalities":"video","videoModes":"t2v,i2v","pricePerImage":null,"billingModel":"second","pricePerMegapixel":null,"pricePerSecondJson":{"withAudio":{"480p":0.05,"720p":0.08},"withoutAudio":{"480p":0.05,"720p":0.08}},"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"MiniMax H3 Max is a video-generation model from MiniMax, jointly released with fal.ai. Derived through additional training from MiniMax H3, it is designed for faster text-to-video and image-to-video generation with...","supportedParameters":"max_tokens,temperature,top_p"},{"id":"qwen/qwen-image-3","label":"Qwen Image 3","provider":"Qwen","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":null,"releasedAt":null,"modality":"text+image->image","inputModalities":"text,image","outputModalities":"image","videoModes":null,"pricePerImage":0.03,"billingModel":"image","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":"max_tokens,presence_penalty,response_format,seed,temperature,top_p"},{"id":"bytedance/seedance-2.0","label":"Seedance 2.0","provider":"ByteDance","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":"2026-04-15T00:02:42.904Z","modality":"text+image+audio+video->video","inputModalities":"text,image,video,audio","outputModalities":"video","videoModes":"t2v,i2v","pricePerImage":null,"billingModel":"second","pricePerMegapixel":null,"pricePerSecondJson":{"withAudio":{"480p":0.067256,"720p":0.1512,"1080p":0.3402},"withoutAudio":{"480p":0.067256,"720p":0.1512,"1080p":0.3402}},"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Seedance 2.0 is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It is particularly strong at preserving character consistency, visual style, and camera movement from reference material. The number of tokens is given by (height of output video * width of output video * duration * 24) / 1024","supportedParameters":"frequency_penalty"},{"id":"openai/sora-2-pro","label":"Sora 2 Pro","provider":"OpenAI","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":"2026-03-23T14:52:01.486Z","modality":"text+image->video","inputModalities":"text,image","outputModalities":"video","videoModes":"t2v,i2v","pricePerImage":null,"billingModel":"second","pricePerMegapixel":null,"pricePerSecondJson":{"withAudio":{"720p":0.3,"1024p":0.5,"1080p":0.5}},"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"OpenAI's flagship video generation model, delivering production-quality video with physics-accurate motion, synchronized audio, and world-state persistence across shots. Sora 2 Pro follows intricate multi-shot instructions while maintaining consistent spatial relationships — objects don't disappear or change shape between cuts. Supports text-to-video and image-to-video, with synchronized background soundscapes, speech, and sound effects. Includes advanced content safety with C2PA metadata provenance and SynthID-style watermarking.","supportedParameters":"frequency_penalty,logit_bias,logprobs,presence_penalty,stop,top_logprobs"},{"id":"qwen/qwen-image-3-pro","label":"Qwen Image 3 Pro","provider":"Qwen","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":null,"releasedAt":null,"modality":"text+image->image","inputModalities":"text,image","outputModalities":"image","videoModes":null,"pricePerImage":0.075,"billingModel":"image","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":"max_tokens,presence_penalty,response_format,seed,temperature,top_p"},{"id":"bytedance/seedance-2.0-fast","label":"Seedance 2.0 Fast","provider":"ByteDance","category":["cheap"],"inputPrice":0,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":0,"releasedAt":"2026-04-15T00:02:42.904Z","modality":"text+image+audio+video->video","inputModalities":"text,image,video,audio","outputModalities":"video","videoModes":"t2v,i2v","pricePerImage":null,"billingModel":"second","pricePerMegapixel":null,"pricePerSecondJson":{"withAudio":{"480p":0.0538048,"720p":0.12096},"withoutAudio":{"480p":0.0538048,"720p":0.12096}},"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"Seedance 2.0 Fast is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It prioritizes generation speed and lower cost over maximum output quality. The number of tokens is given by (height of output video * width of output video * duration * 24) / 1024","supportedParameters":"frequency_penalty"},{"id":"openai/text-embedding-3-small","label":"Text Embedding 3 Small","provider":"OpenAI","category":["fast","cheap"],"inputPrice":0.021,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":8192,"releasedAt":"2025-10-30T20:50:55.000Z","modality":"text->embeddings","inputModalities":"text","outputModalities":"embeddings","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...","supportedParameters":""},{"id":"openai/gpt-oss-20b","label":"GPT Oss 20b","provider":"OpenAI","category":["cheap"],"inputPrice":0.03,"outputPrice":0.13,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2025-08-05T17:17:09.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]},"cacheReadPrice":0.03,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":9,"rankIntelligence":404,"totalIntelligence":664,"rankCoding":200,"totalCoding":259,"rankSpeed":63,"totalSpeed":333,"rankInputPrice":48,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_k,top_logprobs,top_p"},{"id":"openai/gpt-oss-safeguard-20b","label":"GPT Oss Safeguard 20b","provider":"OpenAI","category":["cheap"],"inputPrice":0.075,"outputPrice":0.3,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2025-10-29T15:47:16.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true},"cacheReadPrice":0.0375,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","supportedParameters":"include_reasoning,max_tokens,reasoning,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_p"},{"id":"openai/text-embedding-ada-002","label":"Text Embedding Ada 002","provider":"OpenAI","category":["cheap"],"inputPrice":0.1,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":8192,"releasedAt":"2025-10-30T23:09:58.000Z","modality":"text->embeddings","inputModalities":"text","outputModalities":"embeddings","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"text-embedding-ada-002 is OpenAI's legacy text embedding model.","supportedParameters":""},{"id":"openai/text-embedding-3-large","label":"Text Embedding 3 Large","provider":"OpenAI","category":["cheap"],"inputPrice":0.13,"outputPrice":0,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":8192,"releasedAt":"2025-10-30T22:21:06.000Z","modality":"text->embeddings","inputModalities":"text","outputModalities":"embeddings","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...","supportedParameters":""},{"id":"openai/gpt-oss-120b","label":"GPT Oss 120b","provider":"OpenAI","category":["cheap"],"inputPrice":0.15,"outputPrice":0.6,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":131072,"releasedAt":"2025-08-05T17:17:11.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":11.6,"rankIntelligence":339,"totalIntelligence":664,"rankCoding":164,"totalCoding":259,"rankSpeed":46,"totalSpeed":333,"rankInputPrice":90,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,min_p,presence_penalty,reasoning,reasoning_effort,repetition_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_a,top_k,top_logprobs,top_p"},{"id":"openai/gpt-5.6-luna-pro","label":"GPT 5.6 Luna Pro","provider":"OpenAI","category":["cheap","long-context"],"inputPrice":0.2,"outputPrice":1.2000000000000002,"originalInputPrice":1,"originalOutputPrice":6,"discount":80,"contextLength":1050000,"releasedAt":"2026-07-09T09:54:27.000Z","modality":"text+image+file->text","inputModalities":"file,image,text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]},"cacheReadPrice":0.020000000000000004,"cacheWritePrice":0.25,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":272000,"input":0.4,"output":2.4000000000000004}],"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.6 Luna Pro is the same underlying model as GPT-5.6 Luna, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"openai/gpt-3.5-turbo","label":"GPT 3.5 Turbo","provider":"OpenAI","category":["cheap"],"inputPrice":0.5,"outputPrice":1.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":16385,"releasedAt":"2023-05-28T00:00:00.000Z","modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":null,"rankIntelligence":null,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":null,"totalSpeed":333,"rankInputPrice":null,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,presence_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_logprobs,top_p"},{"id":"openai/gpt-5.6-sol-pro","label":"GPT 5.6 Sol Pro","provider":"OpenAI","category":["long-context"],"inputPrice":4,"outputPrice":20,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1050000,"releasedAt":"2026-07-09T09:54:14.000Z","modality":"text+image+file->text","inputModalities":"file,image,text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]},"cacheReadPrice":0.4,"cacheWritePrice":5,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":272000,"input":8,"output":30}],"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.6 Sol Pro is the same underlying model as GPT-5.6 Sol, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"openai/gpt-5.6-sol","label":"GPT 5.6 Sol","provider":"OpenAI","category":["long-context"],"inputPrice":4,"outputPrice":20,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1050000,"releasedAt":"2026-07-09T09:54:10.000Z","modality":"text+image+file->text","inputModalities":"file,image,text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":"2026-02-16","maxCompletionTokens":128000,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]},"cacheReadPrice":0.4,"cacheWritePrice":5,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":272000,"input":8,"output":30}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":47,"rankIntelligence":19,"totalIntelligence":664,"rankCoding":6,"totalCoding":259,"rankSpeed":204,"totalSpeed":333,"rankInputPrice":389,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"openai/gpt-5.6-terra-pro","label":"GPT 5.6 Terra Pro","provider":"OpenAI","category":["long-context"],"inputPrice":2,"outputPrice":12,"originalInputPrice":2.5,"originalOutputPrice":15,"discount":20,"contextLength":1050000,"releasedAt":"2026-07-09T09:54:21.000Z","modality":"text+image+file->text","inputModalities":"file,image,text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]},"cacheReadPrice":0.2,"cacheWritePrice":2.5,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":272000,"input":4,"output":24}],"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.6 Terra Pro is the same underlying model as GPT-5.6 Terra, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"openai/gpt-image-1-mini","label":"GPT Image 1 Mini","provider":"OpenAI","category":["fast","long-context"],"inputPrice":2.5,"outputPrice":2.5,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":400000,"releasedAt":"2026-06-24T00:00:00.000Z","modality":"text+image->image","inputModalities":"text,image","outputModalities":"image","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":8,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,presence_penalty,response_format,seed,stop,structured_outputs,temperature,top_logprobs,top_p"},{"id":"openai/gpt-4o","label":"GPT 4o","provider":"OpenAI","category":["flagship"],"inputPrice":2.5,"outputPrice":10,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":128000,"releasedAt":"2024-05-13T00:00:00.000Z","modality":"text+image+file->text","inputModalities":"text,image,file","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":1.25,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":8.4,"rankIntelligence":432,"totalIntelligence":664,"rankCoding":null,"totalCoding":259,"rankSpeed":150,"totalSpeed":333,"rankInputPrice":371,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of GPT-4 Turbo while being twice as...","supportedParameters":"frequency_penalty,logit_bias,logprobs,max_completion_tokens,max_tokens,prediction,presence_penalty,response_format,seed,stop,structured_outputs,temperature,tool_choice,tools,top_logprobs,top_p,web_search_options"},{"id":"openai/gpt-image-2.5-sunburst","label":"GPT Image 2.5 Sunburst","provider":"OpenAI","category":["long-context"],"inputPrice":8,"outputPrice":8,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":400000,"releasedAt":null,"modality":"text+image->image","inputModalities":"text,image","outputModalities":"image","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":30,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,presence_penalty,response_format,seed,stop,structured_outputs,temperature,top_logprobs,top_p"},{"id":"openai/gpt-image-2.5-flare","label":"GPT Image 2.5 Flare","provider":"OpenAI","category":["long-context"],"inputPrice":8,"outputPrice":8,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":400000,"releasedAt":null,"modality":"text+image->image","inputModalities":"text,image","outputModalities":"image","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":30,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,presence_penalty,response_format,seed,stop,structured_outputs,temperature,top_logprobs,top_p"},{"id":"openai/gpt-5.4-image-2","label":"GPT 5.4 Image 2","provider":"OpenAI","category":["long-context"],"inputPrice":8,"outputPrice":15,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":272000,"releasedAt":"2026-04-21T18:52:08.000Z","modality":"text+image+file->text+image","inputModalities":"image,text,file","outputModalities":"image,text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":30,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":false,"default_effort":"medium","default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"]},"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-5.4 Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","supportedParameters":"frequency_penalty,include_reasoning,logit_bias,logprobs,max_tokens,presence_penalty,reasoning,reasoning_effort,response_format,seed,stop,structured_outputs,top_logprobs"},{"id":"openai/gpt-6-astra","label":"GPT 6 Astra","provider":"OpenAI","category":["long-context"],"inputPrice":10,"outputPrice":50,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1050000,"releasedAt":"2026-09-04T20:13:58.000Z","modality":"text+image+file->text","inputModalities":"file,image,text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":null,"pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"]},"cacheReadPrice":1,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":[{"abovePromptTokens":272000,"input":20,"output":75}],"effectiveDeprecationDate":null,"aa":{"intelligenceIndex":52.7,"rankIntelligence":6,"totalIntelligence":664,"rankCoding":11,"totalCoding":259,"rankSpeed":256,"totalSpeed":333,"rankInputPrice":422,"totalInputPrice":444},"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","supportedParameters":"include_reasoning,max_completion_tokens,max_tokens,reasoning,reasoning_effort,response_format,seed,structured_outputs,tool_choice,tools"},{"id":"openai/gpt-image-1","label":"GPT Image 1","provider":"OpenAI","category":["long-context"],"inputPrice":10,"outputPrice":10,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":400000,"releasedAt":"2026-06-24T00:00:00.000Z","modality":"text+image->image","inputModalities":"text,image","outputModalities":"image","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":40,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":null,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":false,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":"frequency_penalty,logit_bias,logprobs,max_tokens,presence_penalty,response_format,seed,stop,structured_outputs,temperature,top_logprobs,top_p"},{"id":"deepseek/deepseek-v4-flash-0731free","label":"Deepseek V4 Flash 0731free","provider":"DeepSeek","category":["fast","cheap","long-context"],"inputPrice":0.2,"outputPrice":0.4,"originalInputPrice":null,"originalOutputPrice":null,"discount":null,"contextLength":1048576,"releasedAt":null,"modality":"text->text","inputModalities":"text","outputModalities":"text","videoModes":null,"pricePerImage":null,"billingModel":"token","pricePerMegapixel":null,"pricePerSecondJson":null,"pricePerImageTiersJson":null,"pricePerCharacter":null,"pricePerSearch":null,"pricePerRequest":null,"pricePerImageToken":null,"knowledgeCutoff":null,"maxCompletionTokens":null,"reasoningMeta":null,"cacheReadPrice":0.03,"cacheWritePrice":null,"reasoningOutputPrice":null,"tokenPriceTiers":null,"effectiveDeprecationDate":null,"aa":null,"isFree":true,"freeRpmLimit":null,"freeDailyLimit":null,"description":null,"supportedParameters":null}]}