{"object":"list","data":[{"id":"qwen3.8-max","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.8 Max","description":"Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...","context_length":1000000,"aliases":["qwen/qwen3.8-max"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"xhigh","default_enabled":true,"supported_efforts":["xhigh","high","medium","low","minimal"]}},{"id":"deepseek-v3.2","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek V3.2","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"aliases":["deepseek/deepseek-v3.2"],"pricing":{"prompt":"2.8e-7","completion":"4.2e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"deepseek-v4-flash","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek V4 Flash","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"aliases":["deepseek/deepseek-v4-flash"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","supported_efforts":["xhigh","high"]}},{"id":"minimax-m2.7","object":"model","created":1787387470,"owned_by":"minimax","name":"Minimax M2.7","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","context_length":204800,"aliases":["minimax/minimax-m2.7"],"pricing":{"prompt":"3e-7","completion":"0.0000012"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"glm-5.2","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 5.2","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":1048576,"aliases":["z-ai/glm-5.2"],"pricing":{"prompt":"0.0000014","completion":"0.0000044"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["xhigh","high"]}},{"id":"glm-5","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 5","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","context_length":204800,"aliases":["z-ai/glm-5"],"pricing":{"prompt":"0.000001","completion":"0.0000032000000000000003"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"glm-5.1","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 5.1","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","context_length":204800,"aliases":["z-ai/glm-5.1"],"pricing":{"prompt":"0.0000014","completion":"0.0000044"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"kimi-k2.7-code","object":"model","created":1787387470,"owned_by":"moonshotai","name":"Kimi K2.7 Code","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"aliases":["moonshotai/kimi-k2.7-code"],"pricing":{"prompt":"9.499999999999999e-7","completion":"0.000004"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_enabled":true}},{"id":"glm-5v-turbo","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 5v Turbo","description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","context_length":202752,"aliases":["z-ai/glm-5v-turbo"],"pricing":{"prompt":"0.0000012","completion":"0.000004"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"glm-5-turbo","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 5 Turbo","description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","context_length":202752,"aliases":["z-ai/glm-5-turbo"],"pricing":{"prompt":"0.0000012","completion":"0.000004"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"deepseek-v4-pro","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek V4 Pro","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1048576,"aliases":["deepseek/deepseek-v4-pro"],"pricing":{"prompt":"0.0000024","completion":"0.0000048"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","supported_efforts":["xhigh","high"]}},{"id":"glm-4.7","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 4.7","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","context_length":204800,"aliases":["z-ai/glm-4.7"],"pricing":{"prompt":"4.0000000000000003e-7","completion":"0.00000175"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"kimi-k3","object":"model","created":1787387470,"owned_by":"moonshotai","name":"Kimi K3","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"aliases":["moonshotai/kimi-k3"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"max","default_enabled":true,"supported_efforts":["max","high","low"]}},{"id":"minimax-m2.5","object":"model","created":1787387470,"owned_by":"minimax","name":"Minimax M2.5","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","context_length":204800,"aliases":["minimax/minimax-m2.5"],"pricing":{"prompt":"3e-7","completion":"0.0000012"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"minimax-m3","object":"model","created":1787387470,"owned_by":"minimax","name":"Minimax M3","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":1048576,"aliases":["minimax/minimax-m3"],"pricing":{"prompt":"3e-7","completion":"0.0000012"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"kimi-k2.5","object":"model","created":1787387470,"owned_by":"moonshotai","name":"Kimi K2.5","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","context_length":262144,"aliases":["moonshotai/kimi-k2.5"],"pricing":{"prompt":"6e-7","completion":"0.000003"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"kimi-k2.6","object":"model","created":1787387470,"owned_by":"moonshotai","name":"Kimi K2.6","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","context_length":262144,"aliases":["moonshotai/kimi-k2.6"],"pricing":{"prompt":"8.58e-7","completion":"0.000003566"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"gemini-2.5-flash-lite","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 2.5 Flash Lite","description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"aliases":["google/gemini-2.5-flash-lite"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3.7-max","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.7 Max","description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","context_length":1000000,"aliases":["qwen/qwen3.7-max"],"pricing":{"prompt":"0.000001475","completion":"0.000004425"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"qwen3.7-plus","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.7 Plus","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","context_length":1000000,"aliases":["qwen/qwen3.7-plus"],"pricing":{"prompt":"3.2e-7","completion":"0.00000128"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"gpt-5.3-codex","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.3 Codex","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","context_length":400000,"aliases":["openai/gpt-5.3-codex"],"pricing":{"prompt":"0.00000175","completion":"0.000014"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"gemini-3.1-flash-lite-preview","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3.1 Flash Lite Preview","description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","context_length":1048576,"aliases":["google/gemini-3.1-flash-lite-preview"],"pricing":{"prompt":"2.5e-7","completion":"0.0000015"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]}},{"id":"claude-sonnet-4.6","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Sonnet 4.6","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"aliases":["anthropic/claude-sonnet-4.6"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["max","high","medium","low"]}},{"id":"claude-opus-4.6","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Opus 4.6","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"aliases":["anthropic/claude-opus-4.6"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":false,"supported_efforts":["max","high","medium","low"],"supports_max_tokens":true}},{"id":"gemini-3.1-pro-preview","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3.1 Pro Preview","description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"aliases":["google/gemini-3.1-pro-preview"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]}},{"id":"sonar-pro-search","object":"model","created":1787387470,"owned_by":"perplexity","name":"Sonar Pro Search","description":"Exclusively available on the BazaarLink API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","context_length":200000,"aliases":["perplexity/sonar-pro-search"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"seed-1.6","object":"model","created":1787387470,"owned_by":"bytedance-seed","name":"Seed 1.6","description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","context_length":262144,"aliases":["bytedance-seed/seed-1.6"],"pricing":{"prompt":"2.5e-7","completion":"0.000002"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"deepseek-chat-v3-0324","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek Chat V3 0324","description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the DeepSeek V3 model and performs really well...","context_length":163840,"aliases":["deepseek/deepseek-chat-v3-0324"],"pricing":{"prompt":"2.5e-7","completion":"0.000001"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"gemma-4-31b-it","object":"model","created":1787387470,"owned_by":"google","name":"Gemma 4 31b It","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"aliases":["google/gemma-4-31b-it"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"3.4000000000000003e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"qwen3.5-flash-02-23","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.5 Flash 02 23","description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","context_length":1000000,"aliases":["qwen/qwen3.5-flash-02-23"],"pricing":{"prompt":"6.5e-8","completion":"2.6e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemma-3-4b-it","object":"model","created":1787387470,"owned_by":"google","name":"Gemma 3 4b It","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"aliases":["google/gemma-3-4b-it"],"pricing":{"prompt":"5.0000000000000004e-8","completion":"1.0000000000000001e-7"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"minimax-m2.1","object":"model","created":1787387470,"owned_by":"minimax","name":"Minimax M2.1","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","context_length":204800,"aliases":["minimax/minimax-m2.1"],"pricing":{"prompt":"3e-7","completion":"0.0000012"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"deepseek-v4-pro-0813","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek V4 Pro 0813","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","context_length":1048576,"aliases":["deepseek/deepseek-v4-pro-0813"],"pricing":{"prompt":"0.0000024","completion":"0.0000048"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","supported_efforts":["max","high","low"]}},{"id":"qwen-2.5-7b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen 2.5 7b Instruct","description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"aliases":["qwen/qwen-2.5-7b-instruct"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"2.0000000000000002e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen3-next-80b-a3b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Next 80b A3b Instruct","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"aliases":["qwen/qwen3-next-80b-a3b-instruct"],"pricing":{"prompt":"9e-8","completion":"0.0000011"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen3-coder-plus","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Coder Plus","description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","context_length":1000000,"aliases":["qwen/qwen3-coder-plus"],"pricing":{"prompt":"6.5e-7","completion":"0.00000325"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"deepseek-v3.2-exp","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek V3.2 Exp","description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"aliases":["deepseek/deepseek-v3.2-exp"],"pricing":{"prompt":"2.7e-7","completion":"4.1e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-30b-a3b-instruct-2507","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 30b A3b Instruct 2507","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","context_length":262144,"aliases":["qwen/qwen3-30b-a3b-instruct-2507"],"pricing":{"prompt":"4.815e-8","completion":"1.9305e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"deepseek-r1-0528","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek R1 0528","description":"May 28th update to the original DeepSeek R1 Performance on par with OpenAI o1, but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"aliases":["deepseek/deepseek-r1-0528"],"pricing":{"prompt":"5e-7","completion":"0.0000021499999999999997"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"qwen3.5-plus-20260420","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.5 Plus 20260420","description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","context_length":1000000,"aliases":["qwen/qwen3.5-plus-20260420"],"pricing":{"prompt":"3e-7","completion":"0.0000018000000000000001"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-3.1-flash-image-preview","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3.1 Flash Image Preview","description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","context_length":65536,"aliases":["google/gemini-3.1-flash-image-preview"],"pricing":{"prompt":"5e-7","completion":"0.000003"},"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["text","image"]},"reasoning":{"mandatory":false,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","minimal"]}},{"id":"claude-sonnet-5","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Sonnet 5","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"aliases":["anthropic/claude-sonnet-5"],"pricing":{"prompt":"0.000002","completion":"0.00001"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"qwen3.6-27b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.6 27b","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"aliases":["qwen/qwen3.6-27b"],"pricing":{"prompt":"2.8899999999999995e-7","completion":"0.0000024"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"qwen3.8-27b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.8 27b","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","context_length":1000000,"aliases":["qwen/qwen3.8-27b"],"pricing":{"prompt":"4.5000000000000003e-7","completion":"0.0000032000000000000003"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"xhigh","default_enabled":true,"supported_efforts":["xhigh","medium","low"]}},{"id":"qwen3-8b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 8b","description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","context_length":131072,"aliases":["qwen/qwen3-8b"],"pricing":{"prompt":"1.17e-7","completion":"4.5500000000000004e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"seed-2.0-mini","object":"model","created":1787387470,"owned_by":"bytedance-seed","name":"Seed 2.0 Mini","description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","context_length":262144,"aliases":["bytedance-seed/seed-2.0-mini"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]}},{"id":"kimi-k2-thinking","object":"model","created":1787387470,"owned_by":"moonshotai","name":"Kimi K2 Thinking","description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","context_length":262144,"aliases":["moonshotai/kimi-k2-thinking"],"pricing":{"prompt":"6e-7","completion":"0.0000025"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_enabled":true}},{"id":"qwen3.8-2.4t-a95b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.8 2.4t A95b","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of Qwen3.8 Max, with 95 billion active parameters out of 2.4 trillion total. It is...","context_length":1048576,"aliases":["qwen/qwen3.8-2.4t-a95b"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"xhigh","default_enabled":true,"supported_efforts":["xhigh","medium","low"]}},{"id":"qwen3-vl-30b-a3b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Vl 30b A3b Instruct","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":262144,"aliases":["qwen/qwen3-vl-30b-a3b-instruct"],"pricing":{"prompt":"1.3e-7","completion":"5.2e-7"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"gemma-3n-e4b-it","object":"model","created":1787387470,"owned_by":"google","name":"Gemma 3n E4b It","description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","context_length":32768,"aliases":["google/gemma-3n-e4b-it"],"pricing":{"prompt":"6e-8","completion":"1.2e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"nemotron-3-super-120b-a12b","object":"model","created":1787387470,"owned_by":"nvidia","name":"Nemotron 3 Super 120b A12b","description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":1000000,"aliases":["nvidia/nemotron-3-super-120b-a12b"],"pricing":{"prompt":"3e-7","completion":"9.000000000000001e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["medium","low"],"supports_max_tokens":true}},{"id":"gpt-5.6-terra-pro","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.6 Terra Pro","description":"GPT-5.6 Terra Pro is the same underlying model as GPT-5.6 Terra, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"aliases":["openai/gpt-5.6-terra-pro"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000024"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"qwen3-32b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 32b","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"aliases":["qwen/qwen3-32b"],"pricing":{"prompt":"8e-8","completion":"2.8e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"mimo-v2.5-pro","object":"model","created":1787387470,"owned_by":"xiaomi","name":"Mimo V2.5 Pro","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","context_length":1050000,"aliases":["xiaomi/mimo-v2.5-pro"],"pricing":{"prompt":"4.35e-7","completion":"8.7e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-max","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Max","description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","context_length":262144,"aliases":["qwen/qwen3-max"],"pricing":{"prompt":"7.8e-7","completion":"0.0000039"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-3.5-flash-lite","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3.5 Flash Lite","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"aliases":["google/gemini-3.5-flash-lite"],"pricing":{"prompt":"3e-7","completion":"0.0000025"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]}},{"id":"qwen3-coder-flash","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Coder Flash","description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","context_length":1000000,"aliases":["qwen/qwen3-coder-flash"],"pricing":{"prompt":"1.95e-7","completion":"9.75e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"gpt-5.4","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.4","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"aliases":["openai/gpt-5.4"],"pricing":{"prompt":"0.0000025","completion":"0.000015"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"qwen3-coder-30b-a3b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Coder 30b A3b Instruct","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","context_length":262144,"aliases":["qwen/qwen3-coder-30b-a3b-instruct"],"pricing":{"prompt":"7e-8","completion":"2.8e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen3.5-plus-02-15","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.5 Plus 02 15","description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","context_length":1000000,"aliases":["qwen/qwen3.5-plus-02-15"],"pricing":{"prompt":"2.6e-7","completion":"0.00000156"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"hermes-4-405b","object":"model","created":1787387470,"owned_by":"nousresearch","name":"Hermes 4 405b","description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","context_length":131072,"aliases":["nousresearch/hermes-4-405b"],"pricing":{"prompt":"0.000001","completion":"0.000003"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-vl-30b-a3b-thinking","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Vl 30b A3b Thinking","description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","context_length":262144,"aliases":["qwen/qwen3-vl-30b-a3b-thinking"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"0.0000024"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"claude-sonnet-4.5","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Sonnet 4.5","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"aliases":["anthropic/claude-sonnet-4.5"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"claude-haiku-4.5","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Haiku 4.5","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"aliases":["anthropic/claude-haiku-4.5"],"pricing":{"prompt":"0.000001","completion":"0.000005"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"hermes-3-llama-3.1-70b","object":"model","created":1787387470,"owned_by":"nousresearch","name":"Hermes 3 Llama 3.1 70b","description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"aliases":["nousresearch/hermes-3-llama-3.1-70b"],"pricing":{"prompt":"7e-7","completion":"7e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen-2.5-coder-32b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen 2.5 Coder 32b Instruct","description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","context_length":32768,"aliases":["qwen/qwen-2.5-coder-32b-instruct"],"pricing":{"prompt":"6.6e-7","completion":"0.000001"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"gemini-3-flash-preview","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3 Flash Preview","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"aliases":["google/gemini-3-flash-preview"],"pricing":{"prompt":"5e-7","completion":"0.000003"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]}},{"id":"gpt-5.6-luna-pro","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.6 Luna Pro","description":"GPT-5.6 Luna Pro is the same underlying model as GPT-5.6 Luna, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"aliases":["openai/gpt-5.6-luna-pro"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"0.0000012000000000000002"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"4.0000000000000003e-7","completion":"0.0000024000000000000003"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"o3","object":"model","created":1787387470,"owned_by":"openai","name":"o3","description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"aliases":["openai/o3"],"pricing":{"prompt":"0.000002","completion":"0.000008"},"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"o1","object":"model","created":1787387470,"owned_by":"openai","name":"o1","description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"aliases":["openai/o1"],"pricing":{"prompt":"0.000015","completion":"0.00006"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"glm-4.5v","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 4.5v","description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","context_length":65536,"aliases":["z-ai/glm-4.5v"],"pricing":{"prompt":"6e-7","completion":"0.0000018000000000000001"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-2.5-pro-preview","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 2.5 Pro Preview","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"aliases":["google/gemini-2.5-pro-preview"],"pricing":{"prompt":"0.00000125","completion":"0.00001"},"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"gemini-2.5-flash","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 2.5 Flash","description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"aliases":["google/gemini-2.5-flash"],"pricing":{"prompt":"3e-7","completion":"0.0000025"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-next-80b-a3b-thinking","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Next 80b A3b Thinking","description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","context_length":262144,"aliases":["qwen/qwen3-next-80b-a3b-thinking"],"pricing":{"prompt":"1.5e-7","completion":"0.0000012"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"glm-4.6","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 4.6","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":204800,"aliases":["z-ai/glm-4.6"],"pricing":{"prompt":"5e-7","completion":"0.000002"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-3.5-flash","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3.5 Flash","description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"aliases":["google/gemini-3.5-flash"],"pricing":{"prompt":"0.0000015","completion":"0.000009"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]}},{"id":"gpt-5.4-mini","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.4 Mini","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"aliases":["openai/gpt-5.4-mini"],"pricing":{"prompt":"7.5e-7","completion":"0.0000045"},"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"qwen3-235b-a22b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 235b A22b","description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","context_length":131072,"aliases":["qwen/qwen3-235b-a22b"],"pricing":{"prompt":"4.5500000000000004e-7","completion":"0.0000018200000000000002"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"glm-4.6v","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 4.6v","description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","context_length":131072,"aliases":["z-ai/glm-4.6v"],"pricing":{"prompt":"3e-7","completion":"9.000000000000001e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gpt-5.1","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.1","description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"aliases":["openai/gpt-5.1"],"pricing":{"prompt":"0.00000125","completion":"0.00001"},"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"none","default_enabled":true,"supported_efforts":["high","medium","low","none"]}},{"id":"qwen3.7-flash","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.7 Flash","description":null,"context_length":1000000,"aliases":["qwen/qwen3.7-flash"],"pricing":{"prompt":"3e-8","completion":"1.3e-7"},"pricing_tiers":[{"above_prompt_tokens":32768,"prompt":"1.0000000000000001e-7","completion":"4.0000000000000003e-7"},{"above_prompt_tokens":262144,"prompt":"2.0000000000000002e-7","completion":"8.000000000000001e-7"}],"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true}},{"id":"deepseek-v4-flash-0731","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek V4 Flash 0731","description":null,"context_length":1048576,"aliases":["deepseek/deepseek-v4-flash-0731"],"pricing":{"prompt":"1.4e-7","completion":"2.8e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","high","low"]}},{"id":"sonar-pro","object":"model","created":1787387470,"owned_by":"perplexity","name":"Sonar Pro","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","context_length":200000,"aliases":["perplexity/sonar-pro"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"claude-opus-4.7","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Opus 4.7","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"aliases":["anthropic/claude-opus-4.7"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":false,"supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"glm-4.7-flash","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 4.7 Flash","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","context_length":202752,"aliases":["z-ai/glm-4.7-flash"],"pricing":{"prompt":"6e-8","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"nemotron-3-nano-30b-a3b","object":"model","created":1787387470,"owned_by":"nvidia","name":"Nemotron 3 Nano 30b A3b","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","context_length":262144,"aliases":["nvidia/nemotron-3-nano-30b-a3b"],"pricing":{"prompt":"5.0000000000000004e-8","completion":"2.0000000000000002e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"grok-4.20-multi-agent","object":"model","created":1787387470,"owned_by":"x-ai","name":"Grok 4.20 Multi Agent","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","context_length":2000000,"aliases":["x-ai/grok-4.20-multi-agent"],"pricing":{"prompt":"0.00000125","completion":"0.0000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["xhigh","high","medium","low"]}},{"id":"qwen3-30b-a3b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 30b A3b","description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","context_length":131072,"aliases":["qwen/qwen3-30b-a3b"],"pricing":{"prompt":"1.3e-7","completion":"5.2e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"gemini-3.1-flash-image","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3.1 Flash Image","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","context_length":131072,"aliases":["google/gemini-3.1-flash-image"],"pricing":{"prompt":"5e-7","completion":"0.000003"},"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["text","image"]},"reasoning":{"mandatory":false,"default_effort":"minimal","default_enabled":true,"supported_efforts":["high","minimal"]}},{"id":"qwen3-vl-235b-a22b-thinking","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Vl 235b A22b Thinking","description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","context_length":131072,"aliases":["qwen/qwen3-vl-235b-a22b-thinking"],"pricing":{"prompt":"4.0000000000000003e-7","completion":"0.000004"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"qwen3-vl-8b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Vl 8b Instruct","description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","context_length":262144,"aliases":["qwen/qwen3-vl-8b-instruct"],"pricing":{"prompt":"1.17e-7","completion":"4.5500000000000004e-7"},"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"]}},{"id":"kimi-k2","object":"model","created":1787387470,"owned_by":"moonshotai","name":"Kimi K2","description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","context_length":131072,"aliases":["moonshotai/kimi-k2"],"pricing":{"prompt":"5.699999999999999e-7","completion":"0.0000023"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"gemma-3-27b-it","object":"model","created":1787387470,"owned_by":"google","name":"Gemma 3 27b It","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":262144,"aliases":["google/gemma-3-27b-it"],"pricing":{"prompt":"8e-8","completion":"4.5000000000000003e-7"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"sonar-deep-research","object":"model","created":1787387470,"owned_by":"perplexity","name":"Sonar Deep Research","description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","context_length":128000,"aliases":["perplexity/sonar-deep-research"],"pricing":{"prompt":"0.000002","completion":"0.000008"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"hermes-4-70b","object":"model","created":1787387470,"owned_by":"nousresearch","name":"Hermes 4 70b","description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","context_length":131072,"aliases":["nousresearch/hermes-4-70b"],"pricing":{"prompt":"1.3e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-3-pro-image","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3 Pro Image","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":131072,"aliases":["google/gemini-3-pro-image"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["text","image"]},"reasoning":{"mandatory":true}},{"id":"qwen3-vl-8b-thinking","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Vl 8b Thinking","description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","context_length":131072,"aliases":["qwen/qwen3-vl-8b-thinking"],"pricing":{"prompt":"1.8e-7","completion":"0.0000021000000000000002"},"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"glm-4.5-air","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 4.5 Air","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"aliases":["z-ai/glm-4.5-air"],"pricing":{"prompt":"1.3e-7","completion":"8.5e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"sonar-reasoning-pro","object":"model","created":1787387470,"owned_by":"perplexity","name":"Sonar Reasoning Pro","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","context_length":128000,"aliases":["perplexity/sonar-reasoning-pro"],"pricing":{"prompt":"0.000002","completion":"0.000008"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3.5-27b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.5 27b","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"aliases":["qwen/qwen3.5-27b"],"pricing":{"prompt":"1.95e-7","completion":"0.00000156"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"kimi-k2-0905","object":"model","created":1787387470,"owned_by":"moonshotai","name":"Kimi K2 0905","description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","context_length":262144,"aliases":["moonshotai/kimi-k2-0905"],"pricing":{"prompt":"6e-7","completion":"0.0000025"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen3.5-122b-a10b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.5 122b A10b","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","context_length":262144,"aliases":["qwen/qwen3.5-122b-a10b"],"pricing":{"prompt":"2.6e-7","completion":"0.00000208"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"mixtral-8x22b-instruct","object":"model","created":1787387470,"owned_by":"mistralai","name":"Mixtral 8x22b Instruct","description":"Mistral's official instruct fine-tuned version of Mixtral 8x22B. It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","context_length":65536,"aliases":["mistralai/mixtral-8x22b-instruct"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"]}},{"id":"o4-mini-high","object":"model","created":1787387470,"owned_by":"openai","name":"o4 Mini High","description":"OpenAI o4-mini-high is the same model as o4-mini with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"aliases":["openai/o4-mini-high"],"pricing":{"prompt":"0.0000011","completion":"0.0000044"},"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","supported_efforts":["high"]}},{"id":"qwen2.5-vl-72b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen2.5 Vl 72b Instruct","description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","context_length":128000,"aliases":["qwen/qwen2.5-vl-72b-instruct"],"pricing":{"prompt":"8.000000000000001e-7","completion":"0.000001"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"qwen3-coder","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Coder","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262144,"aliases":["qwen/qwen3-coder"],"pricing":{"prompt":"3e-7","completion":"0.000001"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"llama-3.2-1b-instruct","object":"model","created":1787387470,"owned_by":"meta-llama","name":"Llama 3.2 1b Instruct","description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","context_length":60000,"aliases":["meta-llama/llama-3.2-1b-instruct"],"pricing":{"prompt":"2.7e-8","completion":"2.01e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen3-235b-a22b-thinking-2507","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 235b A22b Thinking 2507","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","context_length":262144,"aliases":["qwen/qwen3-235b-a22b-thinking-2507"],"pricing":{"prompt":"2.3000000000000002e-7","completion":"0.0000023"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"claude-sonnet-4","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Sonnet 4","description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","context_length":1000000,"aliases":["anthropic/claude-sonnet-4"],"pricing":{"prompt":"0.000003","completion":"0.000015"},"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"deepseek-r1-distill-llama-70b","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek R1 Distill Llama 70b","description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on Llama-3.3-70B-Instruct, using outputs from DeepSeek R1. The model combines advanced distillation techniques to achieve high performance across...","context_length":8192,"aliases":["deepseek/deepseek-r1-distill-llama-70b"],"pricing":{"prompt":"8.000000000000001e-7","completion":"8.000000000000001e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"sonar","object":"model","created":1787387470,"owned_by":"perplexity","name":"Sonar","description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","context_length":127072,"aliases":["perplexity/sonar"],"pricing":{"prompt":"0.000001","completion":"0.000001"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"qwen3-coder-next","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Coder Next","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","context_length":262144,"aliases":["qwen/qwen3-coder-next"],"pricing":{"prompt":"1.2e-7","completion":"8.000000000000001e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"claude-opus-4.8","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Opus 4.8","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"aliases":["anthropic/claude-opus-4.8"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":false,"supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"qwen3.5-35b-a3b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.5 35b A3b","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"aliases":["qwen/qwen3.5-35b-a3b"],"pricing":{"prompt":"2.5e-7","completion":"0.00000125"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"mimo-v2.5","object":"model","created":1787387470,"owned_by":"xiaomi","name":"Mimo V2.5","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","context_length":1050000,"aliases":["xiaomi/mimo-v2.5"],"pricing":{"prompt":"1.4e-7","completion":"2.8e-7"},"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"seed-2.0-lite","object":"model","created":1787387470,"owned_by":"bytedance-seed","name":"Seed 2.0 Lite","description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","context_length":262144,"aliases":["bytedance-seed/seed-2.0-lite"],"pricing":{"prompt":"2.5e-7","completion":"0.000002"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]}},{"id":"o1-pro","object":"model","created":1787387470,"owned_by":"openai","name":"o1 Pro","description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"aliases":["openai/o1-pro"],"pricing":{"prompt":"0.00015","completion":"0.0006"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-vl-235b-a22b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Vl 235b A22b Instruct","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"aliases":["qwen/qwen3-vl-235b-a22b-instruct"],"pricing":{"prompt":"2.1e-7","completion":"0.0000018999999999999998"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"qwen3.6-plus","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.6 Plus","description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","context_length":1000000,"aliases":["qwen/qwen3.6-plus"],"pricing":{"prompt":"3.25e-7","completion":"0.00000195"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-3.7-flash","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3.7 Flash","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","context_length":1048576,"aliases":["google/gemini-3.7-flash"],"pricing":{"prompt":"7.5e-7","completion":"0.00000375"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low"]}},{"id":"qwen3.5-9b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.5 9b","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"aliases":["qwen/qwen3.5-9b"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"1.5e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gpt-5","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy in high-stakes use cases. It supports test-time routing features and advanced prompt understanding, including user-specified intent like \"think hard about this.\" Improvements include reductions in hallucination, sycophancy, and better performance in coding, writing, and health-related tasks.","context_length":400000,"aliases":["openai/gpt-5"],"pricing":{"prompt":"0.00000125","completion":"0.00001"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]}},{"id":"qwen3-235b-a22b-2507","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 235b A22b 2507","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","context_length":262144,"aliases":["qwen/qwen3-235b-a22b-2507"],"pricing":{"prompt":"9e-8","completion":"5.5e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"o3-pro","object":"model","created":1787387470,"owned_by":"openai","name":"o3 Pro","description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"aliases":["openai/o3-pro"],"pricing":{"prompt":"0.00002","completion":"0.00008"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"grok-4.20","object":"model","created":1787387470,"owned_by":"x-ai","name":"Grok 4.20","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","context_length":2000000,"aliases":["x-ai/grok-4.20"],"pricing":{"prompt":"0.00000125","completion":"0.0000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"gpt-5.6-luna","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.6 Luna","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"aliases":["openai/gpt-5.6-luna"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"0.0000012000000000000002"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"4.0000000000000003e-7","completion":"0.0000024000000000000003"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"qwen3-max-thinking","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Max Thinking","description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","context_length":262144,"aliases":["qwen/qwen3-max-thinking"],"pricing":{"prompt":"7.8e-7","completion":"0.0000039"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gpt-5-nano","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5 Nano","description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"aliases":["openai/gpt-5-nano"],"pricing":{"prompt":"5.0000000000000004e-8","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]}},{"id":"gemma-3-12b-it","object":"model","created":1787387470,"owned_by":"google","name":"Gemma 3 12b It","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"aliases":["google/gemma-3-12b-it"],"pricing":{"prompt":"5.0000000000000004e-8","completion":"1.5e-7"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"granite-4.0-h-micro","object":"model","created":1787387470,"owned_by":"ibm-granite","name":"Granite 4.0 H Micro","description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","context_length":131000,"aliases":["ibm-granite/granite-4.0-h-micro"],"pricing":{"prompt":"1.7e-8","completion":"1.12e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"o3-mini","object":"model","created":1787387470,"owned_by":"openai","name":"o3 Mini","description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"aliases":["openai/o3-mini"],"pricing":{"prompt":"0.0000011","completion":"0.0000044"},"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gpt-5.4-pro","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.4 Pro","description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"aliases":["openai/gpt-5.4-pro"],"pricing":{"prompt":"0.00003","completion":"0.00018"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["xhigh","high","medium"]}},{"id":"deepseek-v3.1-terminus","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek V3.1 Terminus","description":"DeepSeek-V3.1 Terminus is an update to DeepSeek V3.1 that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","context_length":163840,"aliases":["deepseek/deepseek-v3.1-terminus"],"pricing":{"prompt":"2.7e-7","completion":"0.000001"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-30b-a3b-thinking-2507","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 30b A3b Thinking 2507","description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","context_length":81920,"aliases":["qwen/qwen3-30b-a3b-thinking-2507"],"pricing":{"prompt":"2.0000000000000002e-7","completion":"0.0000024"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"gemini-2.5-flash-image","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 2.5 Flash Image","description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","context_length":32768,"aliases":["google/gemini-2.5-flash-image"],"pricing":{"prompt":"3e-7","completion":"0.0000025"},"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["text","image"]}},{"id":"o3-mini-high","object":"model","created":1787387470,"owned_by":"openai","name":"o3 Mini High","description":"OpenAI o3-mini-high is the same model as o3-mini with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"aliases":["openai/o3-mini-high"],"pricing":{"prompt":"0.0000011","completion":"0.0000044"},"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","supported_efforts":["high"]}},{"id":"glm-4.5","object":"model","created":1787387470,"owned_by":"z-ai","name":"Glm 4.5","description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","context_length":131072,"aliases":["z-ai/glm-4.5"],"pricing":{"prompt":"6e-7","completion":"0.0000022"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"grok-build-0.1","object":"model","created":1787387470,"owned_by":"x-ai","name":"Grok Build 0.1","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","context_length":256000,"aliases":["x-ai/grok-build-0.1"],"pricing":{"prompt":"0.000001","completion":"0.000002"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"gpt-5.6-terra","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.6 Terra","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"aliases":["openai/gpt-5.6-terra"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000024"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"qwen-plus-2025-07-28:thinking","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen Plus 2025 07 28:Thinking","description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"aliases":["qwen/qwen-plus-2025-07-28:thinking"],"pricing":{"prompt":"2.6e-7","completion":"7.8e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen3-14b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 14b","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"aliases":["qwen/qwen3-14b"],"pricing":{"prompt":"1.2e-7","completion":"2.4e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen-plus-2025-07-28","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen Plus 2025 07 28","description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"aliases":["qwen/qwen-plus-2025-07-28"],"pricing":{"prompt":"2.6e-7","completion":"7.8e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gpt-4o-2024-11-20","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 4o 2024 11 20","description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","context_length":128000,"aliases":["openai/gpt-4o-2024-11-20"],"pricing":{"prompt":"0.0000025","completion":"0.00001"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]}},{"id":"deepseek-chat","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek Chat","description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":163840,"aliases":["deepseek/deepseek-chat"],"pricing":{"prompt":"4.0000000000000003e-7","completion":"0.0000013"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"qwen3.6-35b-a3b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.6 35b A3b","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"aliases":["qwen/qwen3.6-35b-a3b"],"pricing":{"prompt":"1.4e-7","completion":"0.000001"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"gemini-2.5-pro","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 2.5 Pro","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"aliases":["google/gemini-2.5-pro"],"pricing":{"prompt":"0.00000125","completion":"0.00001"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"deepseek-chat-v3.1","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek Chat V3.1","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":163840,"aliases":["deepseek/deepseek-chat-v3.1"],"pricing":{"prompt":"2.5e-7","completion":"9.499999999999999e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gpt-4o","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 4o","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of GPT-4 Turbo while being twice as...","context_length":128000,"aliases":["openai/gpt-4o"],"pricing":{"prompt":"0.0000025","completion":"0.00001"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]}},{"id":"gemini-2.5-pro-preview-05-06","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 2.5 Pro Preview 05 06","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"aliases":["google/gemini-2.5-pro-preview-05-06"],"pricing":{"prompt":"0.00000125","completion":"0.00001"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"claude-opus-4.1","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Opus 4.1","description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"aliases":["anthropic/claude-opus-4.1"],"pricing":{"prompt":"0.000015","completion":"0.000075"},"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"qwen-2.5-72b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen 2.5 72b Instruct","description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"aliases":["qwen/qwen-2.5-72b-instruct"],"pricing":{"prompt":"3.6e-7","completion":"4.0000000000000003e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"gpt-5.4-image-2","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.4 Image 2","description":"GPT-5.4 Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","context_length":272000,"aliases":["openai/gpt-5.4-image-2"],"pricing":{"prompt":"0.000008","completion":"0.000015"},"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["text","image"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"gemini-3.1-pro-preview-customtools","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3.1 Pro Preview Customtools","description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","context_length":1048576,"aliases":["google/gemini-3.1-pro-preview-customtools"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","audio","image","video","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]}},{"id":"gemma-4-26b-a4b-it","object":"model","created":1787387470,"owned_by":"google","name":"Gemma 4 26b A4b It","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"aliases":["google/gemma-4-26b-a4b-it"],"pricing":{"prompt":"7e-8","completion":"3.4000000000000003e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"gemma-2-27b-it","object":"model","created":1787387470,"owned_by":"google","name":"Gemma 2 27b It","description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the Gemini models. Gemma models are well-suited for a variety of...","context_length":8192,"aliases":["google/gemma-2-27b-it"],"pricing":{"prompt":"6.5e-7","completion":"6.5e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"gpt-5.6-sol","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.6 Sol","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"aliases":["openai/gpt-5.6-sol"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000024"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"knowledge_cutoff":"2026-02-16","max_completion_tokens":128000,"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"o4-mini","object":"model","created":1787387470,"owned_by":"openai","name":"o4 Mini","description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"aliases":["openai/o4-mini"],"pricing":{"prompt":"0.0000011","completion":"0.0000044"},"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gpt-oss-20b","object":"model","created":1787387470,"owned_by":"openai","name":"GPT Oss 20b","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"aliases":["openai/gpt-oss-20b"],"pricing":{"prompt":"3e-8","completion":"1.3e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]}},{"id":"gpt-oss-safeguard-20b","object":"model","created":1787387470,"owned_by":"openai","name":"GPT Oss Safeguard 20b","description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","context_length":131072,"aliases":["openai/gpt-oss-safeguard-20b"],"pricing":{"prompt":"7.5e-8","completion":"3e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"qwen-plus","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen Plus","description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","context_length":1000000,"aliases":["qwen/qwen-plus"],"pricing":{"prompt":"2.6e-7","completion":"7.8e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"deepseek-r1","object":"model","created":1787387470,"owned_by":"deepseek","name":"Deepseek R1","description":"DeepSeek R1 is here: Performance on par with OpenAI o1, but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","context_length":64000,"aliases":["deepseek/deepseek-r1"],"pricing":{"prompt":"7e-7","completion":"0.0000025"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true}},{"id":"qwen3.5-397b-a17b","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.5 397b A17b","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":262144,"aliases":["qwen/qwen3.5-397b-a17b"],"pricing":{"prompt":"3.9e-7","completion":"0.00000234"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gpt-3.5-turbo","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 3.5 Turbo","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"aliases":["openai/gpt-3.5-turbo"],"pricing":{"prompt":"5e-7","completion":"0.0000015"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"claude-opus-4.5","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Opus 4.5","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"aliases":["anthropic/claude-opus-4.5"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"gemini-3.6-flash","object":"model","created":1787387470,"owned_by":"google","name":"Gemini 3.6 Flash","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"aliases":["google/gemini-3.6-flash"],"pricing":{"prompt":"0.0000015","completion":"0.0000075"},"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","default_enabled":true,"supported_efforts":["high","medium","low","minimal"]}},{"id":"grok-4.3","object":"model","created":1787387470,"owned_by":"x-ai","name":"Grok 4.3","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"aliases":["x-ai/grok-4.3"],"pricing":{"prompt":"0.00000125","completion":"0.0000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"low","default_enabled":true,"supported_efforts":["high","medium","low","none"]}},{"id":"claude-opus-4","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Opus 4","description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","context_length":200000,"aliases":["anthropic/claude-opus-4"],"pricing":{"prompt":"0.000015","completion":"0.000075"},"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"grok-4.5","object":"model","created":1787387470,"owned_by":"x-ai","name":"Grok 4.5","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","context_length":500000,"aliases":["x-ai/grok-4.5"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","default_enabled":true,"supported_efforts":["high","medium","low"]}},{"id":"gpt-5.5","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.5","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"aliases":["openai/gpt-5.5"],"pricing":{"prompt":"0.000005","completion":"0.00003"},"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["xhigh","high","medium","low","none"]}},{"id":"seed-1.6-flash","object":"model","created":1787387470,"owned_by":"bytedance-seed","name":"Seed 1.6 Flash","description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","context_length":262144,"aliases":["bytedance-seed/seed-1.6-flash"],"pricing":{"prompt":"7.5e-8","completion":"3e-7"},"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false}},{"id":"hermes-3-llama-3.1-405b","object":"model","created":1787387470,"owned_by":"nousresearch","name":"Hermes 3 Llama 3.1 405b","description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"aliases":["nousresearch/hermes-3-llama-3.1-405b"],"pricing":{"prompt":"0.000001","completion":"0.000001"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"claude-3-haiku","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude 3 Haiku","description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","context_length":200000,"aliases":["anthropic/claude-3-haiku"],"pricing":{"prompt":"2.5e-7","completion":"0.00000125"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"qwen3-vl-32b-instruct","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3 Vl 32b Instruct","description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_length":131072,"aliases":["qwen/qwen3-vl-32b-instruct"],"pricing":{"prompt":"1.0399999999999999e-7","completion":"4.1599999999999997e-7"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"gpt-5.6-sol-pro","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5.6 Sol Pro","description":"GPT-5.6 Sol Pro is the same underlying model as GPT-5.6 Sol, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"aliases":["openai/gpt-5.6-sol-pro"],"pricing":{"prompt":"0.000002","completion":"0.000012"},"pricing_tiers":[{"above_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000024"}],"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"medium","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"]}},{"id":"claude-fable-5","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Fable 5","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"aliases":["anthropic/claude-fable-5"],"pricing":{"prompt":"0.00001","completion":"0.00005"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"gpt-oss-120b","object":"model","created":1787387470,"owned_by":"openai","name":"GPT Oss 120b","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"aliases":["openai/gpt-oss-120b"],"pricing":{"prompt":"1.5e-7","completion":"6e-7"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low"]}},{"id":"gpt-5-mini","object":"model","created":1787387470,"owned_by":"openai","name":"GPT 5 Mini","description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"aliases":["openai/gpt-5-mini"],"pricing":{"prompt":"2.5e-7","completion":"0.000002"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"medium","supported_efforts":["high","medium","low","minimal"]}},{"id":"mistral-large","object":"model","created":1787387470,"owned_by":"mistralai","name":"Mistral Large","description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":128000,"aliases":["mistralai/mistral-large"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"]}},{"id":"grok-4.6","object":"model","created":1787387470,"owned_by":"x-ai","name":"Grok 4.6","description":"Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","context_length":500000,"aliases":["x-ai/grok-4.6"],"pricing":{"prompt":"0.000002","completion":"0.000006"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":true,"default_effort":"high","default_enabled":true,"supported_efforts":["xhigh","high","medium","low"]}},{"id":"ministral-3b-2512","object":"model","created":1787387470,"owned_by":"mistralai","name":"Ministral 3b 2512","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","context_length":131072,"aliases":["mistralai/ministral-3b-2512"],"pricing":{"prompt":"1.0000000000000001e-7","completion":"1.0000000000000001e-7"},"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]}},{"id":"claude-opus-5","object":"model","created":1787387470,"owned_by":"anthropic","name":"Claude Opus 5","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"aliases":["anthropic/claude-opus-5"],"pricing":{"prompt":"0.000005","completion":"0.000025"},"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_effort":"high","default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"]}},{"id":"qwen/qwen3.7-flash:free","object":"model","created":1787387470,"owned_by":"qwen","name":"Qwen3.7 Flash (free)","description":"Rate-limited free tier.","context_length":1000000,"aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"reasoning":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true}},{"id":"auto","object":"model","created":1787387470,"owned_by":"bazaarlink","name":"Auto Router","description":"Automatically selects the best model for the request.","aliases":[],"pricing":{"prompt":"-1","completion":"-1"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}},{"id":"auto:free","object":"model","created":1787387470,"owned_by":"bazaarlink","name":"Auto Router (free)","description":"Automatically routes to a free model. Always free, rate-limited.","aliases":[],"pricing":{"prompt":"0","completion":"0"},"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"]}}]}