{"object":"list","data":[{"id":"deepseek-v4-flash-vision-exp","object":"model","created":0,"owned_by":"deepseek","name":"DeepSeek V4 Flash Vision","description":"Experimental vision-enabled release of DeepSeek V4 Flash (open weights, MIT, 305B MoE) — the same efficiency tier with image input. DeepSeek labels it experimental; expect the id to move when a stable release lands.","context_window":1048576,"max_output_tokens":384000,"best_for":"Cheap multimodal reasoning on the V4 Flash line","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.297,"output":0.891,"cached":null},"list_price":{"mode":"per-token","input":0.297,"output":0.891,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"efficient","weights":"open","weights_source":{"checkpoint":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","read":"2026-09-04"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"preview","release_date":"2026-08-21","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","analyse-documents"],"recommended_for":["chat","analyse-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen3.8-27b","object":"model","created":0,"owned_by":"alibaba","name":"Qwen 3.8 27B","description":"Alibaba's open-weight (Apache-2.0) dense 27B vision-language model from the Qwen 3.8 line. 1M context, text, image and video input, reasoning, tool calling and structured outputs — the open checkpoint next to the closed Qwen 3.8 Max and Flash.","context_window":1000000,"max_output_tokens":131072,"best_for":"Open-weight multimodal reasoning at 1M context","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.567,"output":4.05,"cached":null},"list_price":{"mode":"per-token","input":0.567,"output":4.05,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","video"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"Qwen/Qwen3.8-27B","source":"https://huggingface.co/Qwen/Qwen3.8-27B","read":"2026-09-03"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-14","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","analyse-documents","write-code"],"recommended_for":["chat","analyse-documents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3.8-flash","object":"model","created":0,"owned_by":"google","name":"Gemini 3.8 Flash","description":"Google's newest Flash tier, GA 2026-09-02 — significant gains over 3.7 Flash on software engineering and agentic tasks. 1M context, 64K output, text, image, audio, video and file input, reasoning, tool calling, structured outputs.","context_window":1048576,"max_output_tokens":65536,"best_for":"High-volume agentic and multimodal work at 1M context","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":1.013,"output":5.063,"cached":null},"list_price":{"mode":"per-token","input":2.026,"output":10.126,"cached":null},"pricing_mode":"per-token","promotion":{"input":1.013,"output":5.063,"list_input":2.026,"list_output":10.126,"percent_off":50,"ends":"2026-09-10"},"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://deepmind.google/models/model-cards/gemini-3-8-flash/","read":"2026-09-03"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-09-02","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code","analyse-documents"],"recommended_for":["chat","build-agents","write-code","analyse-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"claude-fable-5.1","object":"model","created":0,"owned_by":"anthropic","name":"Claude Fable 5.1","description":"Anthropic's newest Mythos-class flagship — improves on Fable 5 across the board, biggest gains in agentic coding and long-running agentic workflows. 1M context, 128K output, vision, file input, adaptive thinking.","context_window":1000000,"max_output_tokens":128000,"best_for":"The hardest reasoning and long-horizon agentic work money can buy","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":13.5,"output":67.5,"cached":null},"list_price":{"mode":"per-token","input":13.5,"output":67.5,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://www.anthropic.com/claude/fable","read":"2026-09-03"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-09-01","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"claude-fable-5","object":"model","created":0,"owned_by":"anthropic","name":"Claude Fable 5","description":"Anthropic's Mythos-class flagship tier — above Opus in capability. 1M context, vision, file input, extended reasoning.","context_window":1000000,"max_output_tokens":128000,"best_for":"The hardest reasoning and agentic work money can buy","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":13.5,"output":67.5,"cached":null},"list_price":{"mode":"per-token","input":13.5,"output":67.5,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://www.anthropic.com/transparency","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-06-09","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"grok-4.6","object":"model","created":0,"owned_by":"xai","name":"Grok 4.6","description":"xAI's newest flagship. 500K context, vision, file input, reasoning; the rate doubles above 200K prompt tokens.","context_window":500000,"max_output_tokens":8192,"best_for":"Frontier reasoning and agentic work at a lower price point","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":2.7,"output":8.1,"cached":null},"list_price":{"mode":"per-token","input":2.7,"output":8.1,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://docs.x.ai/developers/release-notes","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-12","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"llama-4-maverick","object":"model","created":0,"owned_by":"meta","name":"Llama 4 Maverick","description":"Meta's larger Llama 4 tier, above Scout. 1M context, vision, served from two independent providers.","context_window":1048576,"max_output_tokens":16384,"best_for":"Open-weights work at long context with a second supplier behind it","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.27,"output":1.08,"cached":null},"list_price":{"mode":"per-token","input":0.27,"output":1.08,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":false,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"meta-llama/Llama-4-Maverick-17B-128E-Instruct","source":"https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-04-05","type":"language","tags":["tool-use","vision"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"muse-spark-1.1","object":"model","created":0,"owned_by":"meta","name":"Muse Spark 1.1","description":"Meta's Muse Spark tier. 1M context, vision, file input, reasoning — it spends ~160 tokens thinking before a one-word answer.","context_window":1048576,"max_output_tokens":8192,"best_for":"Long-context reasoning where thinking time is acceptable","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":1.542,"output":5.243,"cached":null},"list_price":{"mode":"per-token","input":1.542,"output":5.243,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://about.fb.com/news/2026/04/introducing-muse-spark-meta-superintelligence-labs/","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-16","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"muse-spark-1.2","object":"model","created":0,"owned_by":"meta","name":"Muse Spark 1.2","description":"Meta's newest Muse Spark — the model behind Muse Code, with bigger gains in coding and agentic work over 1.1. 1M context, multimodal input (text, image, video, file, audio), reasoning.","context_window":1048576,"max_output_tokens":8192,"best_for":"Agentic coding and long-horizon tool use","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":1.688,"output":5.738,"cached":null},"list_price":{"mode":"per-token","input":1.688,"output":5.738,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://about.fb.com/news/2026/04/introducing-muse-spark-meta-superintelligence-labs/","read":"2026-08-31"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-05","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"muse-glimmer-30b","object":"model","created":0,"owned_by":"meta","name":"Muse Glimmer 30B","description":"Meta's open-weight (Apache 2.0) 30B dense multimodal model, distilled from Muse Spark and tuned for local agentic work. Vision + reasoning, single-GPU friendly.","context_window":131072,"max_output_tokens":8192,"best_for":"Open-weight multimodal agents at lower cost","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.405,"output":1.62,"cached":null},"list_price":{"mode":"per-token","input":0.405,"output":1.62,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"meta-models/Muse-Glimmer-30B","source":"https://huggingface.co/meta-models/Muse-Glimmer-30B","read":"2026-08-31"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-10","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"mimo-v2.5","object":"model","created":0,"owned_by":"xiaomi","name":"MiMo V2.5","description":"Xiaomi's open-weight (MIT) flagship — multimodal, 1M context, reasoning. Strong agentic and coding work at low cost.","context_window":1050000,"max_output_tokens":16384,"best_for":"Multimodal reasoning and agentic work at low cost","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.189,"output":0.378,"cached":null},"list_price":{"mode":"per-token","input":0.189,"output":0.378,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"XiaomiMiMo/MiMo-V2.5","source":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","read":"2026-08-31"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-01","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"hy3","object":"model","created":0,"owned_by":"tencent","name":"Hunyuan 3","description":"Tencent's Hunyuan 3 flagship. 256K context, reasoning, tool use.","context_window":262144,"max_output_tokens":8192,"best_for":"General reasoning and agentic work","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.108,"output":0.446,"cached":null},"list_price":{"mode":"per-token","input":0.108,"output":0.446,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://huggingface.co/tencent","read":"2026-09-04"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-01","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"grok-4.20-multi-agent","object":"model","created":0,"owned_by":"xai","name":"Grok 4.20 Multi-Agent","description":"Grok 4.20 run as a multi-agent ensemble, at the same list price. 2M context, vision, file input, reasoning.","context_window":2000000,"max_output_tokens":8192,"best_for":"Problems worth spending several agent passes on","speed":"slow","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":1.767,"output":3.534,"cached":null},"list_price":{"mode":"per-token","input":1.767,"output":3.534,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","response_format","structured_outputs","reasoning"],"latency_tier":"slow","cost_tier":"premium","quality_tier":"strong","weights":"closed","weights_source":{"checkpoint":null,"source":"https://docs.x.ai/developers/release-notes","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-03-31","type":"language","tags":["reasoning","vision","web-search"],"use_cases":["chat","build-agents","research-web"],"recommended_for":["chat","build-agents","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"grok-4.20","object":"model","created":0,"owned_by":"xai","name":"Grok 4.20","description":"xAI's 4.20 generation. 2M context, vision, file input, reasoning; the rate doubles above 200K prompt tokens.","context_window":2000000,"max_output_tokens":8192,"best_for":"Very long documents and codebases in one prompt","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":1.763,"output":3.528,"cached":null},"list_price":{"mode":"per-token","input":1.763,"output":3.528,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"strong","weights":"closed","weights_source":{"checkpoint":null,"source":"https://docs.x.ai/developers/models/grok-4.20","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-03-31","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"grok-4.5","object":"model","created":0,"owned_by":"xai","name":"Grok 4.5","description":"xAI's current flagship. 500K context, vision, file input, reasoning; the rate doubles above 200K prompt tokens.","context_window":500000,"max_output_tokens":8192,"best_for":"Hard reasoning and agentic work at long context","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":2.73,"output":8.189,"cached":null},"list_price":{"mode":"per-token","input":2.73,"output":8.189,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://docs.x.ai/developers/release-notes","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-08","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-5.6-sol-pro","object":"model","created":0,"owned_by":"openai","name":"GPT-5.6 Sol Pro","description":"Sol at the same list price with the Pro serving profile. 1.05M context, vision, file input; the rate doubles above 272K prompt tokens.","context_window":1050000,"max_output_tokens":128000,"best_for":"Hardest reasoning where the Pro serving profile matters","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":2.7,"output":13.5,"cached":0.3375},"list_price":{"mode":"per-token","input":5.4,"output":27,"cached":0.675},"pricing_mode":"per-token","promotion":{"input":2.7,"output":13.5,"list_input":5.4,"list_output":27,"percent_off":50,"ends":"2026-09-10"},"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://huggingface.co/openai","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-09","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-5.6-sol","object":"model","created":0,"owned_by":"openai","name":"GPT-5.6 Sol","description":"OpenAI's top 5.6 tier. 1.05M context, vision, file input; the rate doubles above 272K prompt tokens.","context_window":1050000,"max_output_tokens":128000,"best_for":"Hardest reasoning and long-horizon agentic work","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":2.7,"output":13.5,"cached":0.3375},"list_price":{"mode":"per-token","input":5.4,"output":27,"cached":0.675},"pricing_mode":"per-token","promotion":{"input":2.7,"output":13.5,"list_input":5.4,"list_output":27,"percent_off":50,"ends":"2026-09-10"},"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://developers.openai.com/api/docs/models/gpt-5.6-sol","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-09","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-5.6-terra-pro","object":"model","created":0,"owned_by":"openai","name":"GPT-5.6 Terra Pro","description":"Terra's Pro tier, twice Terra's list price. 1.05M context, vision, file input; the rate doubles above 272K prompt tokens.","context_window":1050000,"max_output_tokens":128000,"best_for":"Agentic coding and tool use at long context","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":2.7,"output":16.2,"cached":0.27},"list_price":{"mode":"per-token","input":2.7,"output":16.2,"cached":0.27},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://huggingface.co/openai","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-09","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-5.6-luna-pro","object":"model","created":0,"owned_by":"openai","name":"GPT-5.6 Luna Pro","description":"Luna at the same list price with the Pro serving profile. 1.05M context, vision, file input; the rate doubles above 272K prompt tokens.","context_window":1050000,"max_output_tokens":128000,"best_for":"High-volume work that needs the Pro serving profile","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.27,"output":1.62,"cached":0.027},"list_price":{"mode":"per-token","input":0.27,"output":1.62,"cached":0.027},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"strong","weights":"closed","weights_source":{"checkpoint":null,"source":"https://huggingface.co/openai","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-09","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","research-web"],"recommended_for":["chat","build-agents","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-5.6-luna","object":"model","created":0,"owned_by":"openai","name":"GPT-5.6 Luna","description":"OpenAI's cheapest 5.6 tier. 1.05M context, vision, file input; the rate doubles above 272K prompt tokens.","context_window":1050000,"max_output_tokens":128000,"best_for":"High-volume chat and classification at long context","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.27,"output":1.62,"cached":0.027},"list_price":{"mode":"per-token","input":0.27,"output":1.62,"cached":0.027},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"strong","weights":"closed","weights_source":{"checkpoint":null,"source":"https://developers.openai.com/api/docs/models/gpt-5.6-luna","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-09","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","research-web"],"recommended_for":["chat","build-agents","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"claude-sonnet-5","object":"model","created":0,"owned_by":"anthropic","name":"Claude Sonnet 5","description":"Anthropic's balanced tier. 1M context, vision, file input, prompt caching, extended reasoning.","context_window":1000000,"max_output_tokens":128000,"best_for":"Agentic coding, long-horizon tasks, vision, tool use","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":2.7,"output":13.5,"cached":0.27},"list_price":{"mode":"per-token","input":2.7,"output":13.5,"cached":0.27},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://www.anthropic.com/transparency","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-06-30","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"claude-opus-5-fast","object":"model","created":0,"owned_by":"anthropic","name":"Claude Opus 5 Fast","description":"Opus 5 tuned for latency, at twice the list price. 1M context, vision, file input, prompt caching.","context_window":1000000,"max_output_tokens":128000,"best_for":"Interactive work that needs Opus quality without Opus latency","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":13.5,"output":67.5,"cached":1.35},"list_price":{"mode":"per-token","input":13.5,"output":67.5,"cached":1.35},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://platform.claude.com/docs/en/about-claude/models/whats-new-opus-5","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-24","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"claude-opus-5","object":"model","created":0,"owned_by":"anthropic","name":"Claude Opus 5","description":"Anthropic's flagship tier. 1M context, vision, file input, prompt caching, extended reasoning.","context_window":1000000,"max_output_tokens":128000,"best_for":"Hardest reasoning, complex agentic loops, long-horizon coding","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":6.75,"output":33.75,"cached":0.675},"list_price":{"mode":"per-token","input":6.75,"output":33.75,"cached":0.675},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://platform.claude.com/docs/en/about-claude/models/whats-new-opus-5","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-24","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-5.6-terra","object":"model","created":0,"owned_by":"openai","name":"GPT-5.6 Terra","description":"OpenAI's balanced GPT-5.6 tier, between the Sol flagship and the Luna cost tier. 1M context, accepts text, images and files.","context_window":1050000,"max_output_tokens":8192,"best_for":"Everyday coding, reasoning and agentic work where capability and cost both matter","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":2.7,"output":16.2,"cached":0.27},"list_price":{"mode":"per-token","input":2.7,"output":16.2,"cached":0.27},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://developers.openai.com/api/docs/models/gpt-5.6-terra","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-09","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["write-code","build-agents","chat","analyse-documents","research-web"],"recommended_for":["write-code","build-agents","chat","analyse-documents","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen3.7-flash","object":"model","created":0,"owned_by":"alibaba","name":"Qwen 3.7 Flash","description":"Alibaba's vision-language reasoning model. 1M context, takes text, image and video, and is the cheapest model in the catalogue by a wide margin.","context_window":1000000,"max_output_tokens":8192,"best_for":"Multimodal agents, visual coding and computer use at high volume","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.0498,"output":0.2164,"cached":0.0081},"list_price":{"mode":"per-token","input":0.0498,"output":0.2164,"cached":0.0081},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","video"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"efficient","weights":null,"weights_source":null,"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-28","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code","analyse-documents"],"recommended_for":["chat","build-agents","write-code","analyse-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3.5-flash-lite","object":"model","created":0,"owned_by":"google","name":"Gemini 3.5 Flash Lite","description":"Google's cheapest 1M-context tier. Takes text, image, audio and video; output price includes thinking tokens.","context_window":1048576,"max_output_tokens":8192,"best_for":"High-volume classification, extraction and summarisation at 1M context","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.405,"output":3.375,"cached":0.0405},"list_price":{"mode":"per-token","input":0.405,"output":3.375,"cached":0.0405},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","audio","video"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"efficient","weights":"closed","weights_source":{"checkpoint":null,"source":"https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-21","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","analyse-documents","write-content","research-web"],"recommended_for":["chat","analyse-documents","write-content","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3.1-pro","object":"model","created":0,"owned_by":"google","name":"Gemini 3.1 Pro","description":"Google's Pro reasoning tier — the step above Flash. 1M context, 64K max output, accepts text, image, audio, video and file; the rate steps up above 200K prompt tokens.","context_window":1048576,"max_output_tokens":65536,"best_for":"Hard reasoning and long-horizon agentic work that Flash cannot carry","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":2.7,"output":16.2,"cached":0.27},"list_price":{"mode":"per-token","input":2.7,"output":16.2,"cached":0.27},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","audio","video","file"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://deepmind.google/models/model-cards/gemini-3-1-pro/","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"preview","release_date":"2026-02-19","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["write-code","build-agents","chat","analyse-documents","research-web"],"recommended_for":["write-code","build-agents","chat","analyse-documents","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3.6-flash","object":"model","created":0,"owned_by":"google","name":"Gemini 3.6 Flash","description":"Google's July 2026 Flash tier. 1M context, accepts text, image, audio and video, and costs less per output token than 3.5 Flash.","context_window":1048576,"max_output_tokens":8192,"best_for":"High-volume multimodal work at 1M context","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":1.013,"output":5.063,"cached":0.10125},"list_price":{"mode":"per-token","input":1.013,"output":5.063,"cached":0.10125},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","audio","video"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://deepmind.google/models/model-cards/gemini-3-6-flash/","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-21","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","analyse-documents","write-content","research-web"],"recommended_for":["chat","analyse-documents","write-content","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3.7-flash","object":"model","created":0,"owned_by":"google","name":"Gemini 3.7 Flash","description":"Google's newest Flash tier. 1M context and 64K max output, the widest input set in the family — text, image, audio, video and file — with reasoning, tool calling and structured outputs.","context_window":1048576,"max_output_tokens":65536,"best_for":"High-volume multimodal work at 1M context where the reply itself is long","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":1.013,"output":5.063,"cached":0.050625},"list_price":{"mode":"per-token","input":2.026,"output":10.126,"cached":0.10125},"pricing_mode":"per-token","promotion":{"input":1.013,"output":5.063,"list_input":2.026,"list_output":10.126,"percent_off":50,"ends":"2026-09-10"},"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://deepmind.google/models/model-cards/gemini-3-7-flash/","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-13","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","analyse-documents","write-content","research-web"],"recommended_for":["chat","analyse-documents","write-content","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"kimi-k3","object":"model","created":0,"owned_by":"moonshot","name":"Kimi K3","description":"Moonshot's 2.8T open-weight multimodal reasoning model. Successor to K2.7 — 1M context, vision input, built for long-horizon agentic work.","context_window":1048576,"max_output_tokens":32768,"best_for":"Long-horizon agentic coding, reasoning over large repos, vision input","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":4.05,"output":20.25,"cached":0.405},"list_price":{"mode":"per-token","input":4.05,"output":20.25,"cached":0.405},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"moonshotai/Kimi-K3","source":"https://huggingface.co/moonshotai/Kimi-K3","read":"2026-08-14"},"reduced_precision":{"verdict":"not-reduced","levels":[],"native":{"formats":["mxfp4","mxfp8"],"source":"https://huggingface.co/moonshotai/Kimi-K3","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-07-16","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["write-code","build-agents","chat","analyse-documents"],"recommended_for":["write-code","build-agents","chat","analyse-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen-3.8-max","object":"model","created":0,"owned_by":"alibaba","name":"Qwen 3.8 Max","description":"Alibaba's newest flagship, GA successor to the 3.7 Max line — and the first Max with vision. 1M context, multimodal reasoning over text and images, cheaper per token than the 3.7 it replaces.","context_window":1000000,"max_output_tokens":131072,"best_for":"General reasoning, visual understanding, long context","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":2.2275,"output":6.684,"cached":0.2781},"list_price":{"mode":"per-token","input":2.2275,"output":6.684,"cached":0.2781},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"Qwen/Qwen3.8-2.4T-A95B","source":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","read":"2026-09-04"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-03","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","write-code","analyse-documents"],"recommended_for":["chat","write-code","analyse-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen3.8-flash","object":"model","created":0,"owned_by":"alibaba","name":"Qwen 3.8 Flash","description":"Alibaba's newest Flash tier — ultra cost-efficient multimodal MoE. 1M context, vision, cheaper per token than the Max line.","context_window":1000000,"max_output_tokens":131072,"best_for":"High-volume multimodal chat, cheap long context","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.203,"output":0.635,"cached":null},"list_price":{"mode":"per-token","input":0.203,"output":0.635,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"Qwen/Qwen3.8-Flash-Next","source":"https://huggingface.co/Qwen/Qwen3.8-Flash-Next","read":"2026-09-04"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-26","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","write-code","analyse-documents"],"recommended_for":["chat","write-code","analyse-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen-3.7-max","object":"model","created":0,"owned_by":"alibaba","name":"Qwen 3.7 Max","description":"Alibaba's newest closed-weight flagship. 1M context, top reasoning + multilingual.","context_window":1000000,"max_output_tokens":32768,"best_for":"General, reasoning, multilingual, long context","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":2.304,"output":6.909,"cached":0.16875},"list_price":{"mode":"per-token","input":2.304,"output":6.909,"cached":0.16875},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://qwen.ai/blog?id=qwen3.7","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-05-21","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","write-code","translate"],"recommended_for":["chat","write-code","translate"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen-3.6-plus","object":"model","created":0,"owned_by":"alibaba","name":"Qwen 3.6 Plus","description":"Alibaba's newest flagship. #1 on Kyma.","context_window":1000000,"max_output_tokens":65536,"best_for":"General, reasoning, multilingual","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":0.655,"output":3.93,"cached":null},"list_price":{"mode":"per-token","input":0.655,"output":3.93,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://qwen.ai/blog?id=qwen3.6","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-04-02","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","write-code","translate"],"recommended_for":["chat","write-code","translate"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen-3.7-plus","object":"model","created":0,"owned_by":"alibaba","name":"Qwen 3.7 Plus","description":"Alibaba's newest Plus flagship. 1M context, vision input, top agentic + reasoning.","context_window":1000000,"max_output_tokens":32768,"best_for":"General, agentic coding, reasoning, vision, long context","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":0.4431,"output":1.773,"cached":0.0864},"list_price":{"mode":"per-token","input":0.4431,"output":1.773,"cached":0.0864},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://qwen.ai/blog?id=qwen3.7-plus","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-06-02","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","write-code","build-agents"],"recommended_for":["chat","write-code","build-agents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen-3-coder","object":"model","created":0,"owned_by":"alibaba","name":"Qwen 3 Coder","description":"Purpose-built for code generation.","context_window":131072,"max_output_tokens":32768,"best_for":"Code generation, debugging","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.334,"output":1.519,"cached":0.135},"list_price":{"mode":"per-token","input":0.334,"output":1.519,"cached":0.135},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"Qwen/Qwen3-Coder-480B-A35B-Instruct","source":"https://huggingface.co/Qwen/Qwen3-Coder-480B-A35B-Instruct","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp4","fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/Qwen/Qwen3-Coder-480B-A35B-Instruct","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-07-22","type":"language","tags":["tool-use"],"use_cases":["chat","write-code","build-agents"],"recommended_for":["chat","write-code","build-agents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen-3-32b","object":"model","created":0,"owned_by":"alibaba","name":"Qwen 3 32B","description":"Coding-focused 32B. Strong on code, math and multilingual.","context_window":32768,"max_output_tokens":8192,"best_for":"Code, math, multilingual","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.108,"output":0.378,"cached":null},"list_price":{"mode":"per-token","input":0.108,"output":0.378,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"Qwen/Qwen3-32B","source":"https://huggingface.co/Qwen/Qwen3-32B","read":"2026-08-14"},"reduced_precision":null,"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-04-28","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","write-code","solve-math","translate"],"recommended_for":["chat","write-code","solve-math","translate"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemma-4-31b","object":"model","created":0,"owned_by":"google","name":"Gemma 4 31B","description":"Google's newest open model. Multimodal.","context_window":128000,"max_output_tokens":8192,"best_for":"Multimodal, vision, general","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":0.0763,"output":0.218,"cached":0.135},"list_price":{"mode":"per-token","input":0.0763,"output":0.218,"cached":0.135},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"google/gemma-4-31B-it","source":"https://huggingface.co/google/gemma-4-31B-it","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp4","fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/google/gemma-4-31b-it","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-04-02","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat"],"recommended_for":["chat"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-m2.5","object":"model","created":0,"owned_by":"minimax","name":"MiniMax M2.5","description":"SWE-bench 80.2%. Top agentic coding.","context_window":196608,"max_output_tokens":32768,"best_for":"Coding, agentic workflows","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.3826,"output":1.346,"cached":null},"list_price":{"mode":"per-token","input":0.3826,"output":1.346,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"MiniMaxAI/MiniMax-M2.5","source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.5","read":"2026-08-14"},"reduced_precision":{"verdict":"not-reduced","levels":[],"native":{"formats":["fp8","bf16"],"source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.5/blob/main/config.json","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-02-12","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","write-code","build-agents"],"recommended_for":["chat","write-code","build-agents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-m3","object":"model","created":0,"owned_by":"minimax","name":"MiniMax M3","description":"MSA sparse attention. SWE-Bench Pro 59%, Terminal-Bench 66%. Agentic coding, 1M context, multimodal input.","context_window":1048576,"max_output_tokens":32768,"best_for":"Agentic coding, debugging, long-horizon tasks, multimodal input","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":0.3985,"output":1.594,"cached":0.0756},"list_price":{"mode":"per-token","input":0.3985,"output":1.594,"cached":0.0756},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","video"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"MiniMaxAI/MiniMax-M3","source":"https://www.minimax.io/blog/minimax-m3","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp4","fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/MiniMaxAI/MiniMax-M3/blob/main/config.json","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-05-31","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","write-code","build-agents"],"recommended_for":["chat","write-code","build-agents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"deepseek-v4-pro","object":"model","created":0,"owned_by":"deepseek","name":"DeepSeek V4 Pro","description":"1.6T MoE flagship. 1M context. Top reasoning tier.","context_window":1000000,"max_output_tokens":65536,"best_for":"Top reasoning, complex coding, long context","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":0.6901,"output":1.38,"cached":0.133},"list_price":{"mode":"per-token","input":0.6901,"output":1.38,"cached":0.133},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"deepseek-ai/DeepSeek-V4-Pro","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro","read":"2026-08-14"},"reduced_precision":{"verdict":"not-reduced","levels":[],"native":{"formats":["fp4","fp8"],"source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-04-23","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","write-code","build-agents"],"recommended_for":["chat","write-code","build-agents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"deepseek-v4-flash","object":"model","created":0,"owned_by":"deepseek","name":"DeepSeek V4 Flash","description":"284B MoE. 1M context. Fast + cheap V4 tier.","context_window":1000000,"max_output_tokens":65536,"best_for":"General, coding, long context, value","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":0.1389,"output":0.2778,"cached":0.0243},"list_price":{"mode":"per-token","input":0.1389,"output":0.2778,"cached":0.0243},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"deepseek-ai/DeepSeek-V4-Flash","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash","read":"2026-08-14"},"reduced_precision":{"verdict":"not-reduced","levels":[],"native":{"formats":["fp4","fp8"],"source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-04-23","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","write-code"],"recommended_for":["chat","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"deepseek-v3","object":"model","created":0,"owned_by":"deepseek","name":"DeepSeek V3","description":"Previous-gen flagship. Stable, proven.","context_window":160000,"max_output_tokens":8192,"best_for":"Reasoning, coding, general","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.351,"output":0.513,"cached":null},"list_price":{"mode":"per-token","input":0.351,"output":0.513,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"deepseek-ai/DeepSeek-V3.2","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3.2","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp4"],"native":{"formats":["fp8","bf16"],"source":"https://huggingface.co/deepseek-ai/DeepSeek-V3.2","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-12-01","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","write-code"],"recommended_for":["chat","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"deepseek-r1","object":"model","created":0,"owned_by":"deepseek","name":"DeepSeek R1","description":"Top reasoning model. 96% cheaper than o1.","context_window":64000,"max_output_tokens":32768,"best_for":"Math, logic, complex reasoning","speed":"slow","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.7425,"output":2.957,"cached":null},"list_price":{"mode":"per-token","input":0.7425,"output":2.957,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"slow","cost_tier":"balanced","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"deepseek-ai/DeepSeek-R1-0528","source":"https://huggingface.co/deepseek-ai/DeepSeek-R1-0528","read":"2026-08-14"},"reduced_precision":null,"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-01-20","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","analyse-documents","solve-math"],"recommended_for":["chat","analyse-documents","solve-math"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3.5-flash","object":"model","created":0,"owned_by":"google","name":"Gemini 3.5 Flash","description":"Newest Gemini Flash. 1M context, multimodal input.","context_window":1048576,"max_output_tokens":8192,"best_for":"Long context, multimodal, fast reasoning","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":2.025,"output":12.15,"cached":0.2025},"list_price":{"mode":"per-token","input":2.025,"output":12.15,"cached":0.2025},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","audio","video"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-05-19","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","research-web"],"recommended_for":["chat","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-2.5-flash","object":"model","created":0,"owned_by":"google","name":"Gemini 2.5 Flash","description":"Gemini 2.5 Flash. 1M context. RETIRING 2026-10-16 — Google discontinues Gemini 2.5 on AI Studio (Vertex ~10-16→10-20); migrate to gemini-3-flash / gemini-3.6-flash.","context_window":1048576,"max_output_tokens":8192,"best_for":"Long context, fast","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.405,"output":3.375,"cached":0.0405},"list_price":{"mode":"per-token","input":0.405,"output":3.375,"cached":0.0405},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","audio","video"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"strong","weights":"closed","weights_source":{"checkpoint":null,"source":"https://huggingface.co/google","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":"2026-10-16","tier":null,"capability":null,"release_stage":"stable","release_date":"2025-03-20","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","research-web"],"recommended_for":["chat","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3-flash","object":"model","created":0,"owned_by":"google","name":"Gemini 3 Flash","description":"Newest Gemini. 1M context.","context_window":1048576,"max_output_tokens":8192,"best_for":"Long context, reasoning","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.675,"output":4.05,"cached":0.0675},"list_price":{"mode":"per-token","input":0.675,"output":4.05,"cached":0.0675},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","audio","video"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://deepmind.google/models/model-cards/gemini-3-flash/","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"preview","release_date":"2025-12-17","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","research-web"],"recommended_for":["chat","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"llama-3.3-70b","object":"model","created":0,"owned_by":"meta","name":"Llama 3.3 70B","description":"Most popular open model. Great all-rounder.","context_window":128000,"max_output_tokens":8192,"best_for":"General, code, tool use","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.135,"output":0.432,"cached":null},"list_price":{"mode":"per-token","input":0.135,"output":0.432,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"meta-llama/Llama-3.3-70B-Instruct","source":"https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct","read":"2026-08-14"},"reduced_precision":null,"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2024-12-06","type":"language","tags":["tool-use"],"use_cases":["chat","write-code"],"recommended_for":["chat","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-oss-120b","object":"model","created":0,"owned_by":"openai","name":"GPT-OSS 120B","description":"OpenAI's open source. 120B parameters.","context_window":128000,"max_output_tokens":8192,"best_for":"General intelligence, writing","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.0494,"output":0.2406,"cached":null},"list_price":{"mode":"per-token","input":0.0494,"output":0.2406,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"openai/gpt-oss-120b","source":"https://huggingface.co/openai/gpt-oss-120b","read":"2026-08-14"},"reduced_precision":null,"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-08-05","type":"language","tags":["tool-use","reasoning"],"use_cases":["write-content","chat"],"recommended_for":["write-content","chat"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"nemotron-3-ultra-550b","object":"model","created":0,"owned_by":"nvidia","name":"Nemotron 3 Ultra 550B","description":"NVIDIA's strongest US open-weight. 550B MoE (55B active), hybrid Mamba-Transformer. 1M context, 300+ tok/s.","context_window":1000000,"max_output_tokens":32768,"best_for":"Reasoning, coding, general, long context, fast throughput","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":0.675,"output":3.375,"cached":0.135},"list_price":{"mode":"per-token","input":0.675,"output":3.375,"cached":0.135},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","source":"https://research.nvidia.com/labs/nemotron/Nemotron-3-Ultra/","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp4","fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-06-04","type":"language","tags":["tool-use","reasoning"],"use_cases":["write-code","chat"],"recommended_for":["write-code","chat"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"step-3.7-flash","object":"model","created":0,"owned_by":"stepfun","name":"Step 3.7 Flash","description":"StepFun's fast flash tier. 256K context, multimodal input, tool calling. Cheap throughput.","context_window":256000,"max_output_tokens":8192,"best_for":"Cheap fast throughput, multimodal input, bulk tasks","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.27,"output":1.553,"cached":0.054},"list_price":{"mode":"per-token","input":0.27,"output":1.553,"cached":0.054},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image","video"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"stepfun-ai/Step-3.7-Flash","source":"https://huggingface.co/stepfun-ai/Step-3.7-Flash","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp8"],"native":{"formats":["bf16","fp32"],"source":"https://huggingface.co/stepfun-ai/Step-3.7-Flash","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-05-28","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat"],"recommended_for":["chat"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"kimi-k2.7-code","object":"model","created":0,"owned_by":"moonshot","name":"Kimi K2.7 Code","description":"Coding specialist. +21.8% Kimi Code Bench vs K2.6, ~30% fewer reasoning tokens. Always-thinking. 262K context.","context_window":262144,"max_output_tokens":16384,"best_for":"End-to-end coding, long-horizon programming, agentic tool use","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":1.009,"output":4.774,"cached":null},"list_price":{"mode":"per-token","input":1.009,"output":4.774,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"moonshotai/Kimi-K2.7-Code","source":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","read":"2026-08-14"},"reduced_precision":{"verdict":"not-reduced","levels":[],"native":{"formats":["int4","bf16"],"source":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-06-12","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"kimi-k2.6","object":"model","created":0,"owned_by":"moonshot","name":"Kimi K2.6","description":"Moonshot's newest. Agentic + vision + reasoning. 262K context.","context_window":262144,"max_output_tokens":32768,"best_for":"Agentic coding, vision, reasoning, long horizon","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":0.7856,"output":3.667,"cached":null},"list_price":{"mode":"per-token","input":0.7856,"output":3.667,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"moonshotai/Kimi-K2.6","source":"https://huggingface.co/moonshotai/Kimi-K2.6","read":"2026-08-14"},"reduced_precision":{"verdict":"not-reduced","levels":[],"native":{"formats":["int4","bf16"],"source":"https://huggingface.co/moonshotai/Kimi-K2.6","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-04-20","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"kimi-k2.5","object":"model","created":0,"owned_by":"moonshot","name":"Kimi K2.5","description":"Multimodal agentic. 262K context.","context_window":262144,"max_output_tokens":32768,"best_for":"Agentic coding, vision, tool use","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.6075,"output":3.038,"cached":null},"list_price":{"mode":"per-token","input":0.6075,"output":3.038,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"moonshotai/Kimi-K2.5","source":"https://huggingface.co/moonshotai/Kimi-K2.5","read":"2026-08-14"},"reduced_precision":null,"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-01-26","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"claude-opus-4-7","object":"model","created":0,"owned_by":"anthropic","name":"Claude Opus 4.7","description":"Anthropic's flagship reasoning + agentic tier. 1M context, vision, prompt caching, code execution. Best for hardest tasks.","context_window":1000000,"max_output_tokens":128000,"best_for":"Hardest reasoning, complex agentic loops, long-horizon coding","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":6.75,"output":33.75,"cached":0.675},"list_price":{"mode":"per-token","input":6.75,"output":33.75,"cached":0.675},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://www.anthropic.com/transparency","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-04-16","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"claude-sonnet-4-6","object":"model","created":0,"owned_by":"anthropic","name":"Claude Sonnet 4.6","description":"Anthropic's balanced tier. 1M context, vision, agentic tool use, prompt caching. The default workhorse.","context_window":1000000,"max_output_tokens":128000,"best_for":"Agentic coding, long-horizon tasks, vision, tool use","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":4.05,"output":20.25,"cached":0.405},"list_price":{"mode":"per-token","input":4.05,"output":20.25,"cached":0.405},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://www.anthropic.com/transparency","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-02-17","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","write-code","research-web"],"recommended_for":["chat","build-agents","write-code","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"claude-haiku-4-5","object":"model","created":0,"owned_by":"anthropic","name":"Claude Haiku 4.5","description":"Anthropic's fast tier. Sub-second TTFT, vision, tool use, prompt caching. 200K context.","context_window":200000,"max_output_tokens":64000,"best_for":"Voice agents, real-time UX, bulk classification, fast tool use","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":1.35,"output":6.75,"cached":0.135},"list_price":{"mode":"per-token","input":1.35,"output":6.75,"cached":0.135},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"premium","quality_tier":"strong","weights":"closed","weights_source":{"checkpoint":null,"source":"https://www.anthropic.com/transparency","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-10-15","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","build-agents","research-web"],"recommended_for":["chat","build-agents","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-m2.7","object":"model","created":0,"owned_by":"minimax","name":"MiniMax M2.7","description":"Next-gen agentic productivity.","context_window":204800,"max_output_tokens":32768,"best_for":"Agentic coding, productivity, debugging","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.405,"output":1.62,"cached":null},"list_price":{"mode":"per-token","input":0.405,"output":1.62,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","reasoning"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"MiniMaxAI/MiniMax-M2.7","source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7","read":"2026-08-14"},"reduced_precision":{"verdict":"not-reduced","levels":[],"native":{"formats":["fp8","bf16"],"source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7/blob/main/config.json","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-03-18","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","build-agents","write-code","write-content"],"recommended_for":["chat","build-agents","write-code","write-content"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"glm-5.2","object":"model","created":0,"owned_by":"zhipu","name":"GLM 5.2","description":"Frontier open-weight. #1 Intelligence Index among open models. 1M context, long-horizon agentic coding.","context_window":1000000,"max_output_tokens":131072,"best_for":"Agentic coding, SWE tasks, long-horizon, 1M context","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":1.101,"output":3.76,"cached":0.29835},"list_price":{"mode":"per-token","input":1.101,"output":3.76,"cached":0.29835},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"zai-org/GLM-5.2","source":"https://huggingface.co/zai-org/GLM-5.2","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp4","fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/zai-org/GLM-5.2/blob/main/config.json","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-06-16","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"glm-5.3","object":"model","created":0,"owned_by":"zhipu","name":"GLM 5.3","description":"Zhipu's newest flagship — coding/agentic upgrade over GLM 5.2. ~1.3M context, open weights.","context_window":1310720,"max_output_tokens":131072,"best_for":"Agentic coding, SWE tasks, long-horizon, 1.3M context","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":1.89,"output":5.94,"cached":null},"list_price":{"mode":"per-token","input":1.89,"output":5.94,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"zai-org/GLM-5.3","source":"https://huggingface.co/zai-org/GLM-5.3","read":"2026-08-29"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-14","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"glm-5.1","object":"model","created":0,"owned_by":"zhipu","name":"GLM 5.1","description":"#1 SWE-Bench Pro open-weight. 8-hour agentic runs.","context_window":203000,"max_output_tokens":65536,"best_for":"Agentic coding, SWE tasks, long-horizon","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":1.89,"output":5.94,"cached":0.27675},"list_price":{"mode":"per-token","input":1.89,"output":5.94,"cached":0.27675},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"zai-org/GLM-5.1","source":"https://huggingface.co/zai-org/GLM-5.1","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp4","fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/zai-org/GLM-5.1/blob/main/config.json","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-04-07","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","build-agents","write-code"],"recommended_for":["chat","build-agents","write-code"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"glm-4.5-air","object":"model","created":0,"owned_by":"zhipu","name":"GLM 4.5 Air","description":"Cheap agentic MoE (106B/12B active).","context_window":131072,"max_output_tokens":8192,"best_for":"Bulk agent, long context, cheap","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.1945,"output":1.272,"cached":0.03375},"list_price":{"mode":"per-token","input":0.1945,"output":1.272,"cached":0.03375},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"zai-org/GLM-4.5-Air","source":"https://huggingface.co/zai-org/GLM-4.5-Air","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/zai-org/GLM-4.5-Air","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-07-28","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat","build-agents"],"recommended_for":["chat","build-agents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"glm-4.7-flash","object":"model","created":0,"owned_by":"zhipu","name":"GLM 4.7 Flash","description":"Ultra cheap. 200K context. Fast.","context_window":203000,"max_output_tokens":65536,"best_for":"Cheap long context, bulk","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.081,"output":0.54,"cached":0.0135},"list_price":{"mode":"per-token","input":0.081,"output":0.54,"cached":0.0135},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"efficient","weights":"open","weights_source":{"checkpoint":"zai-org/GLM-4.7-Flash","source":"https://huggingface.co/zai-org/GLM-4.7-Flash","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/zai-org/GLM-4.7-Flash/blob/main/config.json","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"preview","release_date":"2026-01-19","type":"language","tags":["tool-use","reasoning"],"use_cases":["chat"],"recommended_for":["chat"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"glm-5.3-flash","object":"model","created":0,"owned_by":"zhipu","name":"GLM 5.3 Flash","description":"Zhipu's newest Flash — MIT-licensed, multimodal, latency/cost-optimized successor to GLM 4.7 Flash. ~1.3M context, vision.","context_window":1310720,"max_output_tokens":65536,"best_for":"Cheap multimodal long context, bulk","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":0.101,"output":0.338,"cached":null},"list_price":{"mode":"per-token","input":0.203,"output":0.675,"cached":null},"pricing_mode":"per-token","promotion":{"input":0.101,"output":0.338,"list_input":0.203,"list_output":0.675,"percent_off":50,"ends":"2026-09-09"},"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"efficient","weights":"open","weights_source":{"checkpoint":"zai-org/GLM-5.3-Flash","source":"https://huggingface.co/zai-org/GLM-5.3-Flash","read":"2026-09-04"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-08-26","type":"language","tags":["tool-use","reasoning","vision"],"use_cases":["chat"],"recommended_for":["chat"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"grok-4.3","object":"model","created":0,"owned_by":"xai","name":"Grok 4.3","description":"xAI's frontier model. 1M context, strong reasoning + tool use.","context_window":1000000,"max_output_tokens":32768,"best_for":"General, reasoning, agentic coding, long context","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":1.92,"output":3.838,"cached":0.27},"list_price":{"mode":"per-token","input":1.92,"output":3.838,"cached":0.27},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":"closed","weights_source":{"checkpoint":null,"source":"https://docs.x.ai/developers/models/grok-4.3","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2026-04-30","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","write-code","build-agents","research-web"],"recommended_for":["chat","write-code","build-agents","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"grok-build","object":"model","created":0,"owned_by":"xai","name":"Grok Build","description":"xAI's coding-specialized model. Fast, tool-native, built for agentic dev.","context_window":256000,"max_output_tokens":32768,"best_for":"Agentic coding, debugging, fast tool use","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-token","input":1.389,"output":2.777,"cached":0.27},"list_price":{"mode":"per-token","input":1.389,"output":2.777,"cached":0.27},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":true,"supports_vision":true,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching","reasoning"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"strong","weights":"closed","weights_source":{"checkpoint":null,"source":"https://x.ai/news/grok-build-open-source","read":"2026-08-14"},"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"preview","release_date":"2026-05-20","type":"language","tags":["tool-use","reasoning","vision","web-search"],"use_cases":["chat","write-code","build-agents","research-web"],"recommended_for":["chat","write-code","build-agents","research-web"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"sonar","object":"model","created":0,"owned_by":"perplexity","name":"Sonar","description":"Perplexity's live web-search model. Returns current, cited answers — grounded in a real-time search of the web. Bills a small per-request search fee on top of tokens.","context_window":127072,"max_output_tokens":4096,"best_for":"Live web search, current events, research with citations","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":1.35,"output":1.35,"cached":null},"list_price":{"mode":"per-token","input":1.35,"output":1.35,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","prompt_caching"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-02-19","type":"language","tags":["vision","web-search"],"use_cases":["chat","research-web","write-content"],"recommended_for":["chat","research-web","write-content"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"sonar-pro","object":"model","created":0,"owned_by":"perplexity","name":"Sonar Pro","description":"Perplexity's pro web-search model. Deeper multi-step search, 200K context, longer cited answers. Per-request search fee on top of tokens.","context_window":200000,"max_output_tokens":8000,"best_for":"Deep web research, complex current-events questions, longer cited reports","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-token","input":4.05,"output":20.25,"cached":null},"list_price":{"mode":"per-token","input":4.05,"output":20.25,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":true,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","prompt_caching"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":{"verdict":"not-established","levels":[],"native":null,"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-02-19","type":"language","tags":["vision","web-search"],"use_cases":["chat","research-web","write-content"],"recommended_for":["chat","research-web","write-content"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"flux-2-pro","object":"model","created":0,"owned_by":"bfl","name":"FLUX.2 Pro","description":"BFL's 32B flagship (3× larger than Flux 1.1). Photoreal, multi-reference (up to 10 sources), unified gen+edit, ~60% accurate text-in-image. $0.03/MP base + $0.015 per extra MP.","context_window":0,"max_output_tokens":null,"best_for":"Photo realism, multi-reference blend, hero shots, gen+edit unified","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.0405,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.0405,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":null,"release_stage":"stable","release_date":"2025-11-25","type":"image","tags":["vision","image-generation"],"use_cases":["generate-images","edit-images"],"recommended_for":["generate-images","edit-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"flux-1.1-ultra","object":"model","created":0,"owned_by":"bfl","name":"FLUX 1.1 Pro Ultra","description":"Legacy. Recommend flux-2-pro for new projects (cheaper at 1MP, higher quality, multi-reference).","context_window":0,"max_output_tokens":null,"best_for":"Cinematic photo, hero shot, editorial","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.081,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.081,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2024-11-01","type":"image","tags":["image-generation"],"use_cases":["generate-images"],"recommended_for":["generate-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"flux-kontext-pro","object":"model","created":0,"owned_by":"bfl","name":"FLUX.1 Kontext Pro","description":"Image-to-image edit and refinement. Mask + inpaint.","context_window":0,"max_output_tokens":null,"best_for":"Image edit, inpaint, refinement","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.054,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.054,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":null,"release_stage":"stable","release_date":"2025-05-29","type":"image","tags":["vision","image-generation"],"use_cases":["generate-images","edit-images"],"recommended_for":["generate-images","edit-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"ideogram-v3","object":"model","created":0,"owned_by":"ideogram","name":"Ideogram V3","description":"Text-in-image specialist. Best for typography, packaging, logos.","context_window":0,"max_output_tokens":null,"best_for":"Typography, packaging, posters, logos","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.081,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.081,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":null,"release_stage":"stable","release_date":"2025-03-26","type":"image","tags":["image-generation"],"use_cases":["generate-images"],"recommended_for":["generate-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"recraft-v4","object":"model","created":0,"owned_by":"recraft","name":"Recraft V4","description":"Top of HF Arena (#1, beats Midjourney V8 / DALL-E 3 / FLUX). Design-aware composition, lighting, textures.","context_window":0,"max_output_tokens":null,"best_for":"Design-quality default, brand assets, illustration","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.054,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.054,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":null,"release_stage":"stable","release_date":"2026-02-17","type":"image","tags":["image-generation"],"use_cases":["generate-images"],"recommended_for":["generate-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"recraft-v4-pro","object":"model","created":0,"owned_by":"recraft","name":"Recraft V4 Pro","description":"Recraft V4 at 4MP for print-ready / large-scale assets. Same design taste as V4, higher resolution.","context_window":0,"max_output_tokens":null,"best_for":"Print, posters, hero campaign, large displays","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.3375,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.3375,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":null,"release_stage":"stable","release_date":"2026-02-17","type":"image","tags":["image-generation"],"use_cases":["generate-images"],"recommended_for":["generate-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"recraft-v4-vector","object":"model","created":0,"owned_by":"recraft","name":"Recraft V4 Vector","description":"Native SVG output — actual paths + structured layers, edit in Figma/Illustrator. Only model on the market that ships true vector files.","context_window":0,"max_output_tokens":null,"best_for":"Logos, icons, illustrations as editable SVG","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.108,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.108,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"vector","release_stage":"stable","release_date":"2026-02-17","type":"image","tags":["image-generation"],"use_cases":["generate-images"],"recommended_for":["generate-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"recraft-v4-vector-pro","object":"model","created":0,"owned_by":"recraft","name":"Recraft V4 Vector Pro","description":"Native SVG at 4MP for print-ready vector assets. Same as V4 Vector with higher detail / scale.","context_window":0,"max_output_tokens":null,"best_for":"Print-ready logos, packaging illustrations, large-scale signage","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.405,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.405,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"vector","release_stage":"stable","release_date":"2026-02-17","type":"image","tags":["image-generation"],"use_cases":["generate-images"],"recommended_for":["generate-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"recraft-v3","object":"model","created":0,"owned_by":"recraft","name":"Recraft V3","description":"Legacy. Recommend recraft-v4 for new projects (same price, top of HF Arena).","context_window":0,"max_output_tokens":null,"best_for":"Vector, illustration, brand assets","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.054,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.054,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2024-10-30","type":"image","tags":["image-generation"],"use_cases":["generate-images"],"recommended_for":["generate-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"kling-2.5-pro","object":"model","created":0,"owned_by":"kuaishou","name":"Kling 2.5 Pro","description":"Cinematic 5-10s video. Photoreal humans, smooth motion. Cheapest Kling tier. T2V or I2V via image_url.","context_window":0,"max_output_tokens":null,"best_for":"Budget cinematic clips, brand b-roll, character shots","speed":"slow","recommended":false,"hot":true,"pricing":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.0945,"cached":null},"list_price":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.0945,"cached":null},"pricing_mode":"per-second","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"balanced","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":null,"release_stage":"stable","release_date":"2025-09-23","type":"video","tags":["video-generation"],"use_cases":["generate-video"],"recommended_for":["generate-video"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"kling-3-pro","object":"model","created":0,"owned_by":"kuaishou","name":"Kling 3 Pro","description":"Flagship Kling. Photoreal humans, smooth motion, sharper than 2.5. T2V or I2V via image_url. For native audio, use kling-3-pro-audio.","context_window":0,"max_output_tokens":null,"best_for":"Premium cinematic clips, character/face shots, hero brand video","speed":"slow","recommended":true,"hot":true,"pricing":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.1512,"cached":null},"list_price":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.1512,"cached":null},"pricing_mode":"per-second","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":null,"release_stage":"stable","release_date":"2026-02-05","type":"video","tags":["video-generation"],"use_cases":["generate-video"],"recommended_for":["generate-video"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"kling-3-pro-audio","object":"model","created":0,"owned_by":"kuaishou","name":"Kling 3 Pro (Audio)","description":"Kling 3 Pro with native audio (ambient + dialogue). Same visuals as kling-3-pro plus synchronized sound. ~50% premium for audio.","context_window":0,"max_output_tokens":null,"best_for":"Cinematic clips needing diegetic sound, talking-head shots, ambient atmosphere","speed":"slow","recommended":false,"hot":true,"pricing":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.2268,"cached":null},"list_price":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.2268,"cached":null},"pricing_mode":"per-second","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video","audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":null,"release_stage":"stable","release_date":"2026-02-05","type":"video","tags":["video-generation"],"use_cases":["generate-video"],"recommended_for":["generate-video"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"seedance-2-pro","object":"model","created":0,"owned_by":"bytedance","name":"Seedance 2 Pro","description":"ByteDance flagship video. Multi-shot, native audio bundled, dynamic camera moves. T2V or I2V via image_url. 720p.","context_window":0,"max_output_tokens":null,"best_for":"Action, multi-shot scenes, social with synced audio, product motion","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.40959,"cached":null},"list_price":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.40959,"cached":null},"pricing_mode":"per-second","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video","audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":null,"release_stage":"stable","release_date":"2026-04-14","type":"video","tags":["video-generation"],"use_cases":["generate-video","write-content"],"recommended_for":["generate-video","write-content"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"seedance-2-fast","object":"model","created":0,"owned_by":"bytedance","name":"Seedance 2 Fast","description":"Seedance 2 fast tier — quicker generation, ~20% cheaper than Pro. Native audio bundled. Best for short social clips.","context_window":0,"max_output_tokens":null,"best_for":"Social shorts, UI motion, rapid iteration, product demos","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.326565,"cached":null},"list_price":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.326565,"cached":null},"pricing_mode":"per-second","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video","audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":null,"release_stage":"stable","release_date":"2026-04-14","type":"video","tags":["video-generation"],"use_cases":["generate-video","write-content"],"recommended_for":["generate-video","write-content"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"nano-banana","object":"model","created":0,"owned_by":"google","name":"Nano Banana","description":"Google Gemini image-gen. Native edit-mode (image-in + prompt → image-out). 3 size tiers (512/1K/2K). RETIRING 2026-10-20 — Google discontinues the Gemini 2.5 endpoint it runs on; migrate to nano-banana-3-flash.","context_window":0,"max_output_tokens":null,"best_for":"In-context image editing, style transfer, low-cost iteration","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.053,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.053,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":"2026-10-20","tier":"fast","capability":null,"release_stage":"stable","release_date":"2025-08-26","type":"image","tags":["vision","image-generation"],"use_cases":["generate-images","edit-images"],"recommended_for":["generate-images","edit-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"nano-banana-3-flash","object":"model","created":0,"owned_by":"google","name":"Nano Banana 3 Flash","description":"Newer Gemini 3.1 image-gen. Same edit-mode contract as Nano Banana; sharper output. The migration target for nano-banana, which Google retires 2026-10-20.","context_window":0,"max_output_tokens":null,"best_for":"In-context image edit + gen at the newest Gemini image quality","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.061,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.061,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":null,"release_stage":"stable","release_date":"2026-02-26","type":"image","tags":["vision","image-generation"],"use_cases":["generate-images","edit-images"],"recommended_for":["generate-images","edit-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"veo-3-fast","object":"model","created":0,"owned_by":"google","name":"Veo 3 Fast","description":"Google Veo 3 fast tier — 720p, no audio. Cheapest Veo. Balanced quality and speed for social/drafts.","context_window":0,"max_output_tokens":null,"best_for":"Default Veo tier — balanced quality and speed","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.135,"cached":null},"list_price":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.135,"cached":null},"pricing_mode":"per-second","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":null,"release_stage":"stable","release_date":"2025-07-31","type":"video","tags":["video-generation"],"use_cases":["generate-video"],"recommended_for":["generate-video"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"veo-3","object":"model","created":0,"owned_by":"google","name":"Veo 3","description":"Google Veo 3 flagship — 1080p with native audio (dialogue + ambient + lip-sync). Top-quality cinematic clips.","context_window":0,"max_output_tokens":null,"best_for":"Hero brand video, talking-head, premium cinematic with native audio","speed":"slow","recommended":true,"hot":true,"pricing":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.54,"cached":null},"list_price":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.54,"cached":null},"pricing_mode":"per-second","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video","audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":null,"release_stage":"stable","release_date":"2025-05-20","type":"video","tags":["video-generation"],"use_cases":["generate-video"],"recommended_for":["generate-video"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"embeddinggemma-300m","object":"model","created":0,"owned_by":"google","name":"EmbeddingGemma 300M","description":"768-dimension embeddings from a 300M-parameter model. Built for on-device and high-volume indexing — the cheapest way to embed a large corpus, at a fraction of the usual per-million rate.","context_window":2048,"max_output_tokens":null,"best_for":"Bulk corpus indexing, RAG retrieval on a budget, semantic dedup","speed":"fast","recommended":true,"hot":false,"pricing":{"mode":"per-token","input":0.0027,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.0027,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2025-09-04","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen3-embedding-8b","object":"model","created":0,"owned_by":"alibaba","name":"Qwen3 Embedding 8B","description":"4096-dimension embeddings with a 32K input window and strong multilingual retrieval. Use when recall quality matters more than the per-million rate, or when documents are long enough that a 2K window would have to chunk them.","context_window":32768,"max_output_tokens":null,"best_for":"Multilingual retrieval, long documents, high-recall RAG","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2025-06-05","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"bge-m3","object":"model","created":0,"owned_by":"baai","name":"BGE-M3","description":"1024-dimension embeddings across 100+ languages, built for retrieval that has to work in more than English. The 8K window takes a whole page without chunking. Reach for it when the corpus is multilingual and the per-million rate matters.","context_window":8192,"max_output_tokens":null,"best_for":"Multilingual RAG, cross-language search","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2024-01-30","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen3-embedding-0.6b","object":"model","created":0,"owned_by":"alibaba","name":"Qwen3 Embedding 0.6B","description":"1024-dimension embeddings from the smallest of the Qwen3 retrieval line, at the same rate as its larger siblings. The 32K window is the reason to reach for it: long documents embed whole while the vector stays small enough to index cheaply.","context_window":32768,"max_output_tokens":null,"best_for":"High-volume indexing, long documents on a budget","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2025-06-05","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen3-embedding-4b","object":"model","created":0,"owned_by":"alibaba","name":"Qwen3 Embedding 4B","description":"2560-dimension embeddings — the middle of the Qwen3 retrieval line. Recall lands between the 0.6B and the 8B, and so does the price. Use it when the 0.6B misses too much and the 8B's 4096-dimension vectors cost too much to store.","context_window":32768,"max_output_tokens":null,"best_for":"Balanced recall and index size","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.027,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.027,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2025-06-05","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"bge-large-en","object":"model","created":0,"owned_by":"baai","name":"BGE Large EN v1.5","description":"1024-dimension English embeddings, and still one of the most widely benchmarked retrieval models in production. The 512-token window forces chunking, which is why the newer entries above exist — but an index already built on it stays comparable.","context_window":512,"max_output_tokens":null,"best_for":"English-only retrieval, existing BGE indexes","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2023-09-12","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"bge-base-en","object":"model","created":0,"owned_by":"baai","name":"BGE Base EN v1.5","description":"768-dimension English embeddings — half the vector of bge-large-en at the same per-million rate, so an index built on it is half the storage and half the comparison cost. Retrieval quality gives up little on short chunks, which is what the 512-token window enforces anyway.","context_window":512,"max_output_tokens":null,"best_for":"English retrieval where index size matters","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.00675,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.00675,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2023-09-12","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"multilingual-e5-large","object":"model","created":0,"owned_by":"microsoft","name":"Multilingual E5 Large","description":"1024-dimension embeddings trained on 94 languages, and the most-cited multilingual retrieval baseline outside BGE. Pick it against bge-m3 when the corpus is short-chunk and the comparison is cross-language recall rather than window size.","context_window":512,"max_output_tokens":null,"best_for":"Cross-language retrieval on short chunks","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.0135,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2024-02-08","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"all-minilm-l12","object":"model","created":0,"owned_by":"sentence-transformers","name":"all-MiniLM-L12-v2","description":"384-dimension embeddings — the smallest vector on the shelf, and the default of the sentence-transformers library, so an enormous amount of existing code expects exactly this shape. Twelve layers where the L6 has six: better recall, still tiny.","context_window":512,"max_output_tokens":null,"best_for":"Semantic search at minimum index size","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.00675,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.00675,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"capable","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2021-08-30","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"all-minilm-l6","object":"model","created":0,"owned_by":"sentence-transformers","name":"all-MiniLM-L6-v2","description":"384 dimensions from six layers — the most downloaded embedding model there is, and the one most tutorials and starter repos hard-code. Reach for it to match an existing index or to keep a local prototype and a hosted one on the same vectors.","context_window":512,"max_output_tokens":null,"best_for":"Matching an existing MiniLM index","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.00675,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.00675,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"capable","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2021-08-30","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gte-base","object":"model","created":0,"owned_by":"alibaba","name":"GTE Base","description":"768-dimension general text embeddings from Alibaba's Tongyi lab, trained on a broader mixture than BGE and often a point or two ahead of it on out-of-domain retrieval. Same size and same price as bge-base-en, so the choice is which one your corpus likes.","context_window":512,"max_output_tokens":null,"best_for":"Out-of-domain English retrieval","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.00675,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.00675,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"embedding","release_stage":"stable","release_date":"2023-07-27","type":"embedding","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"hermes-3-405b","object":"model","created":0,"owned_by":"nousresearch","name":"Hermes 3 405B","description":"Nous Research's fine-tune of Llama 3.1 405B, tuned for steerability rather than refusal: it follows a system prompt further than the Instruct model it is built on, which is why roleplay, persona and agent-scaffold work keep reaching for it. Reliable function calling and structured output are part of the tune, not bolted on.","context_window":131072,"max_output_tokens":8192,"best_for":"Steerable assistants, persona work, structured output","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":1.35,"output":1.35,"cached":null},"list_price":{"mode":"per-token","input":1.35,"output":1.35,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":"open","weights_source":{"checkpoint":"NousResearch/Hermes-3-Llama-3.1-405B","source":"https://nousresearch.com/wp-content/uploads/2024/08/Hermes-3-Technical-Report.pdf","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/NousResearch/Hermes-3-Llama-3.1-405B","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2024-08-16","type":"language","tags":["tool-use"],"use_cases":["chat","build-agents","write-content"],"recommended_for":["chat","build-agents","write-content"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"hermes-3-70b","object":"model","created":0,"owned_by":"nousresearch","name":"Hermes 3 70B","description":"The 70B of the same tune, at 30% less per million. Keeps the steerability the 405B is chosen for and gives up depth on the hardest reasoning, which is the trade most persona and agent workloads are happy to make.","context_window":131072,"max_output_tokens":8192,"best_for":"Steerable assistants where the 405B is more than the job needs","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.945,"output":0.945,"cached":null},"list_price":{"mode":"per-token","input":0.945,"output":0.945,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":true,"supports_structured_outputs":true,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":"open","weights_source":{"checkpoint":"NousResearch/Hermes-3-Llama-3.1-70B","source":"https://nousresearch.com/wp-content/uploads/2024/08/Hermes-3-Technical-Report.pdf","read":"2026-08-14"},"reduced_precision":{"verdict":"reduced","levels":["fp8"],"native":{"formats":["bf16"],"source":"https://huggingface.co/NousResearch/Hermes-3-Llama-3.1-70B","read":"2026-08-14"},"measured":"2026-09-03"},"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2024-08-16","type":"language","tags":["tool-use"],"use_cases":["chat","build-agents","write-content"],"recommended_for":["chat","build-agents","write-content"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"qwen3-reranker-8b","object":"model","created":0,"owned_by":"alibaba","name":"Qwen3 Reranker 8B","description":"Scores how well each document answers a query, so a retriever's top 50 can be cut to the 5 worth sending to a model. Sits between search and synthesis: embeddings decide what to fetch, this decides what survives. A 40K window means whole documents can be judged without chunking.","context_window":40960,"max_output_tokens":null,"best_for":"Cutting retrieval noise before it reaches the context window","speed":"medium","recommended":false,"hot":false,"pricing":{"mode":"per-token","input":0.135,"output":0,"cached":null},"list_price":{"mode":"per-token","input":0.135,"output":0,"cached":null},"pricing_mode":"per-token","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"reranking","release_stage":"stable","release_date":"2025-06-05","type":"reranking","tags":[],"use_cases":["search-documents"],"recommended_for":["search-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"whisper-v3-turbo","object":"model","created":0,"owned_by":"openai","name":"Whisper Large v3 Turbo","description":"Speech-to-text. 228x realtime inference. Transcripts with timestamps + language detect.","context_window":3600,"max_output_tokens":null,"best_for":"Transcription, captions, voice agents","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.0009,"cached":null},"list_price":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.0009,"cached":null},"pricing_mode":"per-minute","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"stt","release_stage":"stable","release_date":"2024-10-01","type":"transcription","tags":[],"use_cases":["transcribe-audio"],"recommended_for":["transcribe-audio"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-4o-mini-transcribe-2025-12-15","object":"model","created":0,"owned_by":"openai","name":"GPT-4o mini Transcribe","description":"Speech-to-text. OpenAI's premium quality STT — best real-world accuracy on conversational audio, noisy backgrounds, and code-switching (Vi/En etc). RETIRING 2027-02-26 — OpenAI shuts down the gpt-4o-transcribe family; migrate to gpt-transcribe.","context_window":1500,"max_output_tokens":null,"best_for":"High-accuracy dictation, multilingual transcription, conversational audio","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.00405,"cached":null},"list_price":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.00405,"cached":null},"pricing_mode":"per-minute","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":"2027-02-26","tier":null,"capability":"stt","release_stage":"stable","release_date":"2024-03-13","type":"transcription","tags":[],"use_cases":["transcribe-audio","translate"],"recommended_for":["transcribe-audio","translate"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-transcribe","object":"model","created":0,"owned_by":"openai","name":"GPT Transcribe","description":"Speech-to-text. OpenAI's current premium STT and the replacement for the gpt-4o-transcribe family — high real-world accuracy on conversational audio, noisy backgrounds, and code-switching (Vi/En etc).","context_window":1500,"max_output_tokens":null,"best_for":"High-accuracy dictation, multilingual transcription, conversational audio","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.006075,"cached":null},"list_price":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.006075,"cached":null},"pricing_mode":"per-minute","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"stt","release_stage":"stable","release_date":"2026-08-26","type":"transcription","tags":[],"use_cases":["transcribe-audio","translate"],"recommended_for":["transcribe-audio","translate"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3.5-transcribe","object":"model","created":0,"owned_by":"google","name":"Gemini 3.5 Transcribe","description":"Speech-to-text. Google's dedicated file transcription model — 85+ languages, up to 1 hour per request. OpenAI-compatible /v1/audio/transcriptions.","context_window":3600,"max_output_tokens":null,"best_for":"File transcription, multilingual dictation, captions","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.00675,"cached":null},"list_price":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.00675,"cached":null},"pricing_mode":"per-minute","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"stt","release_stage":"stable","release_date":"2026-08-26","type":"transcription","tags":[],"use_cases":["transcribe-audio"],"recommended_for":["transcribe-audio"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3-flash-audio","object":"model","created":0,"owned_by":"google","name":"Gemini 3 Flash (Audio)","description":"Audio understanding. Hears tone, music, SFX, language, speaker emotion — beyond pure transcription. Inline payload up to 30 min.","context_window":1800,"max_output_tokens":4096,"best_for":"Audio scene understanding, mood/emotion detection, music recognition","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.002592,"cached":null},"list_price":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.002592,"cached":null},"pricing_mode":"per-minute","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["text"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"audio-understand","release_stage":"preview","release_date":"2025-12-17","type":"audio-understanding","tags":[],"use_cases":["analyse-documents"],"recommended_for":["analyse-documents"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-realtime-translate","object":"model","created":0,"owned_by":"openai","name":"GPT Realtime Translate","description":"Native audio-to-audio translation with voice cloning. Preserves original speaker tone. 13 target languages (es/pt/fr/ja/ru/zh/de/ko/hi/id/vi/it/en).","context_window":3600,"max_output_tokens":null,"best_for":"Live translation, dubbing, voice agents preserving speaker tone","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.0459,"cached":null},"list_price":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.0459,"cached":null},"pricing_mode":"per-minute","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"realtime","release_stage":"preview","release_date":"2026-05-07","type":"realtime","tags":["websocket-realtime"],"use_cases":["chat","translate","generate-speech"],"recommended_for":["chat","translate","generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-2.5-flash-native-audio-preview-12-2025","object":"model","created":0,"owned_by":"google","name":"Gemini 2.5 Flash Native Audio","description":"Conversational realtime with 30 pickable voices + 24 output languages. WebSocket-based, ephemeral token auth. Native audio understanding + generation in one round-trip. RETIRING 2026-10-20 — Google discontinues the Gemini 2.5 endpoint it runs on; gemini-3.1-flash-live-preview is the successor at the same per-minute price.","context_window":1800,"max_output_tokens":null,"best_for":"Conversational voice agents, multilingual chat, voice picker UX","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.03888,"cached":null},"list_price":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.03888,"cached":null},"pricing_mode":"per-minute","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":"2026-10-20","tier":null,"capability":"realtime","release_stage":"preview","release_date":"2025-12-12","type":"realtime","tags":["websocket-realtime"],"use_cases":["generate-speech","chat"],"recommended_for":["generate-speech","chat"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3.1-flash-live-preview","object":"model","created":0,"owned_by":"google","name":"Gemini 3.1 Flash Live","description":"Audio-to-audio realtime dialogue on the Gemini 3 line, at the same per-minute price as Gemini 2.5 Flash Native Audio. The successor to that model, which retires 2026-10-20. WebSocket-based; native audio in and out in one round-trip.","context_window":131072,"max_output_tokens":65536,"best_for":"Voice agents migrating off Gemini 2.5 native audio, low-latency spoken dialogue","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.03888,"cached":null},"list_price":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.03888,"cached":null},"pricing_mode":"per-minute","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"realtime","release_stage":"preview","release_date":"2026-03-26","type":"realtime","tags":["websocket-realtime"],"use_cases":["generate-speech","chat"],"recommended_for":["generate-speech","chat"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gemini-3.5-live-translate-preview","object":"model","created":0,"owned_by":"google","name":"Gemini 3.5 Live Translate","description":"Low-latency audio-to-audio speech translation. Near real-time speech-to-speech across 70+ languages, preserving the speaker's intonation, pacing, and pitch. WebSocket-based realtime session.","context_window":16384,"max_output_tokens":32768,"best_for":"Live speech-to-speech translation, multilingual voice agents, dubbing","speed":"fast","recommended":true,"hot":true,"pricing":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.06345,"cached":null},"list_price":{"mode":"per-minute","input":0,"output":0,"per_minute_usd":0.06345,"cached":null},"pricing_mode":"per-minute","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"realtime","release_stage":"preview","release_date":"2026-06-09","type":"realtime","tags":["websocket-realtime"],"use_cases":["chat","translate","generate-speech"],"recommended_for":["chat","translate","generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"eleven-multilingual-v2","object":"model","created":0,"owned_by":"elevenlabs","name":"ElevenLabs Multilingual v2","description":"Hero-quality multilingual TTS. 29 languages, expressive voices, brand-safe consistent delivery.","context_window":5000,"max_output_tokens":null,"best_for":"Narration, storytelling, brand voiceovers, multilingual content","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.405,"cached":null},"list_price":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.405,"cached":null},"pricing_mode":"per-char","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":"tts","release_stage":"stable","release_date":"2023-08-22","type":"speech","tags":[],"use_cases":["generate-speech","translate"],"recommended_for":["generate-speech","translate"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"eleven-v3","object":"model","created":0,"owned_by":"elevenlabs","name":"ElevenLabs v3","description":"Most expressive TTS. Emotional range, audio tags, and lifelike delivery across 70+ languages.","context_window":5000,"max_output_tokens":null,"best_for":"Expressive narration, character voices, emotional dialogue, premium voiceovers","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.405,"cached":null},"list_price":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.405,"cached":null},"pricing_mode":"per-char","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":"tts","release_stage":"stable","release_date":"2025-06-05","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"eleven-flash-v2-5","object":"model","created":0,"owned_by":"elevenlabs","name":"ElevenLabs Flash v2.5","description":"Ultra-low-latency TTS, ~75ms time-to-first-byte. Half the per-char cost of Multilingual v2. 32 languages.","context_window":5000,"max_output_tokens":null,"best_for":"Real-time voice agents, conversational AI, low-latency narration","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.2025,"cached":null},"list_price":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.2025,"cached":null},"pricing_mode":"per-char","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":"tts","release_stage":"stable","release_date":"2024-10-30","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"eleven-turbo-v2-5","object":"model","created":0,"owned_by":"elevenlabs","name":"ElevenLabs Turbo v2.5","description":"Balanced TTS — quicker than Multilingual, better quality than Flash. Half cost vs Multilingual. 32 languages.","context_window":5000,"max_output_tokens":null,"best_for":"Default balanced voice for medium-latency narration, podcasts","speed":"fast","recommended":false,"hot":false,"pricing":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.2025,"cached":null},"list_price":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.2025,"cached":null},"pricing_mode":"per-char","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":"tts","release_stage":"stable","release_date":"2024-07-18","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"elevenlabs-music","object":"model","created":0,"owned_by":"elevenlabs","name":"ElevenLabs Music","description":"Prompt-driven music generation. Lyrics, instrumental, configurable duration up to 5 min.","context_window":2000,"max_output_tokens":null,"best_for":"Background music, soundtracks, theme generation, royalty-free music","speed":"slow","recommended":true,"hot":true,"pricing":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.135,"cached":null},"list_price":{"mode":"per-second","input":0,"output":0,"per_second_usd":0.135,"cached":null},"pricing_mode":"per-second","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":"music","release_stage":"stable","release_date":"2025-08-05","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"elevenlabs-sfx","object":"model","created":0,"owned_by":"elevenlabs","name":"ElevenLabs Sound Effects","description":"Generates non-speech audio (whoosh, explosion, rain) from a text prompt. Flat $0.027 per generation, 0.5-22 sec.","context_window":500,"max_output_tokens":null,"best_for":"Sound effects, foley, ambient sounds, video game audio","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-call","input":0,"output":0,"per_call_usd":0.027,"cached":null},"list_price":{"mode":"per-call","input":0,"output":0,"per_call_usd":0.027,"cached":null},"pricing_mode":"per-call","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":"sfx","release_stage":"stable","release_date":"2024-06-24","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-speech-hd","object":"model","created":0,"owned_by":"minimax","name":"MiniMax Speech HD","description":"MiniMax HD voice. Multilingual, expressive, ~2.9× cheaper than ElevenLabs Multilingual v2 at the same production quality tier.","context_window":5000,"max_output_tokens":null,"best_for":"Production narration, multilingual content, budget-tier brand voice","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.07,"cached":null},"list_price":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.07,"cached":null},"pricing_mode":"per-char","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"cheap","capability":"tts","release_stage":"stable","release_date":"2025-04-02","type":"speech","tags":[],"use_cases":["generate-speech","translate"],"recommended_for":["generate-speech","translate"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-speech-turbo","object":"model","created":0,"owned_by":"minimax","name":"MiniMax Speech Turbo","description":"MiniMax low-latency voice. Multilingual, ~2.2× cheaper than ElevenLabs Flash v2.5. Best for bulk TTS, real-time voice agents, conversational AI.","context_window":5000,"max_output_tokens":null,"best_for":"Real-time voice agents, conversational AI, bulk narration, high-throughput TTS","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.04,"cached":null},"list_price":{"mode":"per-char","input":0,"output":0,"per_kchar_usd":0.04,"cached":null},"pricing_mode":"per-char","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"efficient","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"cheap","capability":"tts","release_stage":"stable","release_date":"2025-04-02","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-music","object":"model","created":0,"owned_by":"minimax","name":"MiniMax Music","description":"Lyrics-driven music generation. Music-2.0 family. Up to 5 minutes per call, ~90× cheaper than ElevenLabs Music for non-hero use cases.","context_window":2000,"max_output_tokens":null,"best_for":"Background music, social shorts, royalty-free soundtrack, bulk theme generation","speed":"slow","recommended":false,"hot":true,"pricing":{"mode":"per-song","input":0,"output":0,"per_song_usd":0.2,"cached":null},"list_price":{"mode":"per-song","input":0,"output":0,"per_song_usd":0.2,"cached":null},"pricing_mode":"per-song","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"cheap","capability":"music","release_stage":"stable","release_date":"2025-10-29","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-music-pro","object":"model","created":0,"owned_by":"minimax","name":"MiniMax Music Pro","description":"Music-2.6 (latest pro family). Higher fidelity than Music-2.0, richer arrangements. Still ~19× cheaper than ElevenLabs Music for production-tier output.","context_window":2000,"max_output_tokens":null,"best_for":"Premium background music, brand soundtracks, podcast intros, video scoring","speed":"slow","recommended":true,"hot":true,"pricing":{"mode":"per-song","input":0,"output":0,"per_song_usd":0.2,"cached":null},"list_price":{"mode":"per-song","input":0,"output":0,"per_song_usd":0.2,"cached":null},"pricing_mode":"per-song","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":"music","release_stage":"stable","release_date":"2026-04-10","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-voice-clone","object":"model","created":0,"owned_by":"minimax","name":"MiniMax Voice Clone","description":"Clone a voice from a 10s-5min reference recording. Returns a voice_id usable in /v1/audio/speech with any MiniMax HD/Turbo SKU. Flat one-time charge per cloned voice.","context_window":0,"max_output_tokens":null,"best_for":"Brand voice cloning, character voices, custom narrator profiles","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-call","input":0,"output":0,"per_call_usd":2,"cached":null},"list_price":{"mode":"per-call","input":0,"output":0,"per_call_usd":2,"cached":null},"pricing_mode":"per-call","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["audio"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-04-02","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-voice-design","object":"model","created":0,"owned_by":"minimax","name":"MiniMax Voice Design","description":"Generate a synthesized voice profile from a natural-language description (no reference audio needed). Returns a voice_id usable in /v1/audio/speech with any MiniMax HD/Turbo SKU. Flat one-time charge per designed voice.","context_window":1000,"max_output_tokens":null,"best_for":"Branding when no voice talent available, fictional characters, persona voices from text alone","speed":"medium","recommended":false,"hot":true,"pricing":{"mode":"per-call","input":0,"output":0,"per_call_usd":2,"cached":null},"list_price":{"mode":"per-call","input":0,"output":0,"per_call_usd":2,"cached":null},"pricing_mode":"per-call","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["audio"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":null,"capability":null,"release_stage":"stable","release_date":"2025-06-23","type":"speech","tags":[],"use_cases":["generate-speech"],"recommended_for":["generate-speech"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"gpt-image-2","object":"model","created":0,"owned_by":"openai","name":"GPT Image 2","description":"OpenAI's flagship image model (Apr 2026). Near-perfect text-in-image (multilingual), reasoning-augmented composition, photorealism, logo-grade lettering. Quality tiers low/medium/high — picker default medium. 1024×1024 / 1024×1536 / 1536×1024 / 2048×2048.","context_window":0,"max_output_tokens":null,"best_for":"Text-in-image, photoreal, multilingual typography","speed":"medium","recommended":true,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.072,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.072,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":true,"input_modalities":["text","image"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"medium","cost_tier":"balanced","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"quality","capability":null,"release_stage":"stable","release_date":"2026-04-21","type":"image","tags":["vision","image-generation"],"use_cases":["generate-images"],"recommended_for":["generate-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"minimax-image-01","object":"model","created":0,"owned_by":"minimax","name":"MiniMax Image 01","description":"Sub-cent image generation. Cheapest tier on Kyma — $0.005 per image flat regardless of resolution. 5 aspect ratios (1:1, 16:9, 9:16, 4:3, 3:4). Best for high-volume / budget workflows.","context_window":0,"max_output_tokens":null,"best_for":"Bulk image generation, social shorts, stock-style assets, budget UI mockups","speed":"fast","recommended":false,"hot":true,"pricing":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.005,"cached":null},"list_price":{"mode":"per-image","input":0,"output":0,"per_image_usd":0.005,"cached":null},"pricing_mode":"per-image","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text"],"output_modalities":["image"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"fast","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"cheap","capability":null,"release_stage":"stable","release_date":"2025-02-15","type":"image","tags":["image-generation"],"use_cases":["generate-images"],"recommended_for":["generate-images"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"hailuo-02-512p","object":"model","created":0,"owned_by":"minimax","name":"Hailuo 02 (512p)","description":"MiniMax Hailuo 02 at 512p — cheapest video tier on Kyma. Flat $0.140 per clip (6s or 10s). T2V or I2V via image_url. Best for social shorts, rapid iteration, budget motion.","context_window":0,"max_output_tokens":null,"best_for":"Social shorts, motion mockups, rapid iteration, budget video workflows","speed":"slow","recommended":false,"hot":true,"pricing":{"mode":"per-video","input":0,"output":0,"per_video_usd":0.14,"cached":null},"list_price":{"mode":"per-video","input":0,"output":0,"per_video_usd":0.14,"cached":null},"pricing_mode":"per-video","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"cheap","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"cheap","capability":null,"release_stage":"stable","release_date":"2025-06-18","type":"video","tags":["video-generation"],"use_cases":["generate-video","write-content"],"recommended_for":["generate-video","write-content"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"hailuo-02-768p","object":"model","created":0,"owned_by":"minimax","name":"Hailuo 02 (768p)","description":"Hailuo 02 at 768p — mid tier balanced quality vs cost. Flat $0.420 per clip. T2V or I2V via image_url.","context_window":0,"max_output_tokens":null,"best_for":"Brand b-roll, product motion, character shots at production quality","speed":"slow","recommended":false,"hot":true,"pricing":{"mode":"per-video","input":0,"output":0,"per_video_usd":0.42,"cached":null},"list_price":{"mode":"per-video","input":0,"output":0,"per_video_usd":0.42,"cached":null},"pricing_mode":"per-video","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"balanced","quality_tier":"strong","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":null,"release_stage":"stable","release_date":"2025-06-18","type":"video","tags":["video-generation"],"use_cases":["generate-video"],"recommended_for":["generate-video"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true},{"id":"hailuo-02-1080p","object":"model","created":0,"owned_by":"minimax","name":"Hailuo 02 (1080p)","description":"Hailuo 02 at 1080p — premium tier, full HD output. Flat $0.780 per clip. T2V or I2V via image_url.","context_window":0,"max_output_tokens":null,"best_for":"Hero brand video, premium social campaigns, high-fidelity character motion","speed":"slow","recommended":true,"hot":true,"pricing":{"mode":"per-video","input":0,"output":0,"per_video_usd":0.78,"cached":null},"list_price":{"mode":"per-video","input":0,"output":0,"per_video_usd":0.78,"cached":null},"pricing_mode":"per-video","promotion":null,"supports_caching":false,"supports_tools":false,"supports_structured_outputs":false,"supports_reasoning":false,"supports_vision":false,"input_modalities":["text","image"],"output_modalities":["video"],"supported_parameters":["temperature","top_p","max_tokens","stream","tools","response_format","structured_outputs","prompt_caching"],"latency_tier":"slow","cost_tier":"premium","quality_tier":"frontier-open","weights":null,"weights_source":null,"reduced_precision":null,"retires_on":null,"tier":"fast","capability":null,"release_stage":"stable","release_date":"2025-06-18","type":"video","tags":["video-generation"],"use_cases":["generate-video"],"recommended_for":["generate-video"],"gateway_output_limit":null,"output_limit_source":"upstream_model","max_tokens_passthrough":true}],"aliases":{"best":"qwen-3.6-plus","fast":"gemini-3.5-flash-lite","code":"qwen-3-coder","cheap":"deepseek-v4-flash","long-context":"gemini-3-flash","vision":"gemma-4-31b","reasoning":"deepseek-r1","agent":"kimi-k2.6","best-agent":"kimi-k2.6","balanced":"llama-3.3-70b","glm-flagship":"glm-5.2","search":"sonar","transcribe":"whisper-v3-turbo","transcribe-quality":"gpt-4o-mini-transcribe-2025-12-15","audio-understand":"gemini-3-flash-audio"},"filters":{"type":[],"tags":[],"use_cases":[],"creator":[],"recommended_for":[],"input_modalities":[],"output_modalities":[],"supported_parameters":[],"latency_tier":null,"cost_tier":null,"quality_tier":null,"tier":null,"capability":null,"release_stage":null,"tools":null,"structured_outputs":null,"vision":null,"reasoning":null,"min_context_window":null,"max_input_price":null},"_backend":"cloudflare-workers"}