[{"slug":"anthropic-claude-fable-5-1","name":"Claude Fable 5.1","provider":"Anthropic","api_id":"claude-fable-5-1","status":"ga","context_window":1000000,"max_output_tokens":128000,"input_price":10,"output_price":50,"cached_input_price":0.25,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":true,"knowledge_cutoff":"2026-06","license":"proprietary","released":"2026-09-01","best_for":"The reach-for-it model when Opus 5 at high effort still fails your evals: hardest reasoning and multi-hour agentic runs. Thinking is always on, forced tool_choice returns 400, and cache reads are 2.5% of input ($0.25/MTok) - the cheapest cache in the lineup, so it pays off hardest on long stateful agents."},{"slug":"anthropic-claude-mythos-5-1","name":"Claude Mythos 5.1","provider":"Anthropic","api_id":"claude-mythos-5-1","status":"preview","context_window":1000000,"max_output_tokens":128000,"input_price":10,"output_price":50,"cached_input_price":0.25,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-06","license":"proprietary","released":"2026-09-01","best_for":"Invitation-only twin of Fable 5.1 under Project Glasswing for defensive cybersecurity work - identical specs and price, but without the safety classifiers that make Fable decline some security requests. Not self-serve; requires an Anthropic/AWS/Google Cloud account team."},{"slug":"anthropic-claude-fable-5","name":"Claude Fable 5","provider":"Anthropic","api_id":"claude-fable-5","status":"ga","context_window":1000000,"max_output_tokens":128000,"input_price":10,"output_price":50,"cached_input_price":1,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-01","license":"proprietary","released":"2026-06-09","best_for":"Legacy (still served, retirement not before June 9 2027). Same list price as Fable 5.1 but 4x the cache-read cost ($1 vs $0.25) and a five-month-older knowledge cutoff - there is no cost reason to start new work here; migrate to Fable 5.1 unless you depend on forced tool_choice, which 5.1 removed."},{"slug":"anthropic-claude-mythos-5","name":"Claude Mythos 5","provider":"Anthropic","api_id":"claude-mythos-5","status":"preview","context_window":1000000,"max_output_tokens":128000,"input_price":10,"output_price":50,"cached_input_price":1,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-01","license":"proprietary","released":"2026-06-09","best_for":"Legacy invite-only Project Glasswing model for cybersecurity and biology research; superseded by Mythos 5.1, which costs the same and reads cache at a quarter the price."},{"slug":"anthropic-claude-opus-5","name":"Claude Opus 5","provider":"Anthropic","api_id":"claude-opus-5","status":"ga","context_window":1000000,"max_output_tokens":128000,"input_price":5,"output_price":25,"cached_input_price":0.5,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-05","license":"proprietary","released":"2026-07-24","best_for":"The default choice for complex agentic coding and enterprise work - half the price of Fable at 1M context, and Anthropic's own recommended starting point. Adaptive thinking is ON by default (a break from Opus 4.8) and can only be disabled at effort high or below; 300K output via the Batch API beta; the only model besides Opus 4.8 with fast mode."},{"slug":"anthropic-claude-opus-4-8","name":"Claude Opus 4.8","provider":"Anthropic","api_id":"claude-opus-4-8","status":"ga","context_window":1000000,"max_output_tokens":128000,"input_price":5,"output_price":25,"cached_input_price":0.5,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-01","license":"proprietary","released":"2026-05-28","best_for":"Legacy but identically priced to Opus 5, so keep it only as a pinned snapshot for behaviour-sensitive workloads or as a refusal-fallback target. Note thinking is OFF unless you explicitly send thinking:{type:\"adaptive\"} - the opposite of Opus 5. Retirement not before May 28 2027."},{"slug":"anthropic-claude-opus-4-7","name":"Claude Opus 4.7","provider":"Anthropic","api_id":"claude-opus-4-7","status":"ga","context_window":1000000,"max_output_tokens":128000,"input_price":5,"output_price":25,"cached_input_price":0.5,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-01","license":"proprietary","released":"2026-04-16","best_for":"Legacy; the model that introduced the current tokenizer (~30% more tokens for the same text than 4.6 and earlier, so re-baseline cost when migrating either direction). Fast mode was removed here - speed:\"fast\" returns an error. No reason to pick it over Opus 5 at the same price."},{"slug":"anthropic-claude-opus-4-6","name":"Claude Opus 4.6","provider":"Anthropic","api_id":"claude-opus-4-6","status":"ga","context_window":1000000,"max_output_tokens":128000,"input_price":5,"output_price":25,"cached_input_price":0.5,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2025-05","license":"proprietary","released":"2026-02-05","best_for":"Legacy; the last model that still accepts the old manual thinking budget_tokens escape hatch and temperature/top_p sampling, which makes it the migration bridge for code not yet ported to adaptive thinking + effort. Uses the pre-4.7 tokenizer."},{"slug":"anthropic-claude-opus-4-5","name":"Claude Opus 4.5","provider":"Anthropic","api_id":"claude-opus-4-5-20251101","status":"ga","context_window":200000,"max_output_tokens":64000,"input_price":5,"output_price":25,"cached_input_price":0.5,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2025-05","license":"proprietary","released":"2025-11-24","best_for":"Legacy, alias claude-opus-4-5. Opus pricing but only 200K context and 64K output, so it is strictly worse value than Opus 5 - keep only for pinned reproducibility. Earliest retirement November 24 2026, the nearest of any active Opus."},{"slug":"anthropic-claude-sonnet-5","name":"Claude Sonnet 5","provider":"Anthropic","api_id":"claude-sonnet-5","status":"ga","context_window":1000000,"max_output_tokens":128000,"input_price":2,"output_price":10,"cached_input_price":0.2,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-01","license":"proprietary","released":"2026-06-30","best_for":"Best price/intelligence ratio in the lineup and the right default for high-volume production traffic: 1M context and 128K output at 40% of Opus 5's price. The $2/$10 launch price is now permanent (the planned Sept 1 2026 rise to $3/$15 was cancelled). Note it does NOT support mid-conversation system messages."},{"slug":"anthropic-claude-sonnet-4-6","name":"Claude Sonnet 4.6","provider":"Anthropic","api_id":"claude-sonnet-4-6","status":"ga","context_window":1000000,"max_output_tokens":128000,"input_price":3,"output_price":15,"cached_input_price":0.3,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2025-08","license":"proprietary","released":"2026-02-17","best_for":"Legacy and 50% more expensive than Sonnet 5 for the same 1M/128K envelope - migrate unless you rely on deprecated budget_tokens thinking or sampling parameters, which Sonnet 5 rejects with a 400. Uses the pre-4.7 tokenizer."},{"slug":"anthropic-claude-sonnet-4-5","name":"Claude Sonnet 4.5","provider":"Anthropic","api_id":"claude-sonnet-4-5-20250929","status":"ga","context_window":200000,"max_output_tokens":64000,"input_price":3,"output_price":15,"cached_input_price":0.3,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2025-01","license":"proprietary","released":"2025-09-29","best_for":"Legacy, alias claude-sonnet-4-5. 200K context, 64K output, no effort parameter, and 1.5x the price of Sonnet 5 - migrate. Retires no sooner than September 29 2026, the nearest retirement date of any active model, so audit for it now."},{"slug":"anthropic-claude-haiku-4-5","name":"Claude Haiku 4.5","provider":"Anthropic","api_id":"claude-haiku-4-5-20251001","status":"ga","context_window":200000,"max_output_tokens":64000,"input_price":1,"output_price":5,"cached_input_price":0.1,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2025-02","license":"proprietary","released":"2025-10-15","best_for":"Cheapest and fastest current Claude - the pick for classification, extraction, routing, sub-agent workers and LLM judges at volume (~$37 per 10,000 support-ticket conversations). Caveats: 200K context, no effort parameter, and it still uses old-style manual extended thinking with budget_tokens rather than adaptive. Alias claude-haiku-4-5."},{"slug":"anthropic-claude-mythos-preview","name":"Claude Mythos Preview","provider":"Anthropic","api_id":"claude-mythos-preview","status":"deprecated","context_window":1000000,"max_output_tokens":0,"input_price":25,"output_price":125,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"Deprecated invitation-only research-preview model for Project Glasswing defensive-cybersecurity work; still reachable for invited customers (including on Amazon Bedrock) but migrate to Claude Mythos 5/5.1. No price, context window or output limit is published on any official Anthropic page - all numeric fields are recorded as unknown rather than guessed. It runs no safety classifiers and rejects non-default temperature/top_p/top_k."},{"slug":"anthropic-claude-opus-4-1","name":"Claude Opus 4.1","provider":"Anthropic","api_id":"claude-opus-4-1-20250805","status":"retired","context_window":200000,"max_output_tokens":0,"input_price":15,"output_price":75,"cached_input_price":1.5,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-08-05","best_for":"Retired on the Claude API (August 5 2026) but STILL SERVED on Amazon Bedrock (us. inference profile only) and Google Cloud, where those partners set their own dates. At $15/$75 it is 3x Opus 5's price for a 200K window - only for organisations pinned to it by compliance. Bedrock ID anthropic.claude-opus-4-1-20250805-v1:0. Max output tokens not published on the pages consulted."},{"slug":"anthropic-claude-opus-4","name":"Claude Opus 4","provider":"Anthropic","api_id":"claude-opus-4-20250514","status":"retired","context_window":200000,"max_output_tokens":0,"input_price":15,"output_price":75,"cached_input_price":1.5,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-05-14","best_for":"Retired on the Claude API (June 15 2026) and on Amazon Bedrock; per Anthropic's pricing page it remains available only on Google Cloud Vertex AI. Recommended replacement is claude-opus-4-8. Max output tokens not published on the pages consulted."},{"slug":"anthropic-claude-sonnet-4","name":"Claude Sonnet 4","provider":"Anthropic","api_id":"claude-sonnet-4-20250514","status":"retired","context_window":200000,"max_output_tokens":0,"input_price":3,"output_price":15,"cached_input_price":0.3,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-05-14","best_for":"Retired on the Claude API (June 15 2026); still served, marked deprecated, on Amazon Bedrock (global/us/eu/apac profiles) and Google Cloud. Same $3/$15 as Sonnet 4.5 with a 200K window - replace with claude-sonnet-5 at $2/$10. Bedrock ID anthropic.claude-sonnet-4-20250514-v1:0. Max output tokens not published on the pages consulted."},{"slug":"anthropic-claude-haiku-3-5","name":"Claude Haiku 3.5","provider":"Anthropic","api_id":"claude-3-5-haiku-20241022","status":"retired","context_window":200000,"max_output_tokens":0,"input_price":0.8,"output_price":4,"cached_input_price":0.08,"modalities_in":["text","image"],"capabilities":["tools","vision","prompt-caching","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-10-22","best_for":"Retired on the Claude API (February 19 2026); still served, marked deprecated, on Amazon Bedrock (us. profile) and Google Cloud. Nominally the cheapest Claude at $0.80/$4, but Haiku 4.5 is far more capable for $1/$5 and, unlike this model, does not charge cache reads against your ITPM rate limit. Max output tokens not published on the pages consulted."},{"slug":"openai-gpt-6-astra","name":"GPT-6 Astra","provider":"OpenAI","api_id":"gpt-6-astra","status":"ga","context_window":1050000,"max_output_tokens":128000,"input_price":10,"output_price":50,"cached_input_price":1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":true,"knowledge_cutoff":"2026-04-30","license":"proprietary","released":"2026-09-03","best_for":"The default pick for hard end-to-end agentic work — long-horizon coding, computer use and research. Uniquely supports reasoning.effort up to xhigh/max, async tool calling and mid-turn steering; no temperature/top_p/logprobs, and tool calling requires the Responses API. Cache writes cost $12.50/1M; >272K-token prompts bill at 2x input / 1.5x output."},{"slug":"openai-gpt-5-6-sol","name":"GPT-5.6 Sol","provider":"OpenAI","api_id":"gpt-5.6-sol","status":"ga","context_window":1050000,"max_output_tokens":128000,"input_price":4,"output_price":20,"cached_input_price":0.4,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-02-16","license":"proprietary","released":"2026-07-09","best_for":"Best value at the frontier tier — 60% cheaper than GPT-6 Astra with the same 1.05M context and full tool surface; the migration target OpenAI names for retiring gpt-5, o3 and gpt-4 workloads. Prices shown are promotional at least through 2026-11-21."},{"slug":"openai-gpt-5-6-terra","name":"GPT-5.6 Terra","provider":"OpenAI","api_id":"gpt-5.6-terra","status":"ga","context_window":1050000,"max_output_tokens":128000,"input_price":2,"output_price":12,"cached_input_price":0.2,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-02-16","license":"proprietary","released":"2026-07-09","best_for":"The intelligence-per-dollar sweet spot: half the price of Sol but keeps 1.05M context, computer use and the full hosted-tool set — the right default for production agents that don't need frontier reasoning."},{"slug":"openai-gpt-5-6-luna","name":"GPT-5.6 Luna","provider":"OpenAI","api_id":"gpt-5.6-luna","status":"ga","context_window":1050000,"max_output_tokens":128000,"input_price":0.2,"output_price":1.2,"cached_input_price":0.02,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-02-16","license":"proprietary","released":"2026-07-09","best_for":"High-volume classification, extraction and routing: a nano-tier price with a 1.05M context window and reasoning support, which no earlier nano model offered."},{"slug":"openai-gpt-5-6-cyber","name":"GPT-5.6 Cyber","provider":"OpenAI","api_id":"gpt-5.6-cyber","status":"ga","context_window":400000,"max_output_tokens":128000,"input_price":12.5,"output_price":75,"cached_input_price":1.25,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-02-16","license":"proprietary","released":"2026-08-07","best_for":"Authorized vulnerability reproduction and exploit validation only — requires Daybreak Red approval, and no Batch endpoint. Priced at a large premium over Sol for the same class of reasoning."},{"slug":"openai-daybreak-blue-alias","name":"Daybreak Blue (alias)","provider":"OpenAI","api_id":"gpt-daybreak-blue-latest","status":"ga","context_window":1050000,"max_output_tokens":128000,"input_price":4,"output_price":20,"cached_input_price":0.4,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-02-16","license":"proprietary","released":"2026-08-07","best_for":"Defensive security work (secure code review, detection engineering, IR) under Daybreak approval — an auto-updating alias that currently resolves to gpt-5.6-sol and is billed at whatever model it points to."},{"slug":"openai-daybreak-red-alias","name":"Daybreak Red (alias)","provider":"OpenAI","api_id":"gpt-daybreak-red-latest","status":"ga","context_window":400000,"max_output_tokens":128000,"input_price":12.5,"output_price":75,"cached_input_price":1.25,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2026-02-16","license":"proprietary","released":"2026-08-07","best_for":"Separately-approved offensive-security alias, currently pointing at gpt-5.6-cyber; use it if you want to track the newest cyber model without changing code."},{"slug":"openai-gpt-5-5-cyber","name":"GPT-5.5 Cyber","provider":"OpenAI","api_id":"gpt-5.5-cyber","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":12.5,"output_price":75,"cached_input_price":1.25,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"Previous-generation Daybreak Red model, still billed on the pricing page at the same rate as gpt-5.6-cyber — no reason to pick it over 5.6 Cyber. It has no public model page, so context window and cutoff are unpublished."},{"slug":"openai-gpt-5-5","name":"GPT-5.5","provider":"OpenAI","api_id":"gpt-5.5","status":"ga","context_window":1050000,"max_output_tokens":128000,"input_price":5,"output_price":30,"cached_input_price":0.5,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2025-12-01","license":"proprietary","released":"2026-04-24","best_for":"Previous frontier model, now more expensive than GPT-5.6 Sol for less capability — keep it only for pinned workloads already validated against it."},{"slug":"openai-gpt-5-5-pro","name":"GPT-5.5 Pro","provider":"OpenAI","api_id":"gpt-5.5-pro","status":"ga","context_window":1050000,"max_output_tokens":128000,"input_price":30,"output_price":180,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","structured-output","web-search","code-execution"],"flagship":true,"knowledge_cutoff":"2025-12-01","license":"proprietary","released":"2026-04-24","best_for":"Maximum-compute single answers where being right beats being cheap (hard proofs, deep analysis). No cached-input discount at all, so avoid it for anything with a repeated prompt prefix."},{"slug":"openai-gpt-5-4","name":"GPT-5.4","provider":"OpenAI","api_id":"gpt-5.4","status":"ga","context_window":1050000,"max_output_tokens":128000,"input_price":2.5,"output_price":15,"cached_input_price":0.25,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2025-08-31","license":"proprietary","released":"2026-03-05","best_for":"A cheaper 1.05M-context reasoning model than GPT-5.5, but GPT-5.6 Terra now undercuts it at $2/$12 with the same context — migrate rather than start here."},{"slug":"openai-gpt-5-4-mini","name":"GPT-5.4 Mini","provider":"OpenAI","api_id":"gpt-5.4-mini","status":"ga","context_window":400000,"max_output_tokens":128000,"input_price":0.75,"output_price":4.5,"cached_input_price":0.075,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"2025-08-31","license":"proprietary","released":"2026-03-17","best_for":"Subagent and computer-use workhorse in the mini tier; only worth it over GPT-5.6 Terra ($2/$12, 1.05M ctx) when your volume makes the 2.6x cheaper input decisive."},{"slug":"openai-gpt-5-4-nano","name":"GPT-5.4 nano","provider":"OpenAI","api_id":"gpt-5.4-nano","status":"ga","context_window":400000,"max_output_tokens":128000,"input_price":0.2,"output_price":1.25,"cached_input_price":0.02,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2025-08-31","license":"proprietary","released":"2026-03-17","best_for":"Simple high-volume tasks; priced identically to GPT-5.6 Luna on input but slightly more on output with a smaller context — prefer Luna for new builds."},{"slug":"openai-gpt-5-4-pro","name":"GPT-5.4 Pro","provider":"OpenAI","api_id":"gpt-5.4-pro","status":"ga","context_window":1050000,"max_output_tokens":128000,"input_price":30,"output_price":180,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","structured-output","web-search","computer-use"],"flagship":false,"knowledge_cutoff":"2025-08-31","license":"proprietary","released":"2026-03-05","best_for":"Pro-tier reasoning at the same headline price as GPT-5.5 Pro but an older cutoff; no cached-input discount and no code interpreter. Batch is available and halves the cost."},{"slug":"openai-gpt-5-3-codex","name":"GPT-5.3-Codex","provider":"OpenAI","api_id":"gpt-5.3-codex","status":"ga","context_window":400000,"max_output_tokens":128000,"input_price":1.75,"output_price":14,"cached_input_price":0.175,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2025-08-31","license":"proprietary","released":"2026-02-24","best_for":"The only Codex-branded model still served — long-horizon agentic coding in Codex CLI or a similar harness. Responses API only, no Batch; Fast mode doubles it to $3.50/$28."},{"slug":"openai-chat-latest","name":"Chat Latest","provider":"OpenAI","api_id":"chat-latest","status":"ga","context_window":400000,"max_output_tokens":128000,"input_price":5,"output_price":30,"cached_input_price":0.5,"modalities_in":["text","image"],"capabilities":["tools","vision","prompt-caching","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2025-08-31","license":"proprietary","released":"2026-05","best_for":"Reproducing ChatGPT's non-reasoning \"Instant\" personality in your own product. It is a moving alias with no version pinning and no Batch support, so don't build evals against it."},{"slug":"openai-gpt-5-search-api","name":"GPT-5 Search API","provider":"OpenAI","api_id":"gpt-5-search-api","status":"ga","context_window":0,"max_output_tokens":0,"input_price":1.25,"output_price":10,"cached_input_price":0.125,"modalities_in":["text"],"capabilities":["web-search","prompt-caching","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"The search-specialised GPT-5 variant sold for grounded question answering; billed like gpt-5 plus web-search call fees. It appears only in the pricing table, so context window and cutoff are unpublished."},{"slug":"openai-gpt-5-2","name":"GPT-5.2","provider":"OpenAI","api_id":"gpt-5.2","status":"ga","context_window":400000,"max_output_tokens":128000,"input_price":1.75,"output_price":14,"cached_input_price":0.175,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2025-08-31","license":"proprietary","released":"2025-12-11","best_for":"Still served and not yet deprecated, but GPT-5.6 Terra is cheaper on both input and output with 2.6x the context — treat 5.2 as a pinned-snapshot option only."},{"slug":"openai-gpt-5-2-pro","name":"GPT-5.2 Pro","provider":"OpenAI","api_id":"gpt-5.2-pro","status":"ga","context_window":400000,"max_output_tokens":128000,"input_price":21,"output_price":168,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","streaming","web-search"],"flagship":false,"knowledge_cutoff":"2025-08-31","license":"proprietary","released":"2025-12-11","best_for":"Cheapest of the currently-served pro tier ($21/$168 vs $30/$180), useful if you need max-compute answers on a budget and can live with the Aug-2025 cutoff. No cached-input discount."},{"slug":"openai-gpt-5-1","name":"GPT-5.1","provider":"OpenAI","api_id":"gpt-5.1","status":"ga","context_window":400000,"max_output_tokens":128000,"input_price":1.25,"output_price":10,"cached_input_price":0.125,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2025-11-13","best_for":"Legacy pinned workloads; unlike gpt-5 it carries no announced shutdown date yet, but its Sep-2024 cutoff is now two years stale."},{"slug":"openai-gpt-5","name":"GPT-5","provider":"OpenAI","api_id":"gpt-5","status":"deprecated","context_window":400000,"max_output_tokens":128000,"input_price":1.25,"output_price":10,"cached_input_price":0.125,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2025-08-07","best_for":"Migrate off: gpt-5-2025-08-07 shuts down 2026-12-11 with gpt-5.6-sol as the named replacement."},{"slug":"openai-gpt-5-mini","name":"GPT-5 Mini","provider":"OpenAI","api_id":"gpt-5-mini","status":"deprecated","context_window":400000,"max_output_tokens":128000,"input_price":0.25,"output_price":2,"cached_input_price":0.025,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2024-05-31","license":"proprietary","released":"2025-08-07","best_for":"Migrate off: shuts down 2026-12-11, replacement gpt-5.6-terra. Still the cheapest 400K-context reasoning model in the GPT-5 line until then."},{"slug":"openai-gpt-5-nano","name":"GPT-5 nano","provider":"OpenAI","api_id":"gpt-5-nano","status":"deprecated","context_window":400000,"max_output_tokens":128000,"input_price":0.05,"output_price":0.4,"cached_input_price":0.005,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","code-execution"],"flagship":false,"knowledge_cutoff":"2024-05-31","license":"proprietary","released":"2025-08-07","best_for":"The cheapest model OpenAI still serves ($0.05/$0.40) — but it shuts down 2026-12-11, and GPT-5.6 Luna is the named successor at 4x the input price."},{"slug":"openai-gpt-5-pro","name":"GPT-5 Pro","provider":"OpenAI","api_id":"gpt-5-pro","status":"deprecated","context_window":400000,"max_output_tokens":272000,"input_price":15,"output_price":120,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output","web-search"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2025-10-06","best_for":"Shuts down 2026-12-11 (replacement: gpt-5.6-sol with reasoning.mode pro). Notable for a 272K max output — the largest of any OpenAI model."},{"slug":"openai-gpt-4-1","name":"GPT-4.1","provider":"OpenAI","api_id":"gpt-4.1","status":"ga","context_window":1047576,"max_output_tokens":32768,"input_price":2,"output_price":8,"cached_input_price":0.5,"modalities_in":["text","image"],"capabilities":["tools","vision","prompt-caching","batch","streaming","structured-output","fine-tuning","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"proprietary","released":"2025-04-14","best_for":"The best remaining non-reasoning long-context model, and one of the few still fine-tunable — pick it when you need deterministic latency and no reasoning tokens."},{"slug":"openai-gpt-4-1-mini","name":"GPT-4.1 Mini","provider":"OpenAI","api_id":"gpt-4.1-mini","status":"ga","context_window":1047576,"max_output_tokens":32768,"input_price":0.4,"output_price":1.6,"cached_input_price":0.1,"modalities_in":["text","image"],"capabilities":["tools","vision","batch","streaming","structured-output","fine-tuning","web-search"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"proprietary","released":"2025-04-14","best_for":"Cheap 1M-context non-reasoning model that still supports fine-tuning; a good base for distillation before the fine-tuning platform closes on 2027-01-06."},{"slug":"openai-gpt-4-1-nano","name":"GPT-4.1 nano","provider":"OpenAI","api_id":"gpt-4.1-nano","status":"deprecated","context_window":1047576,"max_output_tokens":32768,"input_price":0.1,"output_price":0.4,"cached_input_price":0.025,"modalities_in":["text","image"],"capabilities":["tools","vision","prompt-caching","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"proprietary","released":"2025-04-14","best_for":"Shuts down 2026-10-23 (replacement gpt-5.6-luna); fine-tuned derivatives (ft-gpt-4.1-nano) die on the same date, so plan retraining now."},{"slug":"openai-gpt-4o","name":"GPT-4o","provider":"OpenAI","api_id":"gpt-4o","status":"ga","context_window":128000,"max_output_tokens":16384,"input_price":2.5,"output_price":10,"cached_input_price":1.25,"modalities_in":["text","image"],"capabilities":["tools","vision","batch","streaming","structured-output","fine-tuning","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2024-05-13","best_for":"Legacy multimodal chat only. Its cached-input discount is a weak 50% (vs 90% on GPT-5.x), and the gpt-4o-2024-05-13 snapshot shuts down 2026-10-23; the alias defaults to gpt-4o-2024-08-06."},{"slug":"openai-gpt-4o-mini","name":"GPT-4o Mini","provider":"OpenAI","api_id":"gpt-4o-mini","status":"ga","context_window":128000,"max_output_tokens":16384,"input_price":0.15,"output_price":0.6,"cached_input_price":0.075,"modalities_in":["text","image"],"capabilities":["tools","vision","batch","streaming","structured-output","fine-tuning","web-search"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2024-07-18","best_for":"Still the cheapest fine-tunable vision model ($3/1M training tokens), but for inference GPT-5.6 Luna is better and only slightly dearer."},{"slug":"openai-o1","name":"o1","provider":"OpenAI","api_id":"o1","status":"deprecated","context_window":200000,"max_output_tokens":100000,"input_price":15,"output_price":60,"cached_input_price":7.5,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2024-12-17","best_for":"Shuts down 2026-10-23 (replacement gpt-5.6-sol). Nothing recommends it now: 4x Sol's input price for weaker reasoning and a 2023 cutoff."},{"slug":"openai-o1-pro","name":"o1-pro","provider":"OpenAI","api_id":"o1-pro","status":"deprecated","context_window":200000,"max_output_tokens":100000,"input_price":150,"output_price":600,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","structured-output"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2025-03-19","best_for":"The most expensive model OpenAI has ever served ($150/$600) and it shuts down 2026-10-23 — migrate to gpt-5.6-sol with reasoning.mode pro immediately."},{"slug":"openai-o3","name":"o3","provider":"OpenAI","api_id":"o3","status":"deprecated","context_window":200000,"max_output_tokens":100000,"input_price":2,"output_price":8,"cached_input_price":0.5,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"proprietary","released":"2025-04-16","best_for":"o3-2025-04-16 shuts down 2026-12-11 (replacement gpt-5.6-sol). One of the few older models with a Flex tier ($1/$0.25/$4)."},{"slug":"openai-o3-pro","name":"o3-pro","provider":"OpenAI","api_id":"o3-pro","status":"deprecated","context_window":200000,"max_output_tokens":100000,"input_price":20,"output_price":80,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","structured-output"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"proprietary","released":"2025-06-10","best_for":"Shuts down 2026-12-11. Cheapest pro-class option still live, but no caching, no streaming and a mid-2024 cutoff."},{"slug":"openai-o3-mini","name":"o3-mini","provider":"OpenAI","api_id":"o3-mini","status":"deprecated","context_window":200000,"max_output_tokens":100000,"input_price":1.1,"output_price":4.4,"cached_input_price":0.55,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2025-01-31","best_for":"Text-only reasoning model shutting down 2026-10-23 (replacement gpt-5.6-sol); GPT-5.4 Mini is cheaper and adds vision."},{"slug":"openai-o4-mini","name":"o4-mini","provider":"OpenAI","api_id":"o4-mini","status":"deprecated","context_window":200000,"max_output_tokens":100000,"input_price":1.1,"output_price":4.4,"cached_input_price":0.275,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","fine-tuning","code-execution"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"proprietary","released":"2025-04-16","best_for":"The only reasoning model that was ever fine-tunable ($100/hour RFT training). Both it and its fine-tunes shut down 2026-10-23; replacement gpt-5.6-terra. Has a Flex tier at $0.55/$0.138/$2.20."},{"slug":"openai-gpt-4-turbo","name":"GPT-4 Turbo","provider":"OpenAI","api_id":"gpt-4-turbo","status":"deprecated","context_window":128000,"max_output_tokens":4096,"input_price":10,"output_price":30,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","batch","streaming"],"flagship":false,"knowledge_cutoff":"2023-12-01","license":"proprietary","released":"2024-04-09","best_for":"Shuts down 2026-10-23 (replacement gpt-5.6-sol). No caching, 4K max output — legacy compatibility only."},{"slug":"openai-gpt-4","name":"GPT-4","provider":"OpenAI","api_id":"gpt-4","status":"deprecated","context_window":8192,"max_output_tokens":8192,"input_price":30,"output_price":60,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","batch","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-12-01","license":"proprietary","released":"2023-06-13","best_for":"The original GPT-4 (gpt-4-0613), 8K context, shutting down 2026-10-23 along with its fine-tunes. Nothing to recommend it at $30/$60."},{"slug":"openai-gpt-3-5-turbo","name":"GPT-3.5 Turbo","provider":"OpenAI","api_id":"gpt-3.5-turbo","status":"deprecated","context_window":16385,"max_output_tokens":4096,"input_price":0.5,"output_price":1.5,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","batch","fine-tuning","tools"],"flagship":false,"knowledge_cutoff":"2021-09-01","license":"proprietary","released":"2024-01-25","best_for":"Default snapshot gpt-3.5-turbo-0125; shuts down 2026-10-23 (replacement gpt-5.6-terra). GPT-5.6 Luna is cheaper on input and far more capable — no reason to stay."},{"slug":"openai-gpt-3-5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","provider":"OpenAI","api_id":"gpt-3.5-turbo-instruct","status":"deprecated","context_window":4096,"max_output_tokens":4096,"input_price":1.5,"output_price":2,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"2021-09-01","license":"proprietary","released":"2023-09","best_for":"The last completions-endpoint (v1/completions) instruct model; shuts down 2026-09-28. Only for code that cannot be moved off raw completions."},{"slug":"openai-davinci-002","name":"davinci-002","provider":"OpenAI","api_id":"davinci-002","status":"deprecated","context_window":16384,"max_output_tokens":16384,"input_price":2,"output_price":2,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["fine-tuning","batch"],"flagship":false,"knowledge_cutoff":"2021-09-01","license":"proprietary","released":"2023-08","best_for":"Legacy base model for raw completions and classic fine-tuning; shuts down 2026-09-28. Fine-tuning it costs $6/1M training tokens and $12/$12 inference."},{"slug":"openai-babbage-002","name":"babbage-002","provider":"OpenAI","api_id":"babbage-002","status":"deprecated","context_window":16384,"max_output_tokens":16384,"input_price":0.4,"output_price":0.4,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["fine-tuning","batch"],"flagship":false,"knowledge_cutoff":"2021-09-01","license":"proprietary","released":"2023-08","best_for":"Smallest legacy base model, shutting down 2026-09-28; the cheapest fine-tune training on the platform at $0.40/1M tokens, but the clock has run out."},{"slug":"openai-gpt-oss-120b","name":"gpt-oss-120b","provider":"OpenAI","api_id":"gpt-oss-120b","status":"ga","context_window":131072,"max_output_tokens":131072,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","streaming","structured-output","batch"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"Apache-2.0","released":"2025-08-05","best_for":"Self-hosting a frontier-ish reasoning model on a single H100 with full chain-of-thought visibility and no per-token bill. OpenAI publishes no hosted price for it on the API pricing page, so cost depends on your own or a third party's infrastructure."},{"slug":"openai-gpt-oss-20b","name":"gpt-oss-20b","provider":"OpenAI","api_id":"gpt-oss-20b","status":"ga","context_window":131072,"max_output_tokens":131072,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","streaming","structured-output","batch"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"Apache-2.0","released":"2025-08-05","best_for":"On-device or low-latency local inference with tool calling; runs in ~16GB. No OpenAI-hosted price is published on the pricing page."},{"slug":"openai-gpt-realtime-2-1","name":"GPT-Realtime-2.1","provider":"OpenAI","api_id":"gpt-realtime-2.1","status":"ga","context_window":128000,"max_output_tokens":32000,"input_price":4,"output_price":24,"cached_input_price":0.4,"modalities_in":["text","audio","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2026-07-06","best_for":"Current best speech-to-speech model, with configurable reasoning effort and improved alphanumeric/noise handling. Text prices shown; AUDIO tokens are $32 in / $0.40 cached / $64 out and image input $5/$0.50 — audio dominates the bill."},{"slug":"openai-gpt-realtime-2-1-mini","name":"GPT-Realtime-2.1 Mini","provider":"OpenAI","api_id":"gpt-realtime-2.1-mini","status":"ga","context_window":128000,"max_output_tokens":32000,"input_price":0.6,"output_price":2.4,"cached_input_price":0.06,"modalities_in":["text","audio","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2026-07-06","best_for":"Cost-controlled voice agents: audio in/out is $10/$0.30 cached/$20 per 1M — roughly a third of full Realtime-2.1 — with the same 128K context. Text prices shown; image input $0.80/$0.08."},{"slug":"openai-gpt-realtime-2","name":"GPT-Realtime-2","provider":"OpenAI","api_id":"gpt-realtime-2","status":"ga","context_window":128000,"max_output_tokens":32000,"input_price":4,"output_price":24,"cached_input_price":0.4,"modalities_in":["text","audio","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2026-05-07","best_for":"Identical pricing to 2.1 (audio $32/$0.40/$64, image $5/$0.50) with none of the 2.1 quality fixes — pin it only if you have validated against this exact snapshot."},{"slug":"openai-gpt-realtime-1-5","name":"GPT-Realtime-1.5","provider":"OpenAI","api_id":"gpt-realtime-1.5","status":"ga","context_window":32000,"max_output_tokens":4096,"input_price":4,"output_price":16,"cached_input_price":0.4,"modalities_in":["text","audio","image"],"capabilities":["tools","prompt-caching","vision","streaming"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2026-02-23","best_for":"Non-reasoning voice model with cheaper text output ($16 vs $24) than Realtime-2.x but only 32K context; audio $32/$0.40/$64. Good for scripted, low-latency voice flows."},{"slug":"openai-gpt-realtime","name":"GPT-Realtime","provider":"OpenAI","api_id":"gpt-realtime","status":"deprecated","context_window":32000,"max_output_tokens":4096,"input_price":4,"output_price":16,"cached_input_price":0.4,"modalities_in":["text","audio","image"],"capabilities":["tools","vision","prompt-caching"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2025-08-28","best_for":"Shuts down 2027-01-20; replacement gpt-realtime-2.1. Audio $32/$0.40/$64."},{"slug":"openai-gpt-realtime-mini","name":"GPT-Realtime Mini","provider":"OpenAI","api_id":"gpt-realtime-mini","status":"deprecated","context_window":32000,"max_output_tokens":4096,"input_price":0.6,"output_price":2.4,"cached_input_price":0.06,"modalities_in":["text","image","audio"],"capabilities":["tools","vision","prompt-caching"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2025-10-06","best_for":"Shuts down 2027-01-20 (the 2025-10-06 snapshot is already gone); replacement gpt-realtime-2.1-mini at the same audio rates of $10/$0.30/$20."},{"slug":"openai-gpt-4o-realtime-preview","name":"GPT-4o Realtime Preview","provider":"OpenAI","api_id":"gpt-4o-realtime-preview","status":"deprecated","context_window":32000,"max_output_tokens":4096,"input_price":5,"output_price":20,"cached_input_price":2.5,"modalities_in":["text","audio"],"capabilities":["tools","prompt-caching"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2024-10-01","best_for":"Shuts down 2027-01-20. Its audio rate of $40/$2.50/$80 is the most expensive realtime audio still served — move to gpt-realtime-2.1 for 20% less."},{"slug":"openai-gpt-4o-mini-realtime-preview","name":"GPT-4o Mini Realtime Preview","provider":"OpenAI","api_id":"gpt-4o-mini-realtime-preview","status":"deprecated","context_window":16000,"max_output_tokens":4096,"input_price":0.6,"output_price":2.4,"cached_input_price":0.3,"modalities_in":["text","audio"],"capabilities":["tools","prompt-caching"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2024-12-17","best_for":"Shuts down 2027-01-20; audio $10/$0.30/$20. Only a 16K context — replace with gpt-realtime-2.1-mini, which gives 128K at the same audio price."},{"slug":"openai-gpt-audio-1-5","name":"GPT-Audio-1.5","provider":"OpenAI","api_id":"gpt-audio-1.5","status":"ga","context_window":128000,"max_output_tokens":16384,"input_price":2.5,"output_price":10,"cached_input_price":-1,"modalities_in":["text","audio"],"capabilities":["tools","streaming"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2026-02-23","best_for":"Best current audio-in/audio-out model for turn-based Chat Completions (not streaming voice sessions). Text prices shown; audio tokens $32 in / $64 out, with no cached-audio discount."},{"slug":"openai-gpt-audio","name":"GPT-Audio","provider":"OpenAI","api_id":"gpt-audio","status":"deprecated","context_window":128000,"max_output_tokens":16384,"input_price":2.5,"output_price":10,"cached_input_price":-1,"modalities_in":["text","audio"],"capabilities":["tools","streaming"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2025-08-28","best_for":"Shuts down 2027-01-20; audio $32/$64. Identical price to gpt-audio-1.5, so migrate — there is no cost reason to stay."},{"slug":"openai-gpt-audio-mini","name":"GPT-Audio Mini","provider":"OpenAI","api_id":"gpt-audio-mini","status":"deprecated","context_window":128000,"max_output_tokens":16384,"input_price":0.6,"output_price":2.4,"cached_input_price":-1,"modalities_in":["text","audio"],"capabilities":["tools","prompt-caching"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2025-10-06","best_for":"Shuts down 2027-01-20 (replacement gpt-audio-1.5, which is 4x the text price). Audio tokens $10 in / $20 out — the cheapest audio-out model OpenAI serves."},{"slug":"openai-gpt-4o-audio-preview","name":"GPT-4o Audio Preview","provider":"OpenAI","api_id":"gpt-4o-audio-preview","status":"deprecated","context_window":128000,"max_output_tokens":16384,"input_price":2.5,"output_price":10,"cached_input_price":-1,"modalities_in":["text","audio"],"capabilities":["tools","streaming"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2024-10-17","best_for":"Shuts down 2027-01-20; audio $40/$80, the priciest audio tokens on the platform. Replacement is gpt-audio-1.5 at $32/$64."},{"slug":"openai-gpt-4o-mini-audio-preview","name":"GPT-4o Mini Audio Preview","provider":"OpenAI","api_id":"gpt-4o-mini-audio-preview","status":"deprecated","context_window":128000,"max_output_tokens":16384,"input_price":0.15,"output_price":0.6,"cached_input_price":-1,"modalities_in":["text","audio"],"capabilities":["tools","streaming"],"flagship":false,"knowledge_cutoff":"2023-10-01","license":"proprietary","released":"2024-12-17","best_for":"Shuts down 2027-01-20; cheapest text side of any audio model ($0.15/$0.60) with audio at $10/$20. Replacement gpt-audio-1.5."},{"slug":"openai-gpt-realtime-translate","name":"GPT-Realtime-Translate","provider":"OpenAI","api_id":"gpt-realtime-translate","status":"ga","context_window":16000,"max_output_tokens":2000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2026-05-07","best_for":"Streaming speech-to-speech translation. Not token-priced at all: OpenAI bills $0.034 per minute of audio via v1/realtime/translations."},{"slug":"openai-gpt-realtime-whisper","name":"GPT-Realtime-Whisper","provider":"OpenAI","api_id":"gpt-realtime-whisper","status":"ga","context_window":16000,"max_output_tokens":2000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio","text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"2024-09-30","license":"proprietary","released":"2026-05-07","best_for":"Realtime streaming transcription priced by duration at $0.017/minute — same rate as gpt-live-transcribe; pick whichever accuracy profile suits your audio."},{"slug":"openai-gpt-live-transcribe","name":"GPT-Live-Transcribe","provider":"OpenAI","api_id":"gpt-live-transcribe","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio","text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-07-28","best_for":"Low-latency live captions and realtime agent transcription; $0.017 per minute, the named replacement for whisper-1 and the gpt-4o-transcribe family."},{"slug":"openai-gpt-transcribe","name":"GPT-Transcribe","provider":"OpenAI","api_id":"gpt-transcribe","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio","text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-07-28","best_for":"Cheapest and most accurate file transcription OpenAI sells: $0.0045 per minute, a quarter of Whisper's $0.006 and a third of live transcription. Use it for batch/archive audio."},{"slug":"openai-gpt-4o-transcribe","name":"GPT-4o Transcribe","provider":"OpenAI","api_id":"gpt-4o-transcribe","status":"deprecated","context_window":16000,"max_output_tokens":2000,"input_price":2.5,"output_price":10,"cached_input_price":-1,"modalities_in":["audio","text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"proprietary","released":"2025-03-20","best_for":"Shuts down 2027-02-26. Token-billed (audio in $2.50/1M, text out $10/1M) which OpenAI estimates at $0.006/minute — gpt-transcribe does the same job for $0.0045."},{"slug":"openai-gpt-4o-mini-transcribe","name":"GPT-4o Mini Transcribe","provider":"OpenAI","api_id":"gpt-4o-mini-transcribe","status":"deprecated","context_window":16000,"max_output_tokens":2000,"input_price":1.25,"output_price":5,"cached_input_price":-1,"modalities_in":["audio","text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"proprietary","released":"2025-03-20","best_for":"Shuts down 2027-02-26; ~$0.003/minute. Default snapshot is now gpt-4o-mini-transcribe-2025-12-15 (the 2025-03-20 snapshot was already retired)."},{"slug":"openai-gpt-4o-transcribe-diarize","name":"GPT-4o Transcribe Diarize","provider":"OpenAI","api_id":"gpt-4o-transcribe-diarize","status":"deprecated","context_window":16000,"max_output_tokens":2000,"input_price":2.5,"output_price":10,"cached_input_price":-1,"modalities_in":["text","audio"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"2024-06-01","license":"proprietary","released":"","best_for":"The only OpenAI model that labels who is speaking when, at ~$0.006/minute — but it shuts down 2027-02-26 and the named replacements do not advertise diarization, so plan a third-party fallback."},{"slug":"openai-whisper","name":"Whisper","provider":"OpenAI","api_id":"whisper-1","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary (hosted); open-weights Whisper released under MIT separately","released":"2023-03","best_for":"Shuts down 2027-02-26. Billed at $0.006 per minute; gpt-transcribe is both cheaper and more accurate. Note the hosted whisper-1 endpoint is distinct from the MIT-licensed open Whisper weights on GitHub."},{"slug":"openai-gpt-4o-mini-tts","name":"GPT-4o Mini TTS","provider":"OpenAI","api_id":"gpt-4o-mini-tts","status":"ga","context_window":0,"max_output_tokens":0,"input_price":0.6,"output_price":12,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-03-20","best_for":"Steerable text-to-speech where you can instruct tone and delivery; billed per token ($0.60/1M text in, $12/1M audio out) rather than per character like tts-1. Default snapshot gpt-4o-mini-tts-2025-12-15."},{"slug":"openai-tts-1","name":"TTS-1","provider":"OpenAI","api_id":"tts-1","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2023-11-06","best_for":"Lowest-latency speech generation, billed per character not per token: $15.00 per 1M input characters. Use it when you need predictable per-character cost; gpt-4o-mini-tts if you need voice steering."},{"slug":"openai-tts-1-hd","name":"TTS-1 HD","provider":"OpenAI","api_id":"tts-1-hd","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2023-11-06","best_for":"Quality-optimised TTS at $30.00 per 1M input characters — double tts-1. Worth it only for published/produced audio, not interactive apps."},{"slug":"openai-gpt-image-2","name":"GPT-Image-2","provider":"OpenAI","api_id":"gpt-image-2","status":"ga","context_window":0,"max_output_tokens":0,"input_price":5,"output_price":-1,"cached_input_price":1.25,"modalities_in":["text","image"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-04-21","best_for":"State-of-the-art generation and editing (inpainting, transparent backgrounds in preview) and the migration target for every older image model. Text-input prices shown; IMAGE tokens are $8 in / $2 cached / $30 out per 1M — cheaper output than gpt-image-1.5's $32."},{"slug":"openai-gpt-image-1-5","name":"GPT-Image-1.5","provider":"OpenAI","api_id":"gpt-image-1.5","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":5,"output_price":10,"cached_input_price":1.25,"modalities_in":["text","image"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-12-16","best_for":"Shuts down 2026-12-01; replace with gpt-image-2. Image tokens $8/$2/$32; per-image list prices run $0.009 (low, 1024x1024) to $0.20 (high, 1536x1024)."},{"slug":"openai-gpt-image-1-mini","name":"GPT-Image-1 Mini","provider":"OpenAI","api_id":"gpt-image-1-mini","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":2,"output_price":-1,"cached_input_price":0.2,"modalities_in":["text","image"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-10-06","best_for":"Shuts down 2026-12-01. Cheapest image generation OpenAI has offered — $0.005 for a low-quality 1024x1024, image tokens $2.50/$0.25/$8 — with no equally cheap successor."},{"slug":"openai-gpt-image-1","name":"GPT-Image-1","provider":"OpenAI","api_id":"gpt-image-1","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":5,"output_price":-1,"cached_input_price":1.25,"modalities_in":["text","image"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-04","best_for":"Shuts down 2026-10-23, earlier than the other image models. Most expensive of the family (image tokens $10/$2.50/$40, up to $0.25 per high-quality image) — migrate to gpt-image-2."},{"slug":"openai-chatgpt-image-latest","name":"ChatGPT Image Latest","provider":"OpenAI","api_id":"chatgpt-image-latest","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":5,"output_price":10,"cached_input_price":1.25,"modalities_in":["text","image"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-12-16","best_for":"Matches the image model used inside ChatGPT; shuts down 2026-12-01. Priced identically to gpt-image-1.5 (image tokens $8/$2/$32)."},{"slug":"openai-sora-2","name":"Sora 2","provider":"OpenAI","api_id":"sora-2","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-10-06","best_for":"Video with synced audio, billed per second of output ($0.10/s at 720p, $0.05/s via Batch), not per token. The whole Videos API and both Sora models shut down 2026-09-24 with no replacement — do not start new work here."},{"slug":"openai-sora-2-pro","name":"Sora 2 Pro","provider":"OpenAI","api_id":"sora-2-pro","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-10-06","best_for":"Higher-fidelity video up to 1080p, priced per second: $0.30 (720p), $0.50 (1024p), $0.70 (1080p); half those rates via Batch. Also shuts down 2026-09-24 with no successor."},{"slug":"openai-text-embedding-3-small","name":"text-embedding-3-small","provider":"OpenAI","api_id":"text-embedding-3-small","status":"ga","context_window":0,"max_output_tokens":0,"input_price":0.02,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-01-25","best_for":"Default embedding model for RAG at scale — 6.5x cheaper than 3-large and supports the `dimensions` parameter to shrink vectors. Embeddings bill on input tokens only; there is no output charge."},{"slug":"openai-text-embedding-3-large","name":"text-embedding-3-large","provider":"OpenAI","api_id":"text-embedding-3-large","status":"ga","context_window":0,"max_output_tokens":0,"input_price":0.13,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-01-25","best_for":"Highest-quality OpenAI embedding; use it when retrieval accuracy on a large heterogeneous corpus matters more than the 6.5x cost over 3-small."},{"slug":"openai-text-embedding-ada-002","name":"text-embedding-ada-002","provider":"OpenAI","api_id":"text-embedding-ada-002","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":0.1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2022-12","best_for":"Legacy embedding kept alive only so existing vector stores keep matching; it costs 5x text-embedding-3-small for worse quality. Re-embed when you can."},{"slug":"openai-omni-moderation","name":"omni-moderation","provider":"OpenAI","api_id":"omni-moderation-latest","status":"ga","context_window":0,"max_output_tokens":0,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-09-26","best_for":"Free text+image safety classification via v1/moderations — run it on user input and model output; the pricing page lists it as Free with no token charge. Default snapshot omni-moderation-2024-09-26."},{"slug":"google-deepmind-gemini-3-8-flash","name":"Gemini 3.8 Flash","provider":"Google DeepMind","api_id":"gemini-3.8-flash","status":"ga","context_window":1048576,"max_output_tokens":65536,"input_price":0.75,"output_price":3.75,"cached_input_price":0.075,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":true,"knowledge_cutoff":"March 2026","license":"proprietary","released":"2026-09","best_for":"The current default for long-horizon agentic coding and enterprise workflows — frontier-class reasoning at Flash cost, but note $0.75/$3.75 is introductory and doubles to $1.50/$7.50 on 2027-01-01; thinking levels are low/medium/high only ('minimal' errors)."},{"slug":"google-deepmind-gemini-3-7-flash","name":"Gemini 3.7 Flash","provider":"Google DeepMind","api_id":"gemini-3.7-flash","status":"ga","context_window":1048576,"max_output_tokens":65536,"input_price":0.75,"output_price":3.75,"cached_input_price":0.075,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"March 2026","license":"proprietary","released":"2026-08","best_for":"Everyday coding and reliable multi-step tool use; identical price and limits to 3.8 Flash, so pick it only if you have pinned regression-tested behaviour — otherwise take 3.8."},{"slug":"google-deepmind-gemini-3-6-flash","name":"Gemini 3.6 Flash","provider":"Google DeepMind","api_id":"gemini-3.6-flash","status":"ga","context_window":1048576,"max_output_tokens":65536,"input_price":0.75,"output_price":3.75,"cached_input_price":0.075,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"March 2026","license":"proprietary","released":"2026-07","best_for":"Code generation, agentic execution and spatial reasoning; same price as 3.7/3.8 Flash and superseded by both — keep only for pinned deployments."},{"slug":"google-deepmind-gemini-3-5-flash","name":"Gemini 3.5 Flash","provider":"Google DeepMind","api_id":"gemini-3.5-flash","status":"ga","context_window":1048576,"max_output_tokens":65536,"input_price":1.5,"output_price":9,"cached_input_price":0.15,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-05","best_for":"Sub-agent deployment and long-horizon agentic loops — but it is now the most expensive Flash on the list ($1.50/$9.00 vs $0.75/$3.75 for 3.6-3.8 through 2026); migrate off it unless you need its exact behaviour."},{"slug":"google-deepmind-gemini-3-5-flash-lite","name":"Gemini 3.5 Flash-Lite","provider":"Google DeepMind","api_id":"gemini-3.5-flash-lite","status":"ga","context_window":1048576,"max_output_tokens":65536,"input_price":0.3,"output_price":2.5,"cached_input_price":0.03,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"March 2026","license":"proprietary","released":"2026-07","best_for":"High-throughput sub-agent fan-out and document parsing where you still want a 1M window and preview computer-use; costs 20% more input than 3.1 Flash-Lite for a fresher March 2026 cutoff."},{"slug":"google-deepmind-gemini-3-1-flash-lite","name":"Gemini 3.1 Flash-Lite","provider":"Google DeepMind","api_id":"gemini-3.1-flash-lite","status":"ga","context_window":1048576,"max_output_tokens":65536,"input_price":0.25,"output_price":1.5,"cached_input_price":0.025,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-05","best_for":"Cheapest Gemini 3 model — high-volume translation, classification and extraction at scale; audio input costs double ($0.50/1M) so keep it to text/image/video workloads."},{"slug":"google-deepmind-gemini-3-1-pro-preview","name":"Gemini 3.1 Pro Preview","provider":"Google DeepMind","api_id":"gemini-3.1-pro-preview","status":"preview","context_window":1048576,"max_output_tokens":65536,"input_price":2,"output_price":12,"cached_input_price":0.2,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-02","best_for":"The only Pro-tier Gemini 3 currently served — hardest software-engineering and precise multi-step tool tasks; prices double above 200k tokens ($4.00/$18.00) and there is no free tier, plus a gemini-3.1-pro-preview-customtools endpoint tuned for bash + custom tools."},{"slug":"google-deepmind-gemini-3-flash-preview","name":"Gemini 3 Flash Preview","provider":"Google DeepMind","api_id":"gemini-3-flash-preview","status":"preview","context_window":1048576,"max_output_tokens":65536,"input_price":0.5,"output_price":3,"cached_input_price":0.05,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-12","best_for":"Google's own docs call it the legacy baseline Flash — cheaper than 3.5 Flash but beaten on price and capability by 3.6-3.8 Flash during the intro period; audio input is $1.00/1M."},{"slug":"google-deepmind-gemini-2-5-pro","name":"Gemini 2.5 Pro","provider":"Google DeepMind","api_id":"gemini-2.5-pro","status":"ga","context_window":1048576,"max_output_tokens":65536,"input_price":1.25,"output_price":10,"cached_input_price":0.125,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"January 2025","license":"proprietary","released":"2025-06","best_for":"The last GA (non-preview) Pro model — choose it over 3.1 Pro Preview when you need production stability guarantees or a free tier; prices step to $2.50/$15.00 above 200k tokens."},{"slug":"google-deepmind-gemini-2-5-flash","name":"Gemini 2.5 Flash","provider":"Google DeepMind","api_id":"gemini-2.5-flash","status":"ga","context_window":1048576,"max_output_tokens":65536,"input_price":0.3,"output_price":2.5,"cached_input_price":0.03,"modalities_in":["text","image","audio","video"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"January 2025","license":"proprietary","released":"2025-06","best_for":"Stable, well-understood workhorse for existing production traffic; same headline price as 3.5 Flash-Lite but weaker — only stay on it if migration cost outweighs the gain."},{"slug":"google-deepmind-gemini-2-5-flash-lite","name":"Gemini 2.5 Flash-Lite","provider":"Google DeepMind","api_id":"gemini-2.5-flash-lite","status":"ga","context_window":1048576,"max_output_tokens":65536,"input_price":0.1,"output_price":0.4,"cached_input_price":0.01,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"January 2025","license":"proprietary","released":"2025-07","best_for":"Absolute cheapest token price in the whole Gemini lineup ($0.10/$0.40) — the right pick for massive-volume classification, tagging and simple extraction where Gemini 3 quality is unnecessary."},{"slug":"google-deepmind-gemini-2-5-computer-use-preview","name":"Gemini 2.5 Computer Use Preview","provider":"Google DeepMind","api_id":"gemini-2.5-computer-use-preview-10-2025","status":"preview","context_window":128000,"max_output_tokens":64000,"input_price":1.25,"output_price":10,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","computer-use","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-10","best_for":"Legacy browser-automation and UI-testing endpoint only — Gemini 3 Pro and Flash now do computer use natively at Flash prices, so use this only for pinned existing agents; $2.50/$15.00 above 200k tokens, no free tier."},{"slug":"google-deepmind-gemini-3-1-flash-image-nano-banana-2","name":"Gemini 3.1 Flash Image (Nano Banana 2)","provider":"Google DeepMind","api_id":"gemini-3.1-flash-image","status":"ga","context_window":131072,"max_output_tokens":32768,"input_price":0.5,"output_price":60,"cached_input_price":-1,"modalities_in":["text","image","video","pdf"],"capabilities":["vision","reasoning","batch","streaming","web-search"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-02","best_for":"Default high-throughput image generation and conversational editing: $60/1M image tokens works out to $0.045 (0.5K), $0.067 (1K), $0.101 (2K), $0.151 (4K) per image; text/thinking output is billed separately at $3.00/1M. No free tier."},{"slug":"google-deepmind-gemini-3-1-flash-lite-image-nano-banana-2-lite","name":"Gemini 3.1 Flash Lite Image (Nano Banana 2 Lite)","provider":"Google DeepMind","api_id":"gemini-3.1-flash-lite-image","status":"ga","context_window":65536,"max_output_tokens":4096,"input_price":0.25,"output_price":30,"cached_input_price":-1,"modalities_in":["text","image","video","pdf"],"capabilities":["vision","reasoning","batch","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026","best_for":"Cheapest Gemini image endpoint at $0.0336 per 1K image (1024px only — no 2K/4K); pick it for ultra-low-latency, high-volume generation where you don't need search grounding or large output."},{"slug":"google-deepmind-gemini-3-pro-image-nano-banana-pro","name":"Gemini 3 Pro Image (Nano Banana Pro)","provider":"Google DeepMind","api_id":"gemini-3-pro-image","status":"ga","context_window":65536,"max_output_tokens":32768,"input_price":2,"output_price":120,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","reasoning","batch","streaming","web-search"],"flagship":true,"knowledge_cutoff":"","license":"proprietary","released":"2025-11","best_for":"Studio-grade image work needing accurate text rendering and Search-grounded factual visuals: $0.134 per 1K/2K image, $0.24 per 4K; input images are a flat 560 tokens ($0.0011 each) and text/thinking output is $12.00/1M."},{"slug":"google-deepmind-gemini-2-5-flash-image-nano-banana","name":"Gemini 2.5 Flash Image (Nano Banana)","provider":"Google DeepMind","api_id":"gemini-2.5-flash-image","status":"ga","context_window":65536,"max_output_tokens":32768,"input_price":0.3,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","batch","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-08","best_for":"Original Nano Banana, still served: output is priced per image ($0.039/image) rather than per token, so there is no per-1M output rate — Nano Banana 2 Lite is cheaper per 1K image ($0.0336) and generally better."},{"slug":"google-deepmind-gemini-3-1-flash-live-preview","name":"Gemini 3.1 Flash Live Preview","provider":"Google DeepMind","api_id":"gemini-3.1-flash-live-preview","status":"preview","context_window":131072,"max_output_tokens":65536,"input_price":0.75,"output_price":4.5,"cached_input_price":-1,"modalities_in":["text","image","audio","video"],"capabilities":["tools","vision","reasoning","streaming","web-search"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-03","best_for":"Current best real-time audio-to-audio voice agent over the Live API (WebSockets); the recorded $0.75/$4.50 is the TEXT rate — audio is $3.00/1M in and $12.00/1M out (or $0.005/$0.018 per minute), image/video input $1.00/1M."},{"slug":"google-deepmind-gemini-3-1-flash-tts-preview","name":"Gemini 3.1 Flash TTS Preview","provider":"Google DeepMind","api_id":"gemini-3.1-flash-tts-preview","status":"preview","context_window":8192,"max_output_tokens":16384,"input_price":1,"output_price":20,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["batch","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-04","best_for":"Latest controllable speech synthesis with expressive audio tags; audio bills at 25 tokens per second, so $20/1M output is roughly $0.0005 per second — 2x the price of 2.5 Flash TTS for better naturalness and multilinguality."},{"slug":"google-deepmind-gemini-2-5-flash-preview-tts","name":"Gemini 2.5 Flash Preview TTS","provider":"Google DeepMind","api_id":"gemini-2.5-flash-preview-tts","status":"preview","context_window":8192,"max_output_tokens":16384,"input_price":0.5,"output_price":10,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["batch","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-12","best_for":"Cheapest Gemini TTS at half the price of 3.1 Flash TTS — right for high-volume narration and real-time assistants where prosody control matters more than the newest expressive tags."},{"slug":"google-deepmind-gemini-2-5-pro-preview-tts","name":"Gemini 2.5 Pro Preview TTS","provider":"Google DeepMind","api_id":"gemini-2.5-pro-preview-tts","status":"preview","context_window":8192,"max_output_tokens":16384,"input_price":1,"output_price":20,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["batch","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-12","best_for":"Studio-quality long-form narration where vocal clarity justifies Pro pricing; no free tier, and batch enqueue is capped at only 25k-100k tokens."},{"slug":"google-deepmind-gemini-2-5-flash-native-audio-live-api","name":"Gemini 2.5 Flash Native Audio (Live API)","provider":"Google DeepMind","api_id":"gemini-2.5-flash-native-audio-preview-12-2025","status":"preview","context_window":131072,"max_output_tokens":8192,"input_price":0.5,"output_price":2,"cached_input_price":-1,"modalities_in":["text","image","audio","video"],"capabilities":["tools","vision","reasoning","streaming","web-search"],"flagship":false,"knowledge_cutoff":"January 2025","license":"proprietary","released":"2025-09","best_for":"Cheaper legacy Live API voice endpoint: recorded $0.50/$2.00 is the TEXT rate; audio/video input is $3.00/1M and audio output $12.00/1M — same audio rates as 3.1 Flash Live, so upgrade unless pinned."},{"slug":"google-deepmind-gemini-3-5-transcribe","name":"Gemini 3.5 Transcribe","provider":"Google DeepMind","api_id":"gemini-3.5-transcribe","status":"ga","context_window":0,"max_output_tokens":0,"input_price":2,"output_price":12,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":["streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-08","best_for":"Batch speech-to-text for stored audio files: 85+ languages with code-switching, speaker diarization, word-level timestamps and custom vocabulary; ~$0.003/min in + $0.002/min out, up to 1 hour per request (30 min with diarization or word timestamps)."},{"slug":"google-deepmind-gemini-3-5-transcribe-live","name":"Gemini 3.5 Transcribe Live","provider":"Google DeepMind","api_id":"gemini-3.5-transcribe-live","status":"ga","context_window":0,"max_output_tokens":0,"input_price":3.5,"output_price":21,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-08","best_for":"Real-time streaming transcription over WebSockets at ~$0.005/min in + $0.004/min out — 75% dearer than the file endpoint, so only use it when you genuinely need sub-second partial results."},{"slug":"google-deepmind-gemini-3-5-live-translate-preview","name":"Gemini 3.5 Live Translate Preview","provider":"Google DeepMind","api_id":"gemini-3.5-live-translate-preview","status":"preview","context_window":131072,"max_output_tokens":65536,"input_price":3.5,"output_price":21,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06","best_for":"Purpose-built bidirectional speech-to-speech interpretation (~$0.0053/min in, $0.0315/min out); no tools, no function calling, no thinking — a dedicated translation pipe, not a general voice agent."},{"slug":"google-deepmind-gemini-omni-flash","name":"Gemini Omni Flash","provider":"Google DeepMind","api_id":"gemini-omni-1.1-flash","status":"ga","context_window":1048576,"max_output_tokens":0,"input_price":1.5,"output_price":17.5,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["vision","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-08","best_for":"Conversational video generation and editing through the Interactions API (3-10s clips at 360p-4K, 24fps, plus extension and upscaling); output is $17.50/1M for video and $9.00/1M for text, no free tier."},{"slug":"google-deepmind-gemini-omni-flash-preview","name":"Gemini Omni Flash Preview","provider":"Google DeepMind","api_id":"gemini-omni-flash-preview","status":"preview","context_window":1048576,"max_output_tokens":0,"input_price":1.5,"output_price":17.5,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["vision","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026","best_for":"Preview channel for Omni Flash at identical pricing — use the stable gemini-omni-1.1-flash unless you need pre-release video features."},{"slug":"google-deepmind-veo-3-1","name":"Veo 3.1","provider":"Google DeepMind","api_id":"veo-3.1-generate-preview","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-10","best_for":"Highest-fidelity text/image-to-video with native audio; billed per second of output, not per token — $0.40/s at 720p-1080p and $0.60/s at 4K, so a 8s 1080p clip is ~$3.20. No free tier; you are only charged for successfully generated video."},{"slug":"google-deepmind-veo-3-1-fast","name":"Veo 3.1 Fast","provider":"Google DeepMind","api_id":"veo-3.1-fast-generate-preview","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-10","best_for":"4x cheaper than Veo 3.1 Standard at $0.10/s (720p), $0.12/s (1080p), $0.30/s (4K) — the sensible default for iteration and most production video; per-second pricing, not per-token."},{"slug":"google-deepmind-veo-3-1-lite","name":"Veo 3.1 Lite","provider":"Google DeepMind","api_id":"veo-3.1-lite-generate-preview","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-10","best_for":"Cheapest video at $0.05/s (720p) and $0.08/s (1080p) with no 4K support — for drafts, previews and high-volume short-form; per-second pricing, not per-token."},{"slug":"google-deepmind-lyria-3-5","name":"Lyria 3.5","provider":"Google DeepMind","api_id":"lyria-3.5","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026","best_for":"Current music generation model, billed per request not per token: $0.08 per full song. No free tier."},{"slug":"google-deepmind-lyria-3-clip-preview","name":"Lyria 3 Clip Preview","provider":"Google DeepMind","api_id":"lyria-3-clip-preview","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025","best_for":"Legacy 30-second music clips at $0.04 per song — the only sub-$0.08 music option, useful for stings and loops."},{"slug":"google-deepmind-lyria-3-pro-preview","name":"Lyria 3 Pro Preview","provider":"Google DeepMind","api_id":"lyria-3-pro-preview","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025","best_for":"Legacy full-song generation at $0.08 per song — same price as Lyria 3.5, which supersedes it."},{"slug":"google-deepmind-lyria-realtime","name":"Lyria RealTime","provider":"Google DeepMind","api_id":"lyria-realtime-exp","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025","best_for":"Experimental streaming music generation for interactive/live scoring; no price is published on the official pricing page — treat as experimental and confirm billing before production use."},{"slug":"google-deepmind-gemini-embedding-2","name":"Gemini Embedding 2","provider":"Google DeepMind","api_id":"gemini-embedding-2","status":"ga","context_window":8192,"max_output_tokens":0,"input_price":0.2,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-04","best_for":"Google's first multimodal embedding model — one unified space for text, image, audio, video and PDF, with Matryoshka dimensions 128-3072 (768/1536/3072 recommended). Text $0.20/1M, image $0.45/1M ($0.00012/image), audio $6.50/1M, video $12.00/1M ($0.00079/frame); embeddings have no output charge."},{"slug":"google-deepmind-gemini-embedding-001","name":"Gemini Embedding 001","provider":"Google DeepMind","api_id":"gemini-embedding-001","status":"ga","context_window":2048,"max_output_tokens":0,"input_price":0.15,"output_price":0,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-06","best_for":"Text-only embeddings at $0.15/1M with 128-3072 flexible dimensions; cheaper than Embedding 2 for pure text but capped at a 2,048-token input window (vs 8,192) — this is also the rate charged for File Search indexing."},{"slug":"google-deepmind-gemini-robotics-er-2-preview","name":"Gemini Robotics ER 2 Preview","provider":"Google DeepMind","api_id":"gemini-robotics-er-2-preview","status":"preview","context_window":131072,"max_output_tokens":65536,"input_price":2,"output_price":10,"cached_input_price":0.2,"modalities_in":["text","image","audio","video"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026","best_for":"Embodied-reasoning VLM for robot orchestration, multi-robot collaboration, video progress understanding and advanced spatial reasoning; Pro-tier pricing on a 128k window, and it has a free tier unlike Gemini 3.1 Pro."},{"slug":"google-deepmind-gemini-robotics-er-2-streaming-preview","name":"Gemini Robotics ER 2 Streaming Preview","provider":"Google DeepMind","api_id":"gemini-robotics-er-2-streaming-preview","status":"preview","context_window":131072,"max_output_tokens":65536,"input_price":2,"output_price":10,"cached_input_price":-1,"modalities_in":["text","image","audio","video"],"capabilities":["tools","vision","reasoning","streaming","structured-output","web-search"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026","best_for":"Continuous-stream variant of Robotics ER 2 for closed-loop control from live camera feeds; same token price as the unary endpoint but no batch or context caching."},{"slug":"google-deepmind-gemini-robotics-er-1-6-preview","name":"Gemini Robotics ER 1.6 Preview","provider":"Google DeepMind","api_id":"gemini-robotics-er-1.6-preview","status":"preview","context_window":131072,"max_output_tokens":65536,"input_price":1,"output_price":5,"cached_input_price":-1,"modalities_in":["text","image","audio","video"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution","computer-use"],"flagship":false,"knowledge_cutoff":"January 2025","license":"proprietary","released":"2025-12","best_for":"Half the price of Robotics ER 2 ($1.00/$5.00) for spatial reasoning and natural-language action planning; audio input is $2.00/1M. Use it when ER 2's extra capability isn't needed."},{"slug":"google-deepmind-gemma-4-31b-instruct","name":"Gemma 4 31B Instruct","provider":"Google DeepMind","api_id":"gemma-4-31b-it","status":"ga","context_window":262144,"max_output_tokens":0,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Gemma 4 license (Gemma Terms of Use)","released":"2026","best_for":"Largest dense open-weight Gemma for self-hosting on server-grade GPUs with a 256K window; hosted access via the Gemini API is free-tier only (the pricing page lists paid-tier rates as 'Not available'), so plan to run it yourself for production."},{"slug":"google-deepmind-gemma-4-26b-a4b-instruct","name":"Gemma 4 26B A4B Instruct","provider":"Google DeepMind","api_id":"gemma-4-26b-a4b-it","status":"ga","context_window":262144,"max_output_tokens":0,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Gemma 4 license (Gemma Terms of Use)","released":"2026","best_for":"Best open-weight throughput-per-dollar in the family — MoE activating only 4B params per token, so it serves near-4B-dense speed at 26B-class quality; free-tier only on the hosted Gemini API, weights on Kaggle/Hugging Face with QAT quantized builds."},{"slug":"google-deepmind-gemma-4-12b","name":"Gemma 4 12B","provider":"Google DeepMind","api_id":"","status":"ga","context_window":262144,"max_output_tokens":0,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Gemma 4 license (Gemma Terms of Use)","released":"2026","best_for":"Encoder-free multimodal mid-size open model with a 256K window — the sweet spot for single-GPU self-hosting; download-only (not exposed on the hosted Gemini API)."},{"slug":"google-deepmind-gemma-4-e4b","name":"Gemma 4 E4B","provider":"Google DeepMind","api_id":"","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Gemma 4 license (Gemma Terms of Use)","released":"2026","best_for":"On-device/edge deployment with a 128K window; download-only, with QAT quantized builds for phone and laptop inference engines."},{"slug":"google-deepmind-gemma-4-e2b","name":"Gemma 4 E2B","provider":"Google DeepMind","api_id":"","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Gemma 4 license (Gemma Terms of Use)","released":"2026","best_for":"Smallest Gemma 4 — ultra-mobile and embedded inference at 128K context; download-only, no hosted endpoint."},{"slug":"google-deepmind-gemini-deep-research","name":"Gemini Deep Research","provider":"Google DeepMind","api_id":"deep-research-preview-04-2026","status":"preview","context_window":1048576,"max_output_tokens":65536,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","web-search","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-04","best_for":"Autonomous multi-step research producing cited reports, with collaborative planning, MCP servers and File Search over the Interactions API; it has no list price of its own — all inference (including intermediate reasoning tokens) bills at the underlying Gemini model's standard rates plus tool fees."},{"slug":"google-deepmind-gemini-deep-research-max","name":"Gemini Deep Research Max","provider":"Google DeepMind","api_id":"deep-research-max-preview-04-2026","status":"preview","context_window":1048576,"max_output_tokens":65536,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","audio","video","pdf"],"capabilities":["tools","vision","reasoning","web-search","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-04","best_for":"Maximum-comprehensiveness tier of Deep Research — more agentic loops and therefore materially more billed tokens than the standard tier; use when report completeness beats cost and latency. Billed at underlying model rates, no separate list price."},{"slug":"google-deepmind-antigravity-agent","name":"Antigravity Agent","provider":"Google DeepMind","api_id":"antigravity-preview-05-2026","status":"preview","context_window":1048576,"max_output_tokens":65536,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","code-execution","web-search","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-05","best_for":"General-purpose managed agent that plans, runs code, manages files and browses inside a Google-hosted Linux sandbox; context compacts at ~135k despite the 1M ceiling. Sandbox compute is free during preview and only model tokens are billed at standard rates."},{"slug":"meta-muse-spark-1-3","name":"Muse Spark 1.3","provider":"Meta","api_id":"muse-spark-1.3","status":"ga","context_window":1048576,"max_output_tokens":0,"input_price":1.25,"output_price":4.25,"cached_input_price":0.15,"modalities_in":["text","image","video","audio","pdf"],"capabilities":["chat","reasoning","tool-calling","tool-search","parallel-tool-calls","structured-output","prompt-caching","search-grounding","image-understanding","video-understanding","file-handling","computer-use","token-counting","streaming","background-responses"],"flagship":true,"knowledge_cutoff":"","license":"","released":"","best_for":"Meta's default first-party model and the only one to pick for new agentic or coding work — it is the sole version offering the \"max\" reasoning-effort level; use 1.2 instead if your workload feeds it audio, which 1.3 handles poorly."},{"slug":"meta-muse-spark-1-3-contributor-tier","name":"Muse Spark 1.3 (Contributor tier)","provider":"Meta","api_id":"muse-spark-1.3-contributor","status":"ga","context_window":1048576,"max_output_tokens":0,"input_price":0.1,"output_price":0.2,"cached_input_price":0.002,"modalities_in":["text","image","video","audio","pdf"],"capabilities":["chat","reasoning","tool-calling","tool-search","parallel-tool-calls","structured-output","prompt-caching","search-grounding","image-understanding","video-understanding","file-handling","streaming"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Same model at roughly a twelfth the input cost in exchange for letting Meta train on your prompts and completions — fine for prototyping and load-testing, disqualifying for anything covered by a customer confidentiality or DPA obligation; note the tier is also throttled to 100 RPM."},{"slug":"meta-muse-spark-1-2","name":"Muse Spark 1.2","provider":"Meta","api_id":"muse-spark-1.2","status":"ga","context_window":1048576,"max_output_tokens":0,"input_price":1.25,"output_price":4.25,"cached_input_price":0.15,"modalities_in":["text","image","video","audio","pdf"],"capabilities":["chat","reasoning","tool-calling","structured-output","prompt-caching","search-grounding","image-understanding","video-understanding","audio-understanding","file-handling","streaming"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"The version to choose when your prompts contain audio — 1.2 supports audio understanding fully while 1.3 does not, and it costs exactly the same; otherwise 1.3 supersedes it."},{"slug":"meta-muse-spark-1-2-contributor-tier","name":"Muse Spark 1.2 (Contributor tier)","provider":"Meta","api_id":"muse-spark-1.2-contributor","status":"ga","context_window":1048576,"max_output_tokens":0,"input_price":0.1,"output_price":0.2,"cached_input_price":0.002,"modalities_in":["text","image","video","audio","pdf"],"capabilities":["chat","reasoning","tool-calling","structured-output","prompt-caching","search-grounding","image-understanding","video-understanding","audio-understanding","file-handling","streaming"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Cheapest route to full audio understanding on Meta's API, but only for non-confidential data — the discount is paid for with training rights over your traffic."},{"slug":"meta-muse-spark-1-1","name":"Muse Spark 1.1","provider":"Meta","api_id":"muse-spark-1.1","status":"ga","context_window":1048576,"max_output_tokens":0,"input_price":1.25,"output_price":4.25,"cached_input_price":0.15,"modalities_in":["text","image","video","audio","pdf"],"capabilities":["chat","reasoning","tool-calling","structured-output","prompt-caching","search-grounding","image-understanding","video-understanding","audio-understanding","file-handling","streaming"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"The original release, priced identically to 1.3 with no Contributor discount available — pin it only to hold behaviour stable for an existing eval baseline, never for new builds."},{"slug":"meta-muse-image-1-0","name":"Muse Image 1.0","provider":"Meta","api_id":"muse-image-1.0","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["image-generation","image-editing","multi-turn-editing","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Billed per image, not per token: a flat $0.01 per generated image regardless of prompt length, reasoning strength, or the web/image search and code execution it runs internally — a request returning n images costs n cents, and images that fail or are safety-filtered are not charged. Both token prices are -1 because Meta publishes no per-token rate for it. Use the Responses API for conversational multi-turn editing and /v1/images/generations or /v1/images/edits for one-shot calls."},{"slug":"meta-muse-voice-transcribe-1-0","name":"Muse Voice Transcribe 1.0","provider":"Meta","api_id":"muse-voice-transcribe-1.0","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":["speech-to-text","realtime-streaming","speaker-diarization","voice-activity-detection","endpointing","keyword-biasing","turn-level-timestamps","multilingual"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Billed per audio-hour, not per token: $0.18 per hour of audio processed, streaming and file transcription priced the same, with zero-data-retention at price parity and no discounted training-eligible tier — both token prices are -1 because Meta publishes none. Good for voice agents and call/meeting intelligence, but rule it out if you need word-level timestamps, sound-event or emotion detection, or speech synthesis; it does none of those, and the 8-concurrent-stream cap constrains fan-out."},{"slug":"meta-muse-glimmer-30b","name":"Muse Glimmer 30B","provider":"Meta","api_id":"meta-models/Muse-Glimmer-30B","status":"ga","context_window":131072,"max_output_tokens":-1,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["chat","reasoning","tool-calling","image-understanding","local-inference","fine-tuning","quantization","speculative-decoding","multilingual"],"flagship":false,"knowledge_cutoff":"","license":"Apache License 2.0","released":"","best_for":"The one Meta open-weight model with a genuinely permissive licence (Apache 2.0, no Community-License user-count or naming clauses) — a 30B dense multimodal model distilled from Muse Spark, with 128K default context and quantized GGUF builds that fit 24-32GB VRAM. Weights only: Meta charges nothing and sells no API for it, so both prices are -1 and your cost is your own hardware or a third-party host."},{"slug":"meta-llama-4-scout","name":"Llama 4 Scout","provider":"Meta","api_id":"meta-llama/Llama-4-Scout-17B-16E-Instruct","status":"ga","context_window":10000000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["chat","multilingual","image-understanding","early-fusion-multimodal","long-context","fine-tuning","distillation"],"flagship":true,"knowledge_cutoff":"2024-08","license":"Llama 4 Community License Agreement","released":"2025-04-05","best_for":"The long-context option in Meta's open-weight line — a 10M-token window and single-H100 deployability at int4 make it the pick for whole-repository or whole-corpus ingestion; Meta sells no API for it, so both prices are -1 and any per-token figure you see belongs to a third-party host."},{"slug":"meta-llama-4-maverick","name":"Llama 4 Maverick","provider":"Meta","api_id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct","status":"ga","context_window":1000000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["chat","multilingual","image-understanding","early-fusion-multimodal","long-context","fine-tuning","distillation"],"flagship":false,"knowledge_cutoff":"2024-08","license":"Llama 4 Community License Agreement","released":"2025-04-05","best_for":"Higher-quality sibling to Scout with the same 17B activated cost per token but 400B total weights — better reasoning and image understanding, at the price of multi-GPU hosting and a shorter 1M window. Weights-only from Meta: both prices -1."},{"slug":"meta-llama-3-3-70b-instruct","name":"Llama 3.3 70B Instruct","provider":"Meta","api_id":"meta-llama/Llama-3.3-70B-Instruct","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["chat","multilingual","tool-calling","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-12","license":"Llama 3.3 Community License Agreement","released":"2024-12-06","best_for":"The best quality-per-GPU text-only Llama: 405B-class instruction following at 70B serving cost, which is why it remains the most widely hosted Llama despite Llama 4. Text only — reach for Llama 4 if you need vision or a window past 128K. Meta publishes no price; both prices -1."},{"slug":"meta-llama-3-2-90b-vision-instruct","name":"Llama 3.2 90B Vision Instruct","provider":"Meta","api_id":"meta-llama/Llama-3.2-90B-Vision-Instruct","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["chat","image-understanding","document-understanding","chart-reasoning","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-12","license":"Llama 3.2 Community License Agreement","released":"2024-09-25","best_for":"A cross-attention vision adapter bolted onto Llama 3.1 70B — capable at charts, documents and captioning, but image+text is English-only and it is superseded by Llama 4's native early fusion; pick it only when you need a 3.x-lineage vision model. Prices -1: weights only."},{"slug":"meta-llama-3-2-11b-vision-instruct","name":"Llama 3.2 11B Vision Instruct","provider":"Meta","api_id":"meta-llama/Llama-3.2-11B-Vision-Instruct","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["chat","image-understanding","document-understanding","captioning","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-12","license":"Llama 3.2 Community License Agreement","released":"2024-09-25","best_for":"Small enough to fine-tune for a single narrow vision task (receipt or form extraction, screenshot classification) on one commodity GPU; not a general-purpose VLM. Weights only, so both prices -1."},{"slug":"meta-llama-3-2-3b-instruct","name":"Llama 3.2 3B Instruct","provider":"Meta","api_id":"meta-llama/Llama-3.2-3B-Instruct","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["chat","multilingual","summarization","retrieval","on-device","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-12","license":"Llama 3.2 Community License Agreement","released":"2024-10-24","best_for":"On-device and edge summarisation, rewriting and query expansion where latency and privacy beat raw capability; note Meta's own quantized builds drop the context window from 128K to 8K. Prices -1 — weights only."},{"slug":"meta-llama-3-2-1b-instruct","name":"Llama 3.2 1B Instruct","provider":"Meta","api_id":"meta-llama/Llama-3.2-1B-Instruct","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["chat","multilingual","summarization","on-device","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-12","license":"Llama 3.2 Community License Agreement","released":"2024-10-24","best_for":"The smallest Llama, for phone- and microcontroller-class deployment or as a speculative-decoding draft model; expect to fine-tune it for one task rather than use it as a general assistant. Weights only: both prices -1."},{"slug":"meta-llama-3-1-405b-instruct","name":"Llama 3.1 405B Instruct","provider":"Meta","api_id":"meta-llama/Llama-3.1-405B-Instruct","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["chat","multilingual","tool-calling","synthetic-data-generation","distillation","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-12","license":"Llama 3.1 Community License Agreement","released":"2024-07-23","best_for":"Now mainly a teacher model: its licence explicitly permits using outputs to train other models, which is the remaining reason to run something this expensive when Llama 3.3 70B matches it on most chat benchmarks. Meta charges nothing and hosts nothing — both prices -1."},{"slug":"meta-llama-3-1-70b-instruct","name":"Llama 3.1 70B Instruct","provider":"Meta","api_id":"meta-llama/Llama-3.1-70B-Instruct","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["chat","multilingual","tool-calling","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-12","license":"Llama 3.1 Community License Agreement","released":"2024-07-23","best_for":"Superseded at the same size and cost by Llama 3.3 70B — keep it only for pinned reproducibility of existing 3.1 evaluations or fine-tunes. Weights only, so both prices -1."},{"slug":"meta-llama-3-1-8b-instruct","name":"Llama 3.1 8B Instruct","provider":"Meta","api_id":"meta-llama/Llama-3.1-8B-Instruct","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["chat","multilingual","tool-calling","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-12","license":"Llama 3.1 Community License Agreement","released":"2024-07-23","best_for":"Still the default single-GPU fine-tuning baseline across the open-source ecosystem because tooling support is universal; choose Llama 3.2 3B instead if you need something smaller and Muse Glimmer 30B if licence permissiveness matters. Prices -1 — weights only."},{"slug":"xai-grok-4-6","name":"Grok 4.6","provider":"xAI","api_id":"grok-4.6","status":"ga","context_window":500000,"max_output_tokens":0,"input_price":2,"output_price":6,"cached_input_price":0.5,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output","web-search","code-execution"],"flagship":true,"knowledge_cutoff":"2026-01 (Grok 4.6 overview page states January 2026; the Models page note states February 1, 2026)","license":"proprietary","released":"2026-08-12","best_for":"The default pick on xAI today - flagship coding and agentic tool-calling with reasoning_effort low/medium/high/xhigh, no output-token limit, and 500K context; note prompts at or above 200K tokens double to $4/$1/$12 and it cannot use the Batch API, so long-context batch jobs are cheaper on grok-4.3."},{"slug":"xai-grok-4-5","name":"Grok 4.5","provider":"xAI","api_id":"grok-4.5","status":"ga","context_window":500000,"max_output_tokens":0,"input_price":2,"output_price":6,"cached_input_price":0.3,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-07-08","best_for":"Previous flagship at the same headline price as 4.6 but with cheaper cache reads ($0.30 vs $0.50) - keep it only for cache-heavy pipelines already tuned to it, or for EU console teams, since it is the one Grok 4.x model explicitly listed as EU-available. Aliases: grok-4.5-latest, grok-build-latest. Long context (>=200K) bills $4.00/$0.60/$12.00."},{"slug":"xai-grok-4-3","name":"Grok 4.3","provider":"xAI","api_id":"grok-4.3","status":"ga","context_window":1000000,"max_output_tokens":0,"input_price":1.25,"output_price":2.5,"cached_input_price":0.2,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":true,"knowledge_cutoff":"","license":"proprietary","released":"2026-04","best_for":"Best price/performance workhorse and the only 1M-context Grok that is both cheap and batchable - configurable reasoning effort (none/low/medium/high) lets one slug cover latency-sensitive non-reasoning traffic and deeper reasoning alike; it is also the redirect target for every model retired on 2026-05-15. Long context (>=200K) bills $2.50/$0.40/$5.00."},{"slug":"xai-grok-build-0-1","name":"Grok Build 0.1","provider":"xAI","api_id":"grok-build-0.1","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":1,"output_price":2,"cached_input_price":0.2,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-05-19","best_for":"Cheapest xAI model and the purpose-built agentic-coding one - successor to grok-code-fast-1 (whose slug now redirects here); pick it for high-volume IDE/agent loops where per-token cost beats frontier reasoning, but its 256K context is the smallest on offer and it has no Batch support. Aliases: grok-code-fast-1, grok-code-fast, grok-code-fast-1-0825. Long context (>=200K) bills $2.00/$0.40/$4.00."},{"slug":"xai-grok-4-20-reasoning","name":"Grok 4.20 (Reasoning)","provider":"xAI","api_id":"grok-4.20-0309-reasoning","status":"ga","context_window":1000000,"max_output_tokens":0,"input_price":1.25,"output_price":2.5,"cached_input_price":0.2,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-03-10","best_for":"Pinned dated reasoning snapshot at the same price as grok-4.3 - choose it only when you need version-frozen reproducibility; new work should use grok-4.3 or grok-4.6. logprobs/top_logprobs are silently ignored on 4.20 and newer. Long context (>=200K) bills $2.50/$0.40/$5.00."},{"slug":"xai-grok-4-20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","provider":"xAI","api_id":"grok-4.20-0309-non-reasoning","status":"ga","context_window":1000000,"max_output_tokens":0,"input_price":1.25,"output_price":2.5,"cached_input_price":0.2,"modalities_in":["text","image"],"capabilities":["tools","vision","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-03-10","best_for":"Low-latency non-reasoning twin of 4.20 at an identical token price - no cost saving over the reasoning slug, so it only wins where you must guarantee zero thinking tokens; grok-4.3 with reasoning_effort=none is the modern equivalent. Long context (>=200K) bills $2.50/$0.40/$5.00."},{"slug":"xai-grok-4-20-multi-agent","name":"Grok 4.20 Multi-Agent","provider":"xAI","api_id":"grok-4.20-multi-agent-0309","status":"beta","context_window":1000000,"max_output_tokens":0,"input_price":1.25,"output_price":2.5,"cached_input_price":0.2,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-03-10","best_for":"Deep-research slug that fans out multiple agents in parallel inside one API call - same per-token price as 4.20 but total token burn per request is far higher and rate limits are ~4x tighter (9 RPS / 2.5M TPM at Tier 0), so budget for it as a research tool, not a chat backend. Long context (>=200K) bills $2.50/$0.40/$5.00."},{"slug":"xai-grok-imagine-image-2-0","name":"Grok Imagine Image 2.0","provider":"xAI","api_id":"grok-imagine-image-2.0","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["batch","vision"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"Current image generation/editing default - priced per image, not per token: $0.04 (1K low), $0.06 (2K low), $0.06 (1K medium), $0.08 (2K medium), plus $0.01 per input image for editing. quality defaults to auto (low for generation, medium for editing); accepts up to 5 reference images and 21:9 / 5:2 aspect ratios. Batch accepted but at standard rates."},{"slug":"xai-grok-imagine-image","name":"Grok Imagine Image","provider":"xAI","api_id":"grok-imagine-image","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["batch","vision"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-03-02","best_for":"The cheap 1.0 image model - flat $0.02/image at both 1K and 2K plus $0.002 per input image; the volume choice when you do not need 2.0's quality tiers, and explicitly unaffected by the November 2026 retirement. Alias: grok-imagine-image-2026-03-02."},{"slug":"xai-grok-imagine-image-quality","name":"Grok Imagine Image Quality","provider":"xAI","api_id":"grok-imagine-image-quality","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-04-03","best_for":"Do not start new work here - retires 2026-11-02, after which the slug silently serves grok-imagine-image-2.0 at quality=low. Current per-image pricing is $0.05 (1K) / $0.07 (2K) plus $0.01 per input image, i.e. $0.01 dearer than its replacement at every resolution. No Batch support. Aliases: grok-imagine-image-quality-latest, grok-imagine-image-quality-20260403, grok-imagine-image-pro."},{"slug":"xai-grok-imagine-video-1-5","name":"Grok Imagine Video 1.5","provider":"xAI","api_id":"grok-imagine-video-1.5","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","audio"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-05-30","best_for":"The video model to use when you need 1080p or audio-conditioned output - billed per second of video: $0.08 (480p), $0.14 (720p), $0.25 (1080p), plus $0.01 per input image; preset-voice audio input is free. Supports text-to-video, image-to-video and reference-to-video. Aliases: grok-imagine-video-1.5-preview, grok-imagine-video-1.5-2026-05-30."},{"slug":"xai-grok-imagine-video","name":"Grok Imagine Video","provider":"xAI","api_id":"grok-imagine-video","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-01-28","best_for":"Budget video option and the only one accepting video input (used for video extension/editing) - $0.05/sec at 480p and $0.07/sec at 720p, no 1080p; inputs cost $0.01/sec of video and $0.002/image. Video Extension pricing is flagged as a promotional rate subject to change."},{"slug":"xai-grok-voice-think-fast-2-0","name":"Grok Voice Think Fast 2.0","provider":"xAI","api_id":"grok-voice-think-fast-2.0","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","audio"],"capabilities":["tools","streaming","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-07-29","best_for":"Current realtime speech-to-speech model over WebSocket with function calling plus web/X/collections/MCP search - billed by time and events, not tokens: $0.08/min ($4.80/hr) of audio sent or received, plus $0.004 per client conversation.item.create text event (tool results and audio items excluded). Max session 120 minutes, us-east-1 only. Alias: grok-voice-latest."},{"slug":"xai-grok-voice-think-fast-1-0","name":"Grok Voice Think Fast 1.0","provider":"xAI","api_id":"grok-voice-think-fast-1.0","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","audio"],"capabilities":["tools","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-04-23","best_for":"Marked Deprecated in xAI's pricing table but still served and still cheaper on audio than 2.0 - $0.05/min ($3.00/hr) audio plus $0.004 per text input event; worth keeping only for cost-sensitive legacy voice sessions, and grok-voice-latest has pointed at 2.0 since 2026-08-05."},{"slug":"xai-xai-speech-to-text","name":"xAI Speech to Text","provider":"xAI","api_id":"","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-04-15","best_for":"Transcription billed per hour of audio, not per token: $0.10/hr on the REST/batch endpoint and $0.20/hr streaming - very cheap for bulk file transcription. Supports WAV/MP3/WebM/OGG/M4A, multiple languages, keyterm prompting, tunable vad_threshold and ML Smart Turn end-of-turn detection on streaming. No public model slug; addressed by endpoint. us-east-1 only."},{"slug":"xai-xai-text-to-speech","name":"xAI Text to Speech","provider":"xAI","api_id":"","status":"ga","context_window":0,"max_output_tokens":0,"input_price":15,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-03-16","best_for":"Speech synthesis billed at $15.00 per 1M input CHARACTERS - the input_price field here is per 1M chars, not tokens. Supports expressive built-in voices, cloned custom voices shared with the Speech-to-Speech API, streaming or batch output, and MP3/WAV/PCM/mu-law/A-law. No public model slug; addressed by endpoint. us-east-1 only."},{"slug":"deepseek-deepseek-v4-flash","name":"DeepSeek-V4-Flash","provider":"DeepSeek","api_id":"deepseek-v4-flash","status":"ga","context_window":1000000,"max_output_tokens":384000,"input_price":0.44,"output_price":1.32,"cached_input_price":0.014,"modalities_in":["text"],"capabilities":["tools","reasoning","prompt-caching","streaming","structured-output"],"flagship":true,"knowledge_cutoff":"","license":"MIT","released":"2026-07-31","best_for":"The default DeepSeek pick: near-Pro reasoning at a third of the price with the same 1M context, ideal for high-volume agentic coding and long-document work — schedule batchy jobs outside 01:00-04:00/06:00-10:00 UTC weekdays and you pay $0.22/$0.66 instead of $0.44/$1.32."},{"slug":"deepseek-deepseek-v4-pro","name":"DeepSeek-V4-Pro","provider":"DeepSeek","api_id":"deepseek-v4-pro","status":"ga","context_window":1000000,"max_output_tokens":384000,"input_price":1.32,"output_price":3.96,"cached_input_price":0.044,"modalities_in":["text"],"capabilities":["tools","reasoning","prompt-caching","streaming","structured-output"],"flagship":true,"knowledge_cutoff":"","license":"MIT","released":"2026-08-13","best_for":"DeepSeek's frontier tier — worth the 3x premium over Flash only for hard agentic/production coding, deep reasoning and tool-heavy workflows where Flash's quality gap actually shows; note the tighter 500-concurrent limit and use reasoning_effort=max sparingly since thinking tokens bill at the output rate."},{"slug":"deepseek-deepseek-v4-flash-vision-exp","name":"DeepSeek-V4-Flash-Vision-Exp","provider":"DeepSeek","api_id":"deepseek-v4-flash-vision-exp","status":"preview","context_window":1000000,"max_output_tokens":384000,"input_price":0.44,"output_price":1.32,"cached_input_price":0.014,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"2026-08-21","best_for":"The only DeepSeek model that takes images — use it for screenshot-driven and chart/document-reading agents at identical Flash pricing; it is explicitly experimental and drops FIM completion, so keep text-only traffic on deepseek-v4-flash. Images bill as input tokens (up to 384 tokens each), accepted as base64, public URL (max 32 MiB), or Files API file_id (max 64 MiB, upload is free)."},{"slug":"alibaba-qwen-qwen3-8-max","name":"Qwen3.8-Max","provider":"Alibaba Qwen","api_id":"qwen3.8-max","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":2,"output_price":6,"cached_input_price":0.2,"modalities_in":["text","image","video"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search"],"flagship":true,"knowledge_cutoff":"","license":"proprietary","released":"2026-08-03","best_for":"Qwen's frontier flagship — pick it when you need top-end reasoning plus a genuine 1M-token multimodal context; reasoning is on by default and the cached-input rate ($0.20/1M, 10% of standard) makes long repeated system prompts cheap."},{"slug":"alibaba-qwen-qwen3-7-max","name":"Qwen3.7-Max","provider":"Alibaba Qwen","api_id":"qwen3.7-max","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":2.5,"output_price":7.5,"cached_input_price":0.25,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":true,"knowledge_cutoff":"","license":"proprietary","released":"2026-05","best_for":"Previous-generation Max, now strictly more expensive than Qwen3.8-Max ($2.50/$7.50 vs $2/$6) — only worth pinning if you have validated prompts against its snapshots (qwen3.7-max-2026-05-20, -2026-06-08)."},{"slug":"alibaba-qwen-qwen3-max","name":"Qwen3-Max","provider":"Alibaba Qwen","api_id":"qwen3-max","status":"ga","context_window":262144,"max_output_tokens":65536,"input_price":1.2,"output_price":6,"cached_input_price":0.12,"modalities_in":["text"],"capabilities":["tools","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-09","best_for":"Legacy 2025 flagship, still served but superseded; steeply tiered by input length (0-32K $1.20/$6, 32-128K $2.40/$12, 128-256K $3/$15) so long-context calls cost more than Qwen3.8-Max — migrate off it."},{"slug":"alibaba-qwen-qwen-max-legacy","name":"Qwen-Max (legacy)","provider":"Alibaba Qwen","api_id":"qwen-max","status":"deprecated","context_window":32768,"max_output_tokens":8192,"input_price":1.6,"output_price":6.4,"cached_input_price":0.32,"modalities_in":["text"],"capabilities":["tools","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-01-25","best_for":"Avoid for new work — a 32K-context 2025 model priced near the 1M-context Qwen3.8-Max; kept alive only for the qwen-max-2025-01-25 snapshot pin."},{"slug":"alibaba-qwen-qwen3-7-plus","name":"Qwen3.7-Plus","provider":"Alibaba Qwen","api_id":"qwen3.7-plus","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":0.4,"output_price":1.6,"cached_input_price":0.04,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06","best_for":"The best price/performance default in the lineup — 1M context and near-Max quality at a fifth of Max's price; two tiers only (0-256K $0.40/$1.60, 256K-1M $1.20/$4.80), so keep requests under 256K to stay on the cheap bracket."},{"slug":"alibaba-qwen-qwen3-5-plus","name":"Qwen3.5-Plus","provider":"Alibaba Qwen","api_id":"qwen3.5-plus","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":0.4,"output_price":2.4,"cached_input_price":0.04,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"2026-01","license":"proprietary","released":"2026-02","best_for":"Superseded by Qwen3.7-Plus at the same input price but 50% higher output cost ($2.40 vs $1.60) — no reason to choose it new; tiers are 0-256K $0.40/$2.40 and 256K-1M $0.50/$3."},{"slug":"alibaba-qwen-qwen-plus-rolling","name":"Qwen-Plus (rolling)","provider":"Alibaba Qwen","api_id":"qwen-plus","status":"ga","context_window":1000000,"max_output_tokens":32768,"input_price":0.4,"output_price":1.2,"cached_input_price":0.04,"modalities_in":["text"],"capabilities":["tools","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-08-26","best_for":"The cheapest output price in the Plus tier ($1.20/1M) and a rolling alias that auto-upgrades; max thinking length 81,920 tokens but output capped at 32K — use it for high-volume chat, not long-form generation. Tiers: 0-256K $0.40/$1.20, 256K-1M $1.20/$3.60."},{"slug":"alibaba-qwen-qwen3-8-flash","name":"Qwen3.8-Flash","provider":"Alibaba Qwen","api_id":"qwen3.8-flash","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":0.15,"output_price":0.47,"cached_input_price":0.015,"modalities_in":["text","image","video"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-08","best_for":"Flat-rate 1M context with no tier cliff — the standout value pick for long-document and agentic work where Qwen3.7-Flash's 32K cheap bracket would be blown immediately. Hosted counterpart of the open Qwen3.8-Flash-Next weights."},{"slug":"alibaba-qwen-qwen3-7-flash","name":"Qwen3.7-Flash","provider":"Alibaba Qwen","api_id":"qwen3.7-flash","status":"ga","context_window":1000000,"max_output_tokens":65536,"input_price":0.03,"output_price":0.13,"cached_input_price":0.003,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06","best_for":"The absolute cheapest Qwen model for short prompts ($0.03/$0.13 under 32K), but pricing steps hard to $0.10/$0.40 at 32-256K and $0.20/$0.80 above — only economical if you genuinely keep inputs tiny."},{"slug":"alibaba-qwen-qwen3-5-flash","name":"Qwen3.5-Flash","provider":"Alibaba Qwen","api_id":"qwen3.5-flash","status":"ga","context_window":1000000,"max_output_tokens":65536,"input_price":0.1,"output_price":0.4,"cached_input_price":0.01,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"2026-01","license":"proprietary","released":"2026-02","best_for":"Flat $0.10/$0.40 across the full 1M window — simpler to budget than Qwen3.7-Flash's tiers, but Qwen3.8-Flash is newer for only modestly more."},{"slug":"alibaba-qwen-qwen-flash-rolling","name":"Qwen-Flash (rolling)","provider":"Alibaba Qwen","api_id":"qwen-flash","status":"ga","context_window":1000000,"max_output_tokens":32768,"input_price":0.05,"output_price":0.4,"cached_input_price":0.005,"modalities_in":["text"],"capabilities":["tools","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-08-26","best_for":"Rolling cheap alias that auto-tracks the current Flash generation; max input 997,952 tokens, output capped at 32K. Tiers: 0-256K $0.05/$0.40, 256K-1M $0.25/$2."},{"slug":"alibaba-qwen-qwen-turbo","name":"Qwen-Turbo","provider":"Alibaba Qwen","api_id":"qwen-turbo","status":"deprecated","context_window":1000000,"max_output_tokens":16384,"input_price":0.05,"output_price":0.2,"cached_input_price":0.005,"modalities_in":["text"],"capabilities":["tools","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025","best_for":"Legacy budget tier (price quoted for non-thinking mode); the Flash family has replaced it — migrate to qwen-flash or qwen3.8-flash."},{"slug":"alibaba-qwen-qwq-plus","name":"QwQ-Plus","provider":"Alibaba Qwen","api_id":"qwq-plus","status":"ga","context_window":131072,"max_output_tokens":32768,"input_price":0.8,"output_price":2.4,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-03","best_for":"Alibaba's first dedicated reasoning model, kept alive for compatibility only — Qwen3.7-Plus reasons better at half the input price."},{"slug":"alibaba-qwen-qwen3-vl-plus","name":"Qwen3-VL-Plus","provider":"Alibaba Qwen","api_id":"qwen3-vl-plus","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.2,"output_price":1.6,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-09","best_for":"Dedicated vision-language endpoint for image/video understanding when you don't need Max-tier text reasoning; tiers 0-32K $0.20/$1.60, 32-128K $0.30/$2.40, 128-256K $0.60/$4.80."},{"slug":"alibaba-qwen-qwen3-vl-flash","name":"Qwen3-VL-Flash","provider":"Alibaba Qwen","api_id":"qwen3-vl-flash","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.05,"output_price":0.4,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-09","best_for":"Cheapest hosted vision option for bulk image classification/captioning; tiers 0-32K $0.05/$0.40, 32-128K $0.075/$0.60, 128-256K $0.12/$0.96."},{"slug":"alibaba-qwen-qvq-max","name":"QVQ-Max","provider":"Alibaba Qwen","api_id":"qvq-max","status":"ga","context_window":131072,"max_output_tokens":8192,"input_price":1.2,"output_price":4.8,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","reasoning","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-03","best_for":"Legacy visual-reasoning model; superseded by Qwen3-VL-Plus and the multimodal Max/Plus tiers at lower cost."},{"slug":"alibaba-qwen-qwen3-5-omni-plus","name":"Qwen3.5-Omni-Plus","provider":"Alibaba Qwen","api_id":"qwen3.5-omni-plus","status":"ga","context_window":131072,"max_output_tokens":16384,"input_price":1.4,"output_price":8.3,"cached_input_price":-1,"modalities_in":["text","image","audio","video"],"capabilities":["tools","vision","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-02","best_for":"Any-to-any speech/vision/text model for voice agents — but audio is billed separately and steeply: text/image input $1.40/1M vs audio input $11/1M, text output $8.30/1M vs audio output $44/1M. A realtime variant (qwen3.5-omni-plus-realtime) also exists."},{"slug":"alibaba-qwen-qwen3-5-omni-flash","name":"Qwen3.5-Omni-Flash","provider":"Alibaba Qwen","api_id":"qwen3.5-omni-flash","status":"ga","context_window":131072,"max_output_tokens":16384,"input_price":0.4,"output_price":2.2,"cached_input_price":-1,"modalities_in":["text","image","audio","video"],"capabilities":["tools","vision","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-02","best_for":"Budget voice/multimodal agent tier — text/image input $0.40/1M, audio input $3/1M, text output $2.20/1M, audio output $11.90/1M. Roughly 3.5x cheaper than Omni-Plus across the board."},{"slug":"alibaba-qwen-qwen3-coder-plus","name":"Qwen3-Coder-Plus","provider":"Alibaba Qwen","api_id":"qwen3-coder-plus","status":"ga","context_window":1000000,"max_output_tokens":65536,"input_price":1,"output_price":5,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","batch","streaming","structured-output","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-07","best_for":"Hosted agentic coding model tuned for repo-scale edits, but the tier ramp is brutal — 0-32K $1/$5, 32-128K $1.80/$9, 128-256K $3/$15, 256K-1M $6/$60. Check whether qwen3-coder-next at $0.30/$1.50 covers your workload first."},{"slug":"alibaba-qwen-qwen-vl-ocr","name":"Qwen-VL-OCR","provider":"Alibaba Qwen","api_id":"qwen-vl-ocr","status":"ga","context_window":34096,"max_output_tokens":4096,"input_price":0.07,"output_price":0.16,"cached_input_price":-1,"modalities_in":["text","image","pdf"],"capabilities":["vision","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025","best_for":"Purpose-built document/text extraction at $0.07/$0.16 per 1M — an order of magnitude cheaper than routing scans through a general VL model."},{"slug":"alibaba-qwen-qwen-mt-plus","name":"Qwen-MT-Plus","provider":"Alibaba Qwen","api_id":"qwen-mt-plus","status":"ga","context_window":16384,"max_output_tokens":8192,"input_price":2.46,"output_price":7.37,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025","best_for":"Specialist machine-translation endpoint with terminology/domain-prompt control across 90+ languages; premium priced — use qwen-mt-flash unless translation quality is contractual."},{"slug":"alibaba-qwen-qwen-mt-flash","name":"Qwen-MT-Flash","provider":"Alibaba Qwen","api_id":"qwen-mt-flash","status":"ga","context_window":16384,"max_output_tokens":8192,"input_price":0.16,"output_price":0.49,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025","best_for":"High-volume translation at ~15x less than Qwen-MT-Plus; the right default for localisation pipelines. Cheaper qwen-mt-turbo/lite variants exist from $0.12/$0.36."},{"slug":"alibaba-qwen-qwen3-7-text-embedding","name":"Qwen3.7-Text-Embedding","provider":"Alibaba Qwen","api_id":"qwen3.7-text-embedding","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0.07,"output_price":0,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06","best_for":"Current best Qwen embedding — 128K tokens per item (16x text-embedding-v4), Matryoshka dimensions 256-2560, 200+ languages, same $0.07/1M price. Default choice for new RAG builds."},{"slug":"alibaba-qwen-text-embedding-v4-qwen3-embedding","name":"Text-Embedding-V4 (Qwen3-Embedding)","provider":"Alibaba Qwen","api_id":"text-embedding-v4","status":"ga","context_window":8192,"max_output_tokens":0,"input_price":0.07,"output_price":0,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-06","best_for":"Widely-deployed embedding with dimensions from 64 to 2048 — pick the 64/128 dims to shrink vector-DB cost; 100+ languages, 8K tokens per item, max 10 items per call."},{"slug":"alibaba-qwen-text-embedding-v3","name":"Text-Embedding-V3","provider":"Alibaba Qwen","api_id":"text-embedding-v3","status":"ga","context_window":8192,"max_output_tokens":0,"input_price":0.07,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024","best_for":"Superseded by v4 (fewer dimension options, 50+ languages, smaller 500K free quota); no Singapore price published on the official pricing page — migrate to text-embedding-v4."},{"slug":"alibaba-qwen-qwen3-rerank","name":"Qwen3-Rerank","provider":"Alibaba Qwen","api_id":"qwen3-rerank","status":"ga","context_window":30000,"max_output_tokens":0,"input_price":0.1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025","best_for":"Cross-encoder reranker for the second stage of a RAG pipeline; listed as a served model in Model Studio but no Singapore-region price appears on the official pricing page — confirm in console before budgeting."},{"slug":"alibaba-qwen-tongyi-embedding-vision-plus","name":"Tongyi-Embedding-Vision-Plus","provider":"Alibaba Qwen","api_id":"tongyi-embedding-vision-plus","status":"ga","context_window":1024,"max_output_tokens":0,"input_price":0.09,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["vision"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025","best_for":"Multimodal embedding for cross-modal (text-to-image) retrieval; no published Singapore price on the official pricing page."},{"slug":"alibaba-qwen-qwen-image-3-0-pro","name":"Qwen-Image-3.0-Pro","provider":"Alibaba Qwen","api_id":"qwen-image-3.0-pro","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026","best_for":"Current text-to-image and image-editing flagship, billed per image rather than per token; the per-image rate is not published on the English model-pricing page — check the console."},{"slug":"alibaba-qwen-qwen3-8-2-4t-a95b","name":"Qwen3.8-2.4T-A95B","provider":"Alibaba Qwen","api_id":"qwen3.8-2.4t-a95b","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":2,"output_price":6,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"Qwen3.8-Max License (custom, non-Apache)","released":"2026-08","best_for":"The published base weights behind Qwen3.8-Max and the largest open-weight model in existence — text-only, thinking-mode-only, 262K native context (YaRN-extensible to ~1.01M). Note it ships under a custom licence, NOT Apache 2.0; the hosted qwen3.8-max adds vision, non-thinking mode and 1M context by default."},{"slug":"alibaba-qwen-qwen3-8-27b","name":"Qwen3.8-27B","provider":"Alibaba Qwen","api_id":"qwen3.8-27b","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":0.5,"output_price":3,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-08-14","best_for":"The standout self-host pick — a truly Apache-2.0, 27B dense multimodal model with 262K context and adjustable reasoning effort (xhigh/medium/low) that fits on a single high-memory GPU in BF16 or comfortably in FP8."},{"slug":"alibaba-qwen-qwen3-6-35b-a3b","name":"Qwen3.6-35B-A3B","provider":"Alibaba Qwen","api_id":"qwen3.6-35b-a3b","status":"ga","context_window":262144,"max_output_tokens":65536,"input_price":0.375,"output_price":2.25,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-04-16","best_for":"Only 3B active parameters, so it runs fast on modest hardware while punching well above its weight on coding benchmarks — the sweet spot for local agentic coding under Apache 2.0. Thinking is enabled by default on the Qwen3.6 series."},{"slug":"alibaba-qwen-qwen3-6-27b","name":"Qwen3.6-27B","provider":"Alibaba Qwen","api_id":"qwen3.6-27b","status":"ga","context_window":262144,"max_output_tokens":65536,"input_price":0.6,"output_price":3.6,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-04","best_for":"Dense 27B for workloads where MoE routing hurts quality consistency; now priced above the newer Qwen3.8-27B ($0.60/$3.60 vs $0.50/$3) — prefer 3.8-27B unless pinned."},{"slug":"alibaba-qwen-qwen3-5-397b-a17b","name":"Qwen3.5-397B-A17B","provider":"Alibaba Qwen","api_id":"qwen3.5-397b-a17b","status":"ga","context_window":262144,"max_output_tokens":65536,"input_price":0.6,"output_price":3.6,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2026-01","license":"Apache-2.0","released":"2026-02-16","best_for":"The largest fully Apache-2.0 Qwen model — the one to self-host when licence purity matters more than raw scale and the custom-licensed Qwen3.8-2.4T is off the table. 262K native, YaRN-extensible to ~1.01M."},{"slug":"alibaba-qwen-qwen3-5-122b-a10b","name":"Qwen3.5-122B-A10B","provider":"Alibaba Qwen","api_id":"qwen3.5-122b-a10b","status":"ga","context_window":262144,"max_output_tokens":65536,"input_price":0.4,"output_price":3.2,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2026-01","license":"Apache-2.0","released":"2026-02","best_for":"Mid-size Apache-2.0 MoE that fits a single 8-GPU node in FP8 — a reasonable step down from the 397B when memory is the constraint."},{"slug":"alibaba-qwen-qwen3-5-35b-a3b","name":"Qwen3.5-35B-A3B","provider":"Alibaba Qwen","api_id":"qwen3.5-35b-a3b","status":"ga","context_window":262144,"max_output_tokens":65536,"input_price":0.25,"output_price":2,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2026-01","license":"Apache-2.0","released":"2026-02","best_for":"Cheapest hosted Qwen3.5 open model and an easy local run at 3B active params; superseded on quality by Qwen3.6-35B-A3B for a small price premium."},{"slug":"alibaba-qwen-qwen3-5-27b","name":"Qwen3.5-27B","provider":"Alibaba Qwen","api_id":"qwen3.5-27b","status":"ga","context_window":262144,"max_output_tokens":65536,"input_price":0.3,"output_price":2.4,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2026-01","license":"Apache-2.0","released":"2026-02","best_for":"Dense 27B at the lowest price in the 27B line ($0.30/$2.40) — good value if you don't need the multimodal input that Qwen3.6-27B and Qwen3.8-27B add."},{"slug":"alibaba-qwen-qwen3-coder-next","name":"Qwen3-Coder-Next","provider":"Alibaba Qwen","api_id":"qwen3-coder-next","status":"ga","context_window":262144,"max_output_tokens":65536,"input_price":0.3,"output_price":1.5,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","batch","streaming","structured-output","fine-tuning","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-11","best_for":"Best coding value in the catalogue — 5x cheaper than qwen3-coder-plus at the base tier and open-weight. Tiers: 0-32K $0.30/$1.50, 32-128K $0.50/$2.50, 128-256K $0.80/$4."},{"slug":"alibaba-qwen-qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder-480B-A35B-Instruct","provider":"Alibaba Qwen","api_id":"qwen3-coder-480b-a35b-instruct","status":"ga","context_window":200000,"max_output_tokens":65536,"input_price":1.5,"output_price":7.5,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","batch","streaming","structured-output","fine-tuning","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-07","best_for":"The heavyweight open coding model for hard multi-file agentic tasks; expensive and tier-sensitive (0-32K $1.50/$7.50, 32-128K $2.70/$13.50, 128-200K $4.50/$22.50) — benchmark it against qwen3-coder-next before committing."},{"slug":"alibaba-qwen-qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder-30B-A3B-Instruct","provider":"Alibaba Qwen","api_id":"qwen3-coder-30b-a3b-instruct","status":"ga","context_window":200000,"max_output_tokens":65536,"input_price":0.45,"output_price":2.25,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","batch","streaming","structured-output","fine-tuning","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-07","best_for":"Laptop-to-workstation coding model with 3B active params; the practical choice for local IDE autocomplete and small refactors. Tiers: 0-32K $0.45/$2.25, 32-128K $0.75/$3.75, 128-200K $1.20/$6."},{"slug":"alibaba-qwen-qwen3-next-80b-a3b-thinking","name":"Qwen3-Next-80B-A3B-Thinking","provider":"Alibaba Qwen","api_id":"qwen3-next-80b-a3b-thinking","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.15,"output_price":1.2,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-09","best_for":"Very cheap open reasoning at $0.15 input — hybrid-attention architecture with 3B active params makes long-context inference unusually fast per dollar."},{"slug":"alibaba-qwen-qwen3-next-80b-a3b-instruct","name":"Qwen3-Next-80B-A3B-Instruct","provider":"Alibaba Qwen","api_id":"qwen3-next-80b-a3b-instruct","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.15,"output_price":1.2,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-09","best_for":"Non-thinking sibling for latency-sensitive work where you don't want reasoning tokens billed — same price as the thinking variant, so choose on behaviour not cost."},{"slug":"alibaba-qwen-qwen3-235b-a22b-thinking-2507","name":"Qwen3-235B-A22B-Thinking-2507","provider":"Alibaba Qwen","api_id":"qwen3-235b-a22b-thinking-2507","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.23,"output_price":2.3,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-07","best_for":"The 2025 open reasoning workhorse, still one of the cheapest large thinking models at $0.23 input; Qwen3.5-397B beats it but needs far more memory to self-host."},{"slug":"alibaba-qwen-qwen3-235b-a22b-instruct-2507","name":"Qwen3-235B-A22B-Instruct-2507","provider":"Alibaba Qwen","api_id":"qwen3-235b-a22b-instruct-2507","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.23,"output_price":0.92,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-07","best_for":"Excellent bulk-throughput value at $0.23/$0.92 — 2.5x cheaper output than its thinking twin, ideal for summarisation and extraction at scale."},{"slug":"alibaba-qwen-qwen3-30b-a3b-thinking-2507","name":"Qwen3-30B-A3B-Thinking-2507","provider":"Alibaba Qwen","api_id":"qwen3-30b-a3b-thinking-2507","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.2,"output_price":2.4,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-07","best_for":"Small MoE reasoner that self-hosts on a single 48GB card in FP8; output pricing is high relative to size, so prefer local deployment over the API for volume."},{"slug":"alibaba-qwen-qwen3-30b-a3b-instruct-2507","name":"Qwen3-30B-A3B-Instruct-2507","provider":"Alibaba Qwen","api_id":"qwen3-30b-a3b-instruct-2507","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.2,"output_price":0.8,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-07","best_for":"Fast, cheap non-thinking small MoE — a solid drop-in for classification, routing and tool-dispatch layers."},{"slug":"alibaba-qwen-qwen3-235b-a22b","name":"Qwen3-235B-A22B","provider":"Alibaba Qwen","api_id":"qwen3-235b-a22b","status":"ga","context_window":131072,"max_output_tokens":16384,"input_price":0.7,"output_price":2.8,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2025-03","license":"Apache-2.0","released":"2025-04","best_for":"Original hybrid-thinking Qwen3 flagship; output is $2.80 non-thinking / $8.40 thinking. The -2507 refresh is better and 3x cheaper — migrate."},{"slug":"alibaba-qwen-qwen3-32b","name":"Qwen3-32B","provider":"Alibaba Qwen","api_id":"qwen3-32b","status":"ga","context_window":131072,"max_output_tokens":16384,"input_price":0.16,"output_price":0.64,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2025-03","license":"Apache-2.0","released":"2025-04","best_for":"The most widely deployed open Qwen3 dense model and a well-supported fine-tuning base; output $0.64 non-thinking. Newer 27B models beat it, but tooling maturity still favours it."},{"slug":"alibaba-qwen-qwen3-30b-a3b","name":"Qwen3-30B-A3B","provider":"Alibaba Qwen","api_id":"qwen3-30b-a3b","status":"ga","context_window":131072,"max_output_tokens":16384,"input_price":0.2,"output_price":0.8,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2025-03","license":"Apache-2.0","released":"2025-04","best_for":"Original hybrid-mode 30B MoE; output $0.80 non-thinking / $2.40 thinking. Superseded by the -2507 split variants at the same price."},{"slug":"alibaba-qwen-qwen3-14b","name":"Qwen3-14B","provider":"Alibaba Qwen","api_id":"qwen3-14b","status":"ga","context_window":131072,"max_output_tokens":8192,"input_price":0.35,"output_price":1.4,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2025-03","license":"Apache-2.0","released":"2025-04","best_for":"Oddly priced above Qwen3-32B on the API ($0.35 vs $0.16) — only worth it as a self-hosted fine-tune target on a single 24-32GB GPU. Output $1.40 non-thinking / $4.20 thinking."},{"slug":"alibaba-qwen-qwen3-8b","name":"Qwen3-8B","provider":"Alibaba Qwen","api_id":"qwen3-8b","status":"ga","context_window":131072,"max_output_tokens":8192,"input_price":0.18,"output_price":0.7,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"2025-03","license":"Apache-2.0","released":"2025-04","best_for":"Edge/consumer-GPU tier — runs quantised on 8-12GB VRAM. Output $0.70 non-thinking / $2.10 thinking. Smaller 4B/1.7B/0.6B siblings exist on Hugging Face but are not separately priced in Model Studio."},{"slug":"alibaba-qwen-qwen3-vl-235b-a22b-thinking","name":"Qwen3-VL-235B-A22B-Thinking","provider":"Alibaba Qwen","api_id":"qwen3-vl-235b-a22b-thinking","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.4,"output_price":4,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-09","best_for":"Largest open vision-language reasoner — use for chart/diagram reasoning and GUI-agent work where a small VL model fails; output is 2.5x the instruct variant."},{"slug":"alibaba-qwen-qwen3-vl-235b-a22b-instruct","name":"Qwen3-VL-235B-A22B-Instruct","provider":"Alibaba Qwen","api_id":"qwen3-vl-235b-a22b-instruct","status":"ga","context_window":262144,"max_output_tokens":32768,"input_price":0.4,"output_price":1.6,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-09","best_for":"Top open VL model for straightforward description/extraction at $0.40/$1.60 — no reasoning-token overhead."},{"slug":"alibaba-qwen-qwen3-vl-32b-thinking","name":"Qwen3-VL-32B-Thinking","provider":"Alibaba Qwen","api_id":"qwen3-vl-32b-thinking","status":"ga","context_window":262144,"max_output_tokens":16384,"input_price":0.16,"output_price":0.64,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-10","best_for":"Cheapest open VL reasoner on the API ($0.16/$0.64) and dense enough to self-host on two 40GB cards — strong default for visual QA pipelines."},{"slug":"alibaba-qwen-qwen3-vl-32b-instruct","name":"Qwen3-VL-32B-Instruct","provider":"Alibaba Qwen","api_id":"qwen3-vl-32b-instruct","status":"ga","context_window":262144,"max_output_tokens":16384,"input_price":0.16,"output_price":0.64,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-10","best_for":"Same price as the thinking variant with lower latency — pick this for OCR-adjacent and captioning work that needs no deliberation."},{"slug":"alibaba-qwen-qwen3-vl-30b-a3b-thinking","name":"Qwen3-VL-30B-A3B-Thinking","provider":"Alibaba Qwen","api_id":"qwen3-vl-30b-a3b-thinking","status":"ga","context_window":262144,"max_output_tokens":16384,"input_price":0.2,"output_price":2.4,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-10","best_for":"Fast MoE visual reasoner (3B active) — good local throughput, but on the API the 32B-thinking model is cheaper on output ($0.64 vs $2.40)."},{"slug":"alibaba-qwen-qwen3-vl-30b-a3b-instruct","name":"Qwen3-VL-30B-A3B-Instruct","provider":"Alibaba Qwen","api_id":"qwen3-vl-30b-a3b-instruct","status":"ga","context_window":262144,"max_output_tokens":16384,"input_price":0.2,"output_price":0.8,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-10","best_for":"Best latency-per-dollar open VL model for high-volume image pipelines you host yourself."},{"slug":"alibaba-qwen-qwen3-vl-8b-thinking","name":"Qwen3-VL-8B-Thinking","provider":"Alibaba Qwen","api_id":"qwen3-vl-8b-thinking","status":"ga","context_window":262144,"max_output_tokens":8192,"input_price":0.18,"output_price":2.1,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","reasoning","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-10","best_for":"Smallest open VL reasoner — the one to fine-tune for a narrow visual domain on a single consumer GPU."},{"slug":"alibaba-qwen-qwen3-vl-8b-instruct","name":"Qwen3-VL-8B-Instruct","provider":"Alibaba Qwen","api_id":"qwen3-vl-8b-instruct","status":"ga","context_window":262144,"max_output_tokens":8192,"input_price":0.18,"output_price":0.7,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["tools","vision","batch","streaming","structured-output","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-10","best_for":"Edge-deployable multimodal model for on-device captioning and screenshot understanding."},{"slug":"alibaba-qwen-qwen2-5-omni-7b","name":"Qwen2.5-Omni-7B","provider":"Alibaba Qwen","api_id":"qwen2.5-omni-7b","status":"ga","context_window":32768,"max_output_tokens":2048,"input_price":0.1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","audio","video"],"capabilities":["vision","streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-03","best_for":"The small open any-to-any model people actually self-host for offline voice assistants; output pricing varies by modality and is not stated as a single figure on the pricing page. Superseded by the Qwen3.5-Omni hosted tier."},{"slug":"mistral-mistral-medium-3-5","name":"Mistral Medium 3.5","provider":"Mistral AI","api_id":"mistral-medium-3-5","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":1.5,"output_price":7.5,"cached_input_price":0.15,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":true,"knowledge_cutoff":"","license":"Modified MIT","released":"2026-04-28","best_for":"Mistral's current flagship for long-horizon agentic work and agentic coding — but note it costs 3x the input and 5x the output of Mistral Large 3, so only reach for it when tool-calling depth or coding quality actually justifies the premium. Aliases: mistral-medium-3, mistral-medium-latest."},{"slug":"mistral-mistral-large-3","name":"Mistral Large 3","provider":"Mistral AI","api_id":"mistral-large-2512","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":0.5,"output_price":1.5,"cached_input_price":0.05,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-12-02","best_for":"The value pick of the whole lineup: frontier-size open-weight MoE at $0.50/$1.50, cheaper than Medium 3.5 and most rivals' mid-tier models — default choice for general multimodal chat, RAG and bulk generation unless you specifically need Medium 3.5's agentic coding. Alias: mistral-large-latest."},{"slug":"mistral-mistral-small-4","name":"Mistral Small 4","provider":"Mistral AI","api_id":"mistral-small-2603","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":0.15,"output_price":0.6,"cached_input_price":0.015,"modalities_in":["text","image","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-03-16","best_for":"Best price/performance workhorse — a hybrid instruct+reasoning+coding model with only 6.5B active params, so it is fast and cheap while still handling agents and vision; the model to self-host under Apache 2.0 if you want reasoning without a licence conversation. Alias: mistral-small-latest."},{"slug":"mistral-ministral-3-14b","name":"Ministral 3 14B","provider":"Mistral AI","api_id":"ministral-14b-2512","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":0.2,"output_price":0.2,"cached_input_price":0.02,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-12-02","best_for":"Largest edge-class Ministral — flat $0.20 in and out makes it the cheapest option when your workload is output-heavy; separate Base, Instruct and Reasoning weights ship on Hugging Face for local/on-device deployment. Alias: ministral-14b-latest."},{"slug":"mistral-ministral-3-8b","name":"Ministral 3 8B","provider":"Mistral AI","api_id":"ministral-8b-2512","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":0.15,"output_price":0.15,"cached_input_price":0.015,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-12-02","best_for":"The sweet spot for high-volume classification, extraction and routing where you still want vision and 256K context; also the recommended local model for consumer GPUs. Alias: ministral-8b-latest."},{"slug":"mistral-ministral-3-3b","name":"Ministral 3 3B","provider":"Mistral AI","api_id":"ministral-3b-2512","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":0.1,"output_price":0.1,"cached_input_price":0.01,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-12-02","best_for":"Cheapest model Mistral serves and the one to run on-device — pick it for latency-critical edge inference or trivially structured tasks, not for anything needing world knowledge. Alias: ministral-3b-latest."},{"slug":"mistral-z-ai-glm-5-2","name":"Z.ai GLM 5.2","provider":"Mistral AI","api_id":"zai-glm-5-2","status":"preview","context_window":1000000,"max_output_tokens":128000,"input_price":1.4,"output_price":4.4,"cached_input_price":0.14,"modalities_in":["text"],"capabilities":["tools","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"Open (third-party, Z.ai — licence set by Z.ai, not Mistral)","released":"2026-08-06","best_for":"Use when you need a 1M-token window on Mistral's EU-hosted infrastructure — it is a third-party Z.ai model served without Mistral modifications, aimed at long-context coding and agentic workflows. Public preview."},{"slug":"mistral-codestral","name":"Codestral","provider":"Mistral AI","api_id":"codestral-2508","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0.3,"output_price":0.9,"cached_input_price":0.03,"modalities_in":["text"],"capabilities":["tools","prompt-caching","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary (Mistral Premier)","released":"2025-07-30","best_for":"The only Mistral model with a fill-in-the-middle endpoint (/v1/fim/completions) — the right pick for IDE autocomplete and other low-latency, high-frequency code completion; use Small 4 or Medium 3.5 for conversational coding instead. Alias: codestral-latest."},{"slug":"mistral-leanstral-1-5","name":"Leanstral 1.5","provider":"Mistral AI","api_id":"labs-leanstral-1-5","status":"preview","context_window":256000,"max_output_tokens":128000,"input_price":0,"output_price":0,"cached_input_price":0,"modalities_in":["text","image"],"capabilities":["tools","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-06-30","best_for":"Free Labs endpoint for Lean 4 formal proof engineering, autoformalization and automated theorem proving — genuinely $0 while Mistral gathers feedback, but it is a research preview with no availability guarantee, so don't build production on it. (The older labs-leanstral-2603 still shown on the marketing pricing page was retired 30 Jun 2026.)"},{"slug":"mistral-ocr-4-1","name":"OCR 4.1","provider":"Mistral AI","api_id":"mistral-ocr-4-1","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["image","pdf"],"capabilities":["batch","structured-output","prompt-caching"],"flagship":false,"knowledge_cutoff":"","license":"proprietary (Mistral Premier)","released":"2026-07-16","best_for":"Current best document-extraction model: priced per page, not per token — $4/1,000 pages for OCR and $5/1,000 pages for Document AI (cached $0.40/1,000 pages). Choose it over 4.0 for paragraph-level bounding boxes, structural block labels and block-level confidence scores. Aliases: mistral-ocr-4, mistral-ocr-latest."},{"slug":"mistral-ocr-4-0","name":"OCR 4.0","provider":"Mistral AI","api_id":"mistral-ocr-4-0","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["image","pdf"],"capabilities":["batch","structured-output","prompt-caching"],"flagship":false,"knowledge_cutoff":"","license":"proprietary (Mistral Premier)","released":"2026-06-23","best_for":"Page-priced at the same $4/1,000 pages OCR and $5/1,000 pages Document AI as 4.1 but without block-level confidence scores — no reason to pick it for new work; it exists for pinned integrations."},{"slug":"mistral-ocr-3","name":"OCR 3","provider":"Mistral AI","api_id":"mistral-ocr-2512","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["image","pdf"],"capabilities":["batch","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"proprietary (Mistral Premier)","released":"2025-12-18","best_for":"Cheaper legacy OCR at $2/1,000 pages (Document AI $3/1,000 pages) with no bounding-box extraction — worth keeping only for cost-sensitive plain text-and-image extraction or existing production pipelines; Mistral explicitly steers new work to OCR 4.x."},{"slug":"mistral-voxtral-small","name":"Voxtral Small","provider":"Mistral AI","api_id":"voxtral-small-2507","status":"ga","context_window":32000,"max_output_tokens":0,"input_price":0.1,"output_price":0.4,"cached_input_price":-1,"modalities_in":["text","audio"],"capabilities":["tools","batch","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2025-07-15","best_for":"The one Voxtral that does audio *understanding* rather than plain transcription — chat and Q&A over speech via /v1/chat/completions. Text tokens are $0.10/$0.40 per M; audio input is billed separately at $0.004 per audio minute. Alias: voxtral-small-latest."},{"slug":"mistral-voxtral-mini-transcribe-2","name":"Voxtral Mini Transcribe 2","provider":"Mistral AI","api_id":"voxtral-mini-2602","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary (Mistral Premier)","released":"2026-02-04","best_for":"Cheapest batch speech-to-text on the platform at $0.003 per audio minute (cached $0.0003/min) with word timestamps — the default for bulk transcription of recorded files via /v1/audio/transcriptions. Alias: voxtral-mini-latest."},{"slug":"mistral-voxtral-mini-transcribe-realtime","name":"Voxtral Mini Transcribe Realtime","provider":"Mistral AI","api_id":"voxtral-mini-transcribe-realtime-2602","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-02-04","best_for":"Live/streaming transcription at $0.006 per audio minute — double the batch model's rate, so only use it when you actually need low-latency partial results; Apache 2.0 weights (Voxtral-Mini-4B-Realtime-2602) make self-hosting viable. Alias: voxtral-mini-transcribe-realtime-latest."},{"slug":"mistral-voxtral-tts","name":"Voxtral TTS","provider":"Mistral AI","api_id":"voxtral-mini-tts-2603","status":"ga","context_window":0,"max_output_tokens":0,"input_price":0,"output_price":-1,"cached_input_price":0,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2026-03-23","best_for":"Text-to-speech with zero-shot voice cloning (no transcript needed for the voice prompt), 9 languages and ~90ms time-to-first-audio. Billed per character, not per token: input is free, output is $16 per million characters ($0.016 per 1,000 characters). Weights are CC BY-NC 4.0, so commercial self-hosting is not permitted — use the API. Alias: voxtral-mini-tts-latest."},{"slug":"mistral-codestral-embed","name":"Codestral Embed","provider":"Mistral AI","api_id":"codestral-embed-2505","status":"ga","context_window":8000,"max_output_tokens":0,"input_price":0.15,"output_price":-1,"cached_input_price":0.015,"modalities_in":["text"],"capabilities":["batch","prompt-caching"],"flagship":false,"knowledge_cutoff":"","license":"proprietary (Mistral Premier)","released":"2025-05-28","best_for":"Code-specialised embeddings for repo search and code RAG — worth the 50% premium over Mistral Embed only when you are indexing source code rather than prose. Output price is N/A (embeddings return vectors). Alias: codestral-embed."},{"slug":"mistral-mistral-embed","name":"Mistral Embed","provider":"Mistral AI","api_id":"mistral-embed-2312","status":"ga","context_window":8000,"max_output_tokens":0,"input_price":0.1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary (Mistral Premier)","released":"2023-12-11","best_for":"General-purpose text embeddings at $0.10/M input — the default for prose RAG, though it is Mistral's oldest still-served model (Dec 2023) and its 8K window is small by 2026 standards. Alias: mistral-embed."},{"slug":"mistral-mistral-moderation-2","name":"Mistral Moderation 2","provider":"Mistral AI","api_id":"mistral-moderation-2603","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0,"output_price":0,"cached_input_price":0,"modalities_in":["text"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary (Mistral Premier)","released":"2026-03-01","best_for":"Free content-moderation classifier on /v1/moderations with 128K context and jailbreak detection — strong on long multi-turn multilingual conversations, and free means there is no reason not to put it in front of a user-facing app."},{"slug":"mistral-shieldstral-1-0","name":"Shieldstral 1.0","provider":"Mistral AI","api_id":"","status":"preview","context_window":32000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-08-04","best_for":"Compact self-hosted safety classifier: you write policy questions in natural language and it returns yes/no for prompt moderation, response moderation, prompt-response pair classification and refusal detection, over text and images. Weights-only — it is listed in the model catalogue but carries no API model id and appears on neither pricing page, so run it yourself."},{"slug":"amazon-amazon-nova-2-lite","name":"Amazon Nova 2 Lite","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-2-lite-v1:0","status":"ga","context_window":1000000,"max_output_tokens":65536,"input_price":0.3,"output_price":2.5,"cached_input_price":0.075,"modalities_in":["text","image","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","fine-tuning","web-search","code-execution"],"flagship":true,"knowledge_cutoff":"2025-10","license":"proprietary","released":"2025-12-02","best_for":"The default Amazon model to reach for: 1M context, adjustable extended thinking, built-in web grounding and code interpreter at small-model prices; call it through the global.* profile to pay $0.30/$2.50 instead of the $0.33/$2.75 geo rate."},{"slug":"amazon-amazon-nova-2-pro-preview","name":"Amazon Nova 2 Pro (Preview)","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"","status":"preview","context_window":1000000,"max_output_tokens":0,"input_price":1.25,"output_price":10,"cached_input_price":0.3125,"modalities_in":["text","image","video","audio","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-12","best_for":"Amazon's frontier reasoning model for long-horizon planning and hard agentic work - justify it only where Nova 2 Lite measurably fails, since it is ~4x the input cost; still gated preview, so do not plan production capacity on it. Cache figure derived from AWS's published 'cache reads are 75% less than on-demand input' rule."},{"slug":"amazon-amazon-nova-2-omni-preview","name":"Amazon Nova 2 Omni (Preview)","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"","status":"preview","context_window":1000000,"max_output_tokens":0,"input_price":0.3,"output_price":2.5,"cached_input_price":0.075,"modalities_in":["text","image","video","audio","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","web-search","code-execution"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-12","best_for":"One model instead of a pipeline when you need speech-in plus video-in plus image-out together; watch the asymmetric rates - audio input is $1.00/1M and image output is $40.00/1M against $0.30/$2.50 for text. Preview, early access via your AWS account team."},{"slug":"amazon-amazon-nova-premier","name":"Amazon Nova Premier","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-premier-v1:0","status":"deprecated","context_window":1000000,"max_output_tokens":25000,"input_price":2.5,"output_price":12.5,"cached_input_price":-1,"modalities_in":["text","image","video","pdf"],"capabilities":["tools","vision","reasoning","prompt-caching","batch","streaming","web-search"],"flagship":false,"knowledge_cutoff":"2024-10","license":"proprietary","released":"2025-10-31","best_for":"Do not start new work here - Bedrock moved it to Legacy on 2026-03-13 with EOL 2026-09-14, and Nova 2 Pro is both cheaper and stronger; its remaining role was as distillation teacher for Pro/Lite/Micro students."},{"slug":"amazon-amazon-nova-pro","name":"Amazon Nova Pro","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-pro-v1:0","status":"ga","context_window":300000,"max_output_tokens":5000,"input_price":0.8,"output_price":3.2,"cached_input_price":0.2,"modalities_in":["text","image","video","pdf"],"capabilities":["tools","vision","prompt-caching","batch","streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2024-10","license":"proprietary","released":"2024-12-05","best_for":"Previous-generation workhorse worth keeping only if you already fine-tuned it or need Bedrock Agents/Flows/Knowledge Bases, which Nova 2 Lite does not yet support; otherwise Nova 2 Lite is cheaper with 3x the context."},{"slug":"amazon-amazon-nova-lite","name":"Amazon Nova Lite","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-lite-v1:0","status":"ga","context_window":300000,"max_output_tokens":5000,"input_price":0.06,"output_price":0.24,"cached_input_price":0.015,"modalities_in":["text","image","video","pdf"],"capabilities":["tools","vision","prompt-caching","batch","streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2024-10","license":"proprietary","released":"2024-12-05","best_for":"Still the cheapest vision-capable model on Bedrock at $0.06/1M in - right for bulk image, video and document classification or extraction where a 5K output cap and no reasoning mode are acceptable."},{"slug":"amazon-amazon-nova-micro","name":"Amazon Nova Micro","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-micro-v1:0","status":"ga","context_window":128000,"max_output_tokens":5000,"input_price":0.035,"output_price":0.14,"cached_input_price":0.00875,"modalities_in":["text"],"capabilities":["tools","prompt-caching","batch","streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2024-10","license":"proprietary","released":"2024-12-05","best_for":"Cheapest and lowest-latency model in the Bedrock catalog - text-only routing, intent classification, tagging and short summarization at high QPS; no vision, no reasoning."},{"slug":"amazon-amazon-nova-2-sonic","name":"Amazon Nova 2 Sonic","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-2-sonic-v1:0","status":"ga","context_window":1000000,"max_output_tokens":65536,"input_price":3,"output_price":12,"cached_input_price":-1,"modalities_in":["audio","text"],"capabilities":["tools","streaming"],"flagship":true,"knowledge_cutoff":"","license":"proprietary","released":"2025-12-02","best_for":"Current speech-to-speech choice for real-time voice agents over the bidirectional streaming API; the $3.00/$12.00 rates are speech tokens - text tokens (transcription, tool calls, injected history) bill separately at $0.319 in / $2.651 out per 1M. In-region only: us-east-1, us-west-2, eu-north-1, ap-northeast-1."},{"slug":"amazon-amazon-nova-sonic","name":"Amazon Nova Sonic","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-sonic-v1:0","status":"deprecated","context_window":300000,"max_output_tokens":0,"input_price":3.4,"output_price":13.6,"cached_input_price":-1,"modalities_in":["audio","text"],"capabilities":["tools","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-03","best_for":"Migrate off it - Legacy since 2026-03-13 with EOL 2026-09-14, and Nova 2 Sonic is cheaper ($3.00/$12.00 speech) with more languages. Text tokens bill at $0.06 in / $0.24 out per 1M."},{"slug":"amazon-amazon-nova-multimodal-embeddings","name":"Amazon Nova Multimodal Embeddings","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-2-multimodal-embeddings-v1:0","status":"ga","context_window":8000,"max_output_tokens":0,"input_price":0.135,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image","video","audio","pdf"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-10-28","best_for":"The only Bedrock model that puts text, document images, video and audio in one shared embedding space - pick it for cross-modal retrieval and agentic RAG; selectable 3072/1024/384/256 dimensions trade recall against vector-store cost. Non-text bills per unit, not per token: $0.00006/standard image, $0.0006/document image, $0.0007/second of video, $0.00014/second of audio (all 50% off in batch). Max 8K tokens or 30s of video/audio per call."},{"slug":"amazon-amazon-nova-canvas","name":"Amazon Nova Canvas","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-canvas-v1:0","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-12-03","best_for":"Image generation and editing with built-in watermarking and moderation, but Legacy with EOL 2026-09-30 - do not build new on it. Priced per image, not per token: $0.04 standard / $0.06 premium up to 1024x1024, $0.06 / $0.08 up to 2048x2048; fine-tuning is $0.005 per image seen. Max prompt 1024 characters."},{"slug":"amazon-amazon-nova-reel","name":"Amazon Nova Reel","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.nova-reel-v1:1","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-12-03","best_for":"Short text- or image-to-video with camera-motion controls via the async invoke API, but both v1:0 and v1:1 are Legacy with EOL 2026-09-30. Priced at $0.08 per second of generated 720p 24fps video; max prompt 512 characters."},{"slug":"amazon-amazon-titan-text-embeddings-v2","name":"Amazon Titan Text Embeddings V2","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.titan-embed-text-v2:0","status":"ga","context_window":8192,"max_output_tokens":0,"input_price":0.02,"output_price":0,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-04-30","best_for":"The cheap default for English-first text RAG on Bedrock at $0.02/1M ($0.01 batch), with 8,192-token inputs and selectable 1024/512/256 dimensions. Use Nova Multimodal Embeddings if you need anything beyond plain text."},{"slug":"amazon-amazon-titan-embeddings-g1-text","name":"Amazon Titan Embeddings G1 - Text","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.titan-embed-text-v1","status":"ga","context_window":8192,"max_output_tokens":0,"input_price":0.1,"output_price":0,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2023-09-28","best_for":"First-generation text embeddings, still marked Active but 5x the price of Titan Text Embeddings V2 for no gain - keep it only to avoid re-indexing an existing vector store built on it."},{"slug":"amazon-amazon-titan-embeddings-g1-text-v2","name":"Amazon Titan Embeddings G1 - Text v2","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.titan-embed-g1-text-02","status":"ga","context_window":8192,"max_output_tokens":0,"input_price":0.1,"output_price":0,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"A point release of the G1 text embedder (distinct from Titan Text Embeddings V2) that still has a live Bedrock model card; the pricing page bills it under the single 'Amazon Titan Text Embeddings' G1 line at $0.10/1M, so confirm on your bill before relying on that figure."},{"slug":"amazon-amazon-titan-multimodal-embeddings-g1","name":"Amazon Titan Multimodal Embeddings G1","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.titan-embed-image-v1","status":"ga","context_window":256,"max_output_tokens":0,"input_price":0.8,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["batch","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2023-11-29","best_for":"Text-plus-image search in a shared space and the only Amazon embedder you can fine-tune on your own image-caption pairs; note the tiny 256-token text limit and that images bill separately at $0.00006 each ($0.00003 batch). Nova Multimodal Embeddings supersedes it for new work."},{"slug":"amazon-amazon-titan-image-generator-g1-v2","name":"Amazon Titan Image Generator G1 v2","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.titan-image-generator-v2:0","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2023-11-29","best_for":"Only for maintaining an existing us-east-1/us-west-2 integration - Bedrock lists it as Legacy with an EOL date of 2026-06-30 that has already passed. Priced per image: $0.008 standard / $0.010 premium at 512x512 or smaller, $0.010 / $0.012 above that."},{"slug":"amazon-amazon-rerank-1-0","name":"Amazon Rerank 1.0","provider":"Amazon (Amazon Nova & Amazon Titan on Amazon Bedrock)","api_id":"amazon.rerank-v1:0","status":"ga","context_window":512,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-12","best_for":"Second-stage reranker for Bedrock Knowledge Bases retrieval - billed per query rather than per token at $1.00 per 1,000 queries, where one query covers up to 100 document chunks of at most 512 tokens each (350 docs bills as 4 queries). Not available in us-east-1; use ap-northeast-1, ca-central-1, eu-central-1 or us-west-2."},{"slug":"microsoft-mai-thinking-1","name":"MAI-Thinking-1","provider":"Microsoft","api_id":"MAI-Thinking-1","status":"preview","context_window":256000,"max_output_tokens":64000,"input_price":2,"output_price":8,"cached_input_price":0.2,"modalities_in":["text"],"capabilities":["tools","reasoning","prompt-caching","streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06-01","best_for":"Microsoft's own frontier reasoning model — pick it over GPT-5.x on Azure when you want a 256K-context, tool-calling reasoner at roughly a quarter of flagship pricing, and you can live with encrypted (non-inspectable) chain-of-thought and a preview SLA."},{"slug":"microsoft-mai-cyber-1-flash","name":"MAI-Cyber-1-Flash","provider":"Microsoft","api_id":"MAI-Cyber-1-Flash","status":"preview","context_window":0,"max_output_tokens":0,"input_price":0.6,"output_price":3.5,"cached_input_price":0.06,"modalities_in":["text"],"capabilities":["prompt-caching"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"A security-domain MAI model that has live Global Standard billing meters on Azure but no public model card or Foundry catalog entry yet — treat pricing as firm and availability as unannounced; context window and capabilities are unpublished."},{"slug":"microsoft-mai-image-2-6","name":"MAI-Image-2.6","provider":"Microsoft","api_id":"MAI-Image-2.6","status":"preview","context_window":32000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["web-search"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-07-31","best_for":"Newest MAI image model — best text rendering, portraits and commercial/photoreal output in the family, and the only tier with auto_aspect_ratio and Bing web_grounding; no published price yet, so budget against MAI-Image-2.5 rates."},{"slug":"microsoft-mai-image-2-6-flash","name":"MAI-Image-2.6-Flash","provider":"Microsoft","api_id":"MAI-Image-2.6-Flash","status":"preview","context_window":32000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["web-search"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-07-31","best_for":"Same generation and editing quality as MAI-Image-2.6 at lower latency and cost — the volume choice for 2.6-era output, but Azure has not published its meters yet."},{"slug":"microsoft-mai-image-2-5-pro","name":"MAI-Image-2.5-Pro","provider":"Microsoft","api_id":"MAI-Image-2.5-Pro","status":"preview","context_window":32000,"max_output_tokens":0,"input_price":5,"output_price":106,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06-19","best_for":"The quality ceiling of the 2.5 series — buy it only for visually dense scenes needing object/character consistency, material accuracy and spatial reasoning, because image output costs $106/1M tokens vs $47 on plain 2.5 (image input is $8.00/1M). Retires 2026-10-01."},{"slug":"microsoft-mai-image-2-5","name":"MAI-Image-2.5","provider":"Microsoft","api_id":"MAI-Image-2.5","status":"preview","context_window":32000,"max_output_tokens":0,"input_price":5,"output_price":47,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06-02","best_for":"The balanced default for text-to-image and surgical image editing on Azure: $5.00/1M text input, $8.00/1M image input, $47.00/1M image output. Retires 2026-10-01, so start new work on the 2.6 series."},{"slug":"microsoft-mai-image-2-5-flash","name":"MAI-Image-2.5-Flash","provider":"Microsoft","api_id":"MAI-Image-2.5-Flash","status":"preview","context_window":32000,"max_output_tokens":0,"input_price":1.75,"output_price":19.5,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06-02","best_for":"Cheapest MAI image tier with published pricing — $1.75/1M for both text and image input and $19.50/1M image output, roughly 2.4x cheaper output than MAI-Image-2.5 for high-volume concept art and thumbnails. Retires 2026-10-01."},{"slug":"microsoft-mai-voice-2","name":"MAI-Voice-2","provider":"Microsoft","api_id":"MAI-Voice-2","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06-02","best_for":"Highest-fidelity MAI TTS for audiobooks, podcasts and long-form narration — ~46 prebuilt voices across 15 languages/18 locales with SSML style+styledegree control and gated instant voice cloning. Not token-billed: it runs on the Azure Speech Neural HD Text to Speech meter at $22.00 per 1M characters (Personal Voice cloning is $24.00 per 1M characters)."},{"slug":"microsoft-mai-voice-2-flash","name":"MAI-Voice-2-Flash","provider":"Microsoft","api_id":"MAI-Voice-2-Flash","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-07-22","best_for":"The real-time voice-agent / IVR pick in the MAI-Voice family — same expressive multilingual voices as MAI-Voice-2 but tuned for very low latency, and usable inside the Voice Live API. Billed per character on Azure Speech, not per token."},{"slug":"microsoft-mai-voice-1","name":"MAI-Voice-1","provider":"Microsoft","api_id":"MAI-Voice-1","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2025-12-18","best_for":"First-generation MAI text-to-speech, still selectable in the Foundry catalog and in Voice Live personal-voice enums but no longer documented on the MAI-Voice page — migrate to MAI-Voice-2 or MAI-Voice-2-Flash."},{"slug":"microsoft-mai-transcribe-2","name":"MAI-Transcribe-2","provider":"Microsoft","api_id":"MAI-Transcribe-2","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-09-03","best_for":"Microsoft's current in-house ASR: 60 languages with auto language ID, speaker diarization, word-level timestamps, keyword biasing, code-switching (Hinglish/Spanglish) and verbatim-vs-clean transcript styles. Billed by audio hour on Azure Speech (Fast Transcription $0.36/hr base, plus a $0.30/hr Speech-to-Text Enhanced Feature Audio meter), not per token; call it via enhancedMode.model on the Fast Transcription API."},{"slug":"microsoft-mai-transcribe-1-5","name":"MAI-Transcribe-1.5","provider":"Microsoft","api_id":"MAI-Transcribe-1.5","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-06-02","best_for":"Previous-generation MAI ASR covering ~44 languages; only worth pinning if you have validated output against it, otherwise go straight to MAI-Transcribe-2. Billed per audio hour on Azure Speech."},{"slug":"microsoft-mai-transcribe-1","name":"MAI-Transcribe-1","provider":"Microsoft","api_id":"MAI-Transcribe-1","status":"deprecated","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2026-01-23","best_for":"Deprecated on 2026-08-20 with a retirement date of 2026-09-15 and MAI-Transcribe-1.5 named as replacement — do not start new work here."},{"slug":"microsoft-model-router","name":"model-router","provider":"Microsoft","api_id":"model-router","status":"ga","context_window":200000,"max_output_tokens":128000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","streaming"],"flagship":true,"knowledge_cutoff":"","license":"proprietary","released":"2025-11-18","best_for":"A Microsoft-built classifier that picks the cheapest adequate underlying model (GPT-4.1 series, o4-mini, GPT-5 reasoning models, gpt-5-chat) per request; there is no router surcharge — you pay whatever the chosen model costs, so it is a cost lever rather than a priced model. Version 2025-11-18 is GA until 2027-05-20; requests over ~200K context only succeed if routed to a large-context model."},{"slug":"microsoft-mai-ds-r1","name":"MAI-DS-R1","provider":"Microsoft","api_id":"MAI-DS-R1","status":"deprecated","context_window":163840,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"2025-04","best_for":"Microsoft's safety- and unblocking-post-trained variant of DeepSeek-R1 (671B MoE, 37B active, MIT). It has dropped out of the current Foundry catalog alongside the DeepSeek-R1 retirement, though legacy Azure meters still list $1.35/1M in and $5.40/1M out on Global Standard; the weights remain freely downloadable, so self-host rather than plan on the hosted endpoint."},{"slug":"microsoft-phi-4","name":"Phi-4","provider":"Microsoft","api_id":"Phi-4","status":"ga","context_window":16384,"max_output_tokens":16384,"input_price":0.125,"output_price":0.5,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":true,"knowledge_cutoff":"2024-06","license":"MIT","released":"2024-12-12","best_for":"The dense 14B workhorse of the family: strongest general Phi quality per dollar for math, code and reasoning when a 16K context is enough — if you need long context, use Phi-4-mini-instruct instead."},{"slug":"microsoft-phi-4-mini-instruct","name":"Phi-4-mini-instruct","provider":"Microsoft","api_id":"Phi-4-mini-instruct","status":"ga","context_window":131072,"max_output_tokens":4096,"input_price":0.075,"output_price":0.3,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2024-06","license":"MIT","released":"2025-02","best_for":"Cheapest hosted Phi and the default for high-volume classification, extraction and routing across 23 languages with a real 128K window; no native tool calling, so wrap it yourself."},{"slug":"microsoft-phi-4-multimodal-instruct","name":"Phi-4-multimodal-instruct","provider":"Microsoft","api_id":"Phi-4-multimodal-instruct","status":"ga","context_window":131072,"max_output_tokens":4096,"input_price":0.08,"output_price":0.32,"cached_input_price":-1,"modalities_in":["text","image","audio"],"capabilities":["vision","streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2024-06","license":"MIT","released":"2025-02","best_for":"The only Phi that takes audio as well as images — good for cheap on-device-class speech understanding, OCR and chart reading in one 5.6B model. Watch the meter: audio input is billed separately at $4.00 per 1M audio tokens, 50x the text input rate."},{"slug":"microsoft-phi-4-reasoning","name":"Phi-4-reasoning","provider":"Microsoft","api_id":"Phi-4-reasoning","status":"ga","context_window":32768,"max_output_tokens":32768,"input_price":0.125,"output_price":0.5,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","streaming"],"flagship":false,"knowledge_cutoff":"2025-03","license":"MIT","released":"2025-04-30","best_for":"14B SFT-distilled reasoner at Phi-4 prices — the value pick for math and STEM chains of thought when you cannot justify MAI-Thinking-1 or a frontier model; English only."},{"slug":"microsoft-phi-4-reasoning-plus","name":"Phi-4-reasoning-plus","provider":"Microsoft","api_id":"Phi-4-reasoning-plus","status":"ga","context_window":32768,"max_output_tokens":32768,"input_price":0.125,"output_price":0.5,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","streaming"],"flagship":false,"knowledge_cutoff":"2025-03","license":"MIT","released":"2025-04-30","best_for":"RL-tuned sibling of Phi-4-reasoning at identical token prices — higher accuracy but noticeably longer chains of thought, so it costs more per answer in practice. Still priced and listed in the Foundry catalog, but dropped from the current 'Microsoft models' doc table, so treat its hosted lifetime as shorter than Phi-4-reasoning's."},{"slug":"microsoft-phi-4-mini-reasoning","name":"Phi-4-mini-reasoning","provider":"Microsoft","api_id":"Phi-4-mini-reasoning","status":"ga","context_window":128000,"max_output_tokens":128000,"input_price":0.075,"output_price":0.3,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","streaming"],"flagship":false,"knowledge_cutoff":"2025-02","license":"MIT","released":"2025-04","best_for":"The cheapest reasoning model Microsoft sells: 3.8B with a 128K window in and out, aimed at edge/embedded math tutoring and step-by-step solvers where output length matters more than breadth."},{"slug":"microsoft-phi-4-mini-flash-reasoning","name":"Phi-4-mini-flash-reasoning","provider":"Microsoft","api_id":"Phi-4-mini-flash-reasoning","status":"ga","context_window":65536,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning"],"flagship":false,"knowledge_cutoff":"2025-02","license":"MIT","released":"2025-06","best_for":"Hybrid SambaY/Gated-Memory-Unit architecture giving up to ~10x higher decoding throughput than Phi-4-mini-reasoning on long generations — self-host it (or use Foundry managed compute) when tokens/sec on a single GPU is the binding constraint; there is no per-token Azure meter for it."},{"slug":"microsoft-phi-4-reasoning-vision-15b","name":"Phi-4-Reasoning-Vision-15B","provider":"Microsoft","api_id":"Phi-4-Reasoning-Vision-15B","status":"ga","context_window":16384,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","reasoning"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"2026-03-04","best_for":"Newest Phi release — a 15B multimodal reasoner with explicit <think> traces for chart/diagram/document math and GUI element localization (ScreenSpot-V2). Managed-compute or self-host only; no serverless per-token price published."},{"slug":"microsoft-phi-mini-moe-instruct","name":"Phi-mini-MoE-instruct","provider":"Microsoft","api_id":"microsoft/Phi-mini-MoE-instruct","status":"ga","context_window":4096,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2025-06-23","best_for":"SlimMoE compression of Phi-3.5-MoE down to 7.6B total / 2.4B active — near-Phi-3.5-MoE quality at a third of the memory, but a hard 4K context kills it for RAG. Weights only, no hosted endpoint."},{"slug":"microsoft-phi-tiny-moe-instruct","name":"Phi-tiny-MoE-instruct","provider":"Microsoft","api_id":"microsoft/Phi-tiny-MoE-instruct","status":"ga","context_window":4096,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2025-06-23","best_for":"Smallest SlimMoE variant (1.1B active) for CPU and NPU inference where every GB counts; 4K context and an Oct-2023 cutoff make it a component model, not a chatbot. Weights only."},{"slug":"microsoft-phi-ground-any","name":"Phi-Ground-Any","provider":"Microsoft","api_id":"microsoft/Phi-Ground-Any","status":"preview","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"2026-05-07","best_for":"Research GUI-grounding model (Phi-3-V based) that maps a natural-language instruction to on-screen coordinates — a building block for computer-use agents, not a general chat model; specs are thinly documented and it has no hosted endpoint."},{"slug":"microsoft-phi-3-5-mini-instruct","name":"Phi-3.5-mini-instruct","provider":"Microsoft","api_id":"Phi-3.5-mini-instruct","status":"deprecated","context_window":131072,"max_output_tokens":4096,"input_price":0.13,"output_price":0.52,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2024-08-16","best_for":"Still deployable and still metered on Azure, but Microsoft has dropped it from the current Microsoft-models doc table and Phi-4-mini-instruct is both cheaper and better — migrate."},{"slug":"microsoft-phi-3-5-moe-instruct","name":"Phi-3.5-MoE-instruct","provider":"Microsoft","api_id":"Phi-3.5-MoE-instruct","status":"deprecated","context_window":131072,"max_output_tokens":4096,"input_price":0.16,"output_price":0.64,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2024-08-17","best_for":"The largest Phi ever shipped (16x3.8B MoE, 6.6B active) — historically interesting and still self-hostable under MIT, but superseded on quality-per-dollar by Phi-4 and removed from the current Foundry model list."},{"slug":"microsoft-phi-3-5-vision-instruct","name":"Phi-3.5-vision-instruct","provider":"Microsoft","api_id":"Phi-3.5-vision-instruct","status":"deprecated","context_window":131072,"max_output_tokens":4096,"input_price":0.13,"output_price":0.52,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2024-08-16","best_for":"Multi-image and video-frame reasoning at 4.2B; use Phi-4-multimodal-instruct instead unless you specifically need this checkpoint's multi-frame behaviour."},{"slug":"microsoft-phi-3-medium-128k-instruct","name":"Phi-3-medium-128k-instruct","provider":"Microsoft","api_id":"Phi-3-medium-128k-instruct","status":"deprecated","context_window":131072,"max_output_tokens":4096,"input_price":0.17,"output_price":0.68,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2024-05-07","best_for":"Legacy 14B long-context Phi-3; still metered and catalogued but strictly worse and pricier than Phi-4 — kept only for pinned reproducibility."},{"slug":"microsoft-phi-3-medium-4k-instruct","name":"Phi-3-medium-4k-instruct","provider":"Microsoft","api_id":"Phi-3-medium-4k-instruct","status":"deprecated","context_window":4096,"max_output_tokens":4096,"input_price":0.17,"output_price":0.68,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2024-05-07","best_for":"Short-context variant of Phi-3-medium; no reason to choose it today over Phi-4 at a lower price with a larger window."},{"slug":"microsoft-phi-3-small-128k-instruct","name":"Phi-3-small-128k-instruct","provider":"Microsoft","api_id":"Phi-3-small-128k-instruct","status":"deprecated","context_window":131072,"max_output_tokens":4096,"input_price":0.15,"output_price":0.6,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2024-05-07","best_for":"Legacy 7B long-context model; superseded on every axis by Phi-4-mini-instruct at half the price."},{"slug":"microsoft-phi-3-small-8k-instruct","name":"Phi-3-small-8k-instruct","provider":"Microsoft","api_id":"Phi-3-small-8k-instruct","status":"deprecated","context_window":8192,"max_output_tokens":4096,"input_price":0.15,"output_price":0.6,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2024-05-07","best_for":"Short-context legacy 7B; migrate to Phi-4-mini-instruct. (The Foundry catalog record misreports its window as 131072; the model card and name are authoritative at 8K.)"},{"slug":"microsoft-phi-3-mini-128k-instruct","name":"Phi-3-mini-128k-instruct","provider":"Microsoft","api_id":"Phi-3-mini-128k-instruct","status":"deprecated","context_window":131072,"max_output_tokens":4096,"input_price":0.13,"output_price":0.52,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2024-04-22","best_for":"The model that launched the SLM category; still one of the most downloaded Phi checkpoints for offline/edge use, but on Azure Phi-4-mini-instruct is cheaper and stronger."},{"slug":"microsoft-phi-3-mini-4k-instruct","name":"Phi-3-mini-4k-instruct","provider":"Microsoft","api_id":"Phi-3-mini-4k-instruct","status":"deprecated","context_window":4096,"max_output_tokens":4096,"input_price":0.13,"output_price":0.52,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","fine-tuning"],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2024-04-22","best_for":"The canonical tiny Phi for phones and NPUs (also shipped as GGUF and ONNX DirectML/CUDA/CPU builds); pick it only for offline deployment where the 4K window is fine."},{"slug":"microsoft-phi-3-vision-128k-instruct","name":"Phi-3-vision-128k-instruct","provider":"Microsoft","api_id":"Phi-3-vision-128k-instruct","status":"deprecated","context_window":131072,"max_output_tokens":4096,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision"],"flagship":false,"knowledge_cutoff":"2024-03","license":"MIT","released":"2024-05-19","best_for":"Original Phi vision model, managed-compute/self-host only (no serverless per-token meter); superseded by Phi-3.5-vision-instruct and then Phi-4-multimodal-instruct."},{"slug":"microsoft-phi-2","name":"phi-2","provider":"Microsoft","api_id":"microsoft/phi-2","status":"deprecated","context_window":2048,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"2023-10","license":"MIT","released":"2023-12-13","best_for":"Base (non-instruct) 2.7B research model, still the most-downloaded Microsoft checkpoint on Hugging Face and a common fine-tuning starting point — but a 2K context and no chat tuning make it unsuitable for production assistants."},{"slug":"cohere-command-a","name":"Command A+","provider":"Cohere","api_id":"command-a-plus-05-2026","status":"ga","context_window":128000,"max_output_tokens":64000,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["tools","vision","reasoning","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-05-20","best_for":"Cohere's flagship: the only model that folds vision, reasoning, agentic tool use and 48-language coverage into one set of weights, and it runs on 1x B200 or 2x H100 — but the API is free only under trial-grade limits (20 req/min, 1,000 calls/month), so real production means Apache-2.0 self-hosting or a sales contract."},{"slug":"cohere-command-a-2","name":"Command A","provider":"Cohere","api_id":"command-a-03-2025","status":"ga","context_window":256000,"max_output_tokens":8000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","streaming","structured-output"],"flagship":true,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2025-03","best_for":"The long-context (256K) dense workhorse for RAG, tool use and 23-language agents, and the only Command A variant with a real 500 req/min production rate limit — but Cohere pulled its list price off the public pricing page, so budget via sales."},{"slug":"cohere-command-a-reasoning","name":"Command A Reasoning","provider":"Cohere","api_id":"command-a-reasoning-08-2025","status":"ga","context_window":256000,"max_output_tokens":32000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","reasoning","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2025-08","best_for":"Pick when you need an explicit thinking budget on nuanced multi-step or agentic problems in 23 languages and can host it (4x H100 for production); no public price and production API access is contact-sales only."},{"slug":"cohere-command-a-translate","name":"Command A Translate","provider":"Cohere","api_id":"command-a-translate-08-2025","status":"ga","context_window":8000,"max_output_tokens":8000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2025-08","best_for":"A dedicated 23-language machine-translation model for regulated shops that must translate sensitive documents inside their own perimeter; the 8K input cap means you chunk long documents yourself."},{"slug":"cohere-command-a-vision","name":"Command A Vision","provider":"Cohere","api_id":"command-a-vision-07-2025","status":"ga","context_window":128000,"max_output_tokens":8000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2025-07","best_for":"Document/chart/OCR understanding with up to 20 images per request — but it does NOT support tool use, so for agentic multimodal work go to Command A+ instead."},{"slug":"cohere-command-r-08-2024","name":"Command R+ (08-2024)","provider":"Cohere","api_id":"command-r-plus-08-2024","status":"ga","context_window":128000,"max_output_tokens":4000,"input_price":2.5,"output_price":10,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","streaming","structured-output"],"flagship":true,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2024-08","best_for":"Still Live with a 500 req/min production limit, but at $2.50/$10.00 it is priced under the 'existing customers' FAQ and comprehensively beaten by Command A on quality, context and throughput — migrate rather than start here."},{"slug":"cohere-command-r-08-2024-2","name":"Command R (08-2024)","provider":"Cohere","api_id":"command-r-08-2024","status":"ga","context_window":128000,"max_output_tokens":4000,"input_price":0.15,"output_price":0.6,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2024-08","best_for":"The cheapest Cohere model with a published price, a real 500 req/min production limit and full tool use — the practical default for high-volume RAG and single-step tool calling when you want a price you can actually see."},{"slug":"cohere-command-r7b","name":"Command R7B","provider":"Cohere","api_id":"command-r7b-12-2024","status":"ga","context_window":128000,"max_output_tokens":4000,"input_price":0.0375,"output_price":0.15,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2024-12","best_for":"Cohere's cheapest served model by a wide margin ($0.0375/1M in) with a 128K window — right for latency-sensitive chatbots, classification and on-device/consumer-GPU deployment where 4K output is enough."},{"slug":"cohere-command-r-03-2024","name":"Command R (03-2024)","provider":"Cohere","api_id":"command-r-03-2024","status":"deprecated","context_window":128000,"max_output_tokens":4000,"input_price":0.5,"output_price":1.5,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2024-03","best_for":"Deprecated 2025-09-15 and 3.3x the price of command-r-08-2024 for worse results; the alias `command-r` also points here. No reason to choose it — migrate to command-r-08-2024."},{"slug":"cohere-command-r-04-2024","name":"Command R+ (04-2024)","provider":"Cohere","api_id":"command-r-plus-04-2024","status":"deprecated","context_window":128000,"max_output_tokens":4000,"input_price":3,"output_price":15,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","streaming","structured-output"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2024-04","best_for":"Deprecated 2025-09-15 and the most expensive model Cohere still serves; the alias `command-r-plus` resolves here. Move to command-r-plus-08-2024 for a 17%/33% price cut, or Command A."},{"slug":"cohere-command","name":"Command","provider":"Cohere","api_id":"command","status":"deprecated","context_window":4000,"max_output_tokens":4000,"input_price":1,"output_price":2,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"First-generation Command, deprecated 2025-09-15 with only a 4K context and no tool use; kept alive for existing integrations only."},{"slug":"cohere-command-light","name":"Command Light","provider":"Cohere","api_id":"command-light","status":"deprecated","context_window":4000,"max_output_tokens":4000,"input_price":0.3,"output_price":0.6,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"Deprecated 2025-09-15 and now strictly dominated by Command R7B, which is 8x cheaper on input with a 32x larger context window."},{"slug":"cohere-north-mini-code-1-0","name":"North Mini Code 1.0","provider":"Cohere","api_id":"","status":"ga","context_window":0,"max_output_tokens":0,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tools","streaming"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-06","best_for":"Cohere's first agentic coding model: Apache-2.0, free API key plus free weights, and a 3B active footprint that runs locally — aimed at repo-level SWE-agent/OpenCode style harnesses and terminal agents. Context window is not published; exact API model string is not documented (docs page: /docs/north-mini-code-1.0)."},{"slug":"cohere-embed-4","name":"Embed 4","provider":"Cohere","api_id":"embed-v4.0","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0.12,"output_price":0,"cached_input_price":-1,"modalities_in":["text","image","pdf"],"capabilities":["vision","batch"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"The default retrieval embedder: 128K context, native mixed text+image/PDF input with no preprocessing, Matryoshka output dims (256/512/1024/1536) and 100+ languages, at $0.12 per 1M text tokens ($0.47 per 1M image tokens). Dedicated capacity via Model Vault runs $4.00/hr (Small) or $5.00/hr (Medium)."},{"slug":"cohere-embed-english-v3-0","name":"Embed English v3.0","provider":"Cohere","api_id":"embed-english-v3.0","status":"ga","context_window":512,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","batch","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2023-11","best_for":"1024-dim English-only embedder with a 512-token limit — only worth keeping for existing indexes; Embed 4 supersedes it on every axis. No longer publicly priced."},{"slug":"cohere-embed-english-light-v3-0","name":"Embed English Light v3.0","provider":"Cohere","api_id":"embed-english-light-v3.0","status":"ga","context_window":512,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","batch","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2023-11","best_for":"384-dim English embedder for latency- or index-size-constrained systems; use Embed 4 at 256 dims instead unless you are locked in. No longer publicly priced."},{"slug":"cohere-embed-multilingual-v3-0","name":"Embed Multilingual v3.0","provider":"Cohere","api_id":"embed-multilingual-v3.0","status":"ga","context_window":512,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","batch","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2023-11","best_for":"1024-dim 100+ language embedder and the reference model for rerank-v3.5's language support; legacy for new builds. No longer publicly priced."},{"slug":"cohere-embed-multilingual-light-v3-0","name":"Embed Multilingual Light v3.0","provider":"Cohere","api_id":"embed-multilingual-light-v3.0","status":"ga","context_window":512,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","batch","fine-tuning"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2023-11","best_for":"Smallest multilingual embedder (384 dims) for cheap large-scale indexes; superseded by Embed 4 with truncated dimensions. No longer publicly priced."},{"slug":"cohere-rerank-4-pro","name":"Rerank 4 Pro","provider":"Cohere","api_id":"rerank-v4.0-pro","status":"ga","context_window":32000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"Highest-quality multilingual reranker with an 8x larger 32K context than Rerank 3.5, for complex or long-document relevance work. Priced per search, not per token: $2.50 per 1,000 searches (one search = one query against up to 100 documents); Model Vault instances are $5.00/hr (Medium) or $10.00/hr (Large)."},{"slug":"cohere-rerank-4-fast","name":"Rerank 4 Fast","provider":"Cohere","api_id":"rerank-v4.0-fast","status":"ga","context_window":32000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"The throughput/latency variant of Rerank 4 with the same 32K context — the right default for reranking in a live search path. Priced per search, not per token: $2.00 per 1,000 searches (one search = one query against up to 100 documents); $5.00/hr Medium on Model Vault."},{"slug":"cohere-rerank-3-5","name":"Rerank 3.5","provider":"Cohere","api_id":"rerank-v3.5","status":"ga","context_window":4000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-12","best_for":"Previous-generation multilingual reranker (4K context, auto-chunks longer docs), still served and still the version available on Bedrock and Oracle OCI. Per-search list price is no longer published; dedicated capacity is $5.00/hr or $3,250/mo on Model Vault."},{"slug":"cohere-rerank-english-v3-0","name":"Rerank English v3.0","provider":"Cohere","api_id":"rerank-english-v3.0","status":"ga","context_window":4000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-04","best_for":"English-only v3 reranker kept for existing Azure/SageMaker deployments; Rerank 3.5 already covers English plus other languages at equal context. No published price."},{"slug":"cohere-rerank-multilingual-v3-0","name":"Rerank Multilingual v3.0","provider":"Cohere","api_id":"rerank-multilingual-v3.0","status":"ga","context_window":4000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"2024-04","best_for":"Non-English v3 reranker matching embed-multilingual-v3.0's languages; only for pinned legacy stacks — Rerank 3.5 or Rerank 4 otherwise. No published price."},{"slug":"cohere-parse-5","name":"Parse 5","provider":"Cohere","api_id":"parse-v5.0","status":"ga","context_window":8192,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","pdf"],"capabilities":["vision"],"flagship":false,"knowledge_cutoff":"","license":"proprietary","released":"","best_for":"A 2.3B document parser (PDF/PPT/JPEG in, Markdown + HTML tables + bounding boxes out) for high-volume RAG ingestion; nine stable languages, no confidence scores and no JSON output. Priced per page, not per token: $1.50 per 1,000 pages, or $4.00/hr (Medium) / $7.00/hr (XL) on Model Vault."},{"slug":"cohere-cohere-transcribe","name":"Cohere Transcribe","provider":"Cohere","api_id":"cohere-transcribe-03-2026","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-03","best_for":"Open (Apache-2.0) 2B Conformer ASR covering 14 languages with a real-time factor up to 3x faster than similar-size models; no timestamps, no diarization and no language auto-detect, so pin the language. Free on the API but capped at 5 req/min and 25MB per file; production is per-instance on Model Vault from $3.75/hr."},{"slug":"cohere-cohere-transcribe-arabic","name":"Cohere Transcribe Arabic","provider":"Cohere","api_id":"cohere-transcribe-arabic-07-2026","status":"ga","context_window":0,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["audio"],"capabilities":[],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-07","best_for":"Arabic-specialised fine-tune of Cohere Transcribe — use it over the base model for any Arabic audio; same 25MB file cap, and it is not yet on Bedrock/Azure/Oracle. Per-token price not applicable; Model Vault instance pricing applies."},{"slug":"cohere-aya-expanse-32b","name":"Aya Expanse 32B","provider":"Cohere","api_id":"c4ai-aya-expanse-32b","status":"ga","context_window":128000,"max_output_tokens":4000,"input_price":0.5,"output_price":1.5,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2024-10","best_for":"Research-grade 23-language model with a 128K window at a flat $0.50/$1.50 — the cheapest way to test Cohere-family multilingual quality, but CC-BY-NC weights and a research positioning make it a poor commercial default versus Command R."},{"slug":"cohere-aya-vision-32b","name":"Aya Vision 32B","provider":"Cohere","api_id":"c4ai-aya-vision-32b","status":"ga","context_window":16000,"max_output_tokens":4000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","streaming"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2025-03","best_for":"Open multilingual vision-language research model across 23 languages; the pricing FAQ covers only Aya Expanse, so its API price is unstated — for commercial multimodal work use Command A Vision or Command A+."},{"slug":"cohere-tiny-aya-global","name":"Tiny Aya Global","provider":"Cohere","api_id":"tiny-aya-global","status":"ga","context_window":8000,"max_output_tokens":8000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2026-02","best_for":"The balanced 3.35B/70-language Tiny Aya variant — the one to start with when you want broad low-resource language coverage on small hardware (GGUF builds published). No public API price."},{"slug":"cohere-tiny-aya-earth","name":"Tiny Aya Earth","provider":"Cohere","api_id":"tiny-aya-earth","status":"ga","context_window":8000,"max_output_tokens":8000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2026-02","best_for":"Region-specialised Tiny Aya tuned for West Asian and African languages; choose it over Tiny Aya Global only when your traffic is concentrated in that region. No public API price."},{"slug":"cohere-tiny-aya-fire","name":"Tiny Aya Fire","provider":"Cohere","api_id":"tiny-aya-fire","status":"ga","context_window":8000,"max_output_tokens":8000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2026-02","best_for":"Region-specialised Tiny Aya tuned for South Asian languages; worth the swap from Tiny Aya Global for Indic-heavy workloads. No public API price."},{"slug":"cohere-tiny-aya-water","name":"Tiny Aya Water","provider":"Cohere","api_id":"tiny-aya-water","status":"ga","context_window":8000,"max_output_tokens":8000,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["streaming"],"flagship":false,"knowledge_cutoff":"","license":"CC-BY-NC-4.0","released":"2026-02","best_for":"Region-specialised Tiny Aya tuned for European and Asia-Pacific languages; the pick for EU/APAC-focused small-model deployments. No public API price."},{"slug":"chinese-labs-kimi-k3","name":"Kimi K3","provider":"Chinese AI labs","api_id":"kimi-k3","status":"ga","context_window":1048576,"max_output_tokens":-1,"input_price":2.98,"output_price":14.9,"cached_input_price":0.3,"modalities_in":["text","image","video"],"capabilities":["reasoning","agentic tool use","vision","function calling","context caching","OpenAI-compatible API","Anthropic-compatible API"],"flagship":true,"knowledge_cutoff":"","license":"Kimi K3 License (custom, license_name kimi-k3 on Hugging Face)","released":"2026-06","best_for":"Frontier long-context agent work where you can hit cache: the 10x cache-hit discount ($0.30 vs $3.00) matters far more than the headline rate, and $15 output makes it expensive for chatty workloads."},{"slug":"chinese-labs-kimi-k2-7-code","name":"Kimi K2.7 Code","provider":"Chinese AI labs","api_id":"kimi-k2.7-code","status":"ga","context_window":262144,"max_output_tokens":0,"input_price":0.95,"output_price":4,"cached_input_price":0.19,"modalities_in":["text"],"capabilities":["coding","long-horizon agentic coding","function calling","context caching"],"flagship":false,"knowledge_cutoff":"","license":"Modified MIT","released":"2026-06","best_for":"The value pick for autonomous coding agents — roughly a third of K3's input cost and a quarter of its output cost, with 256k context and weights you can self-host under Modified MIT."},{"slug":"chinese-labs-kimi-k2-7-code-highspeed","name":"Kimi K2.7 Code Highspeed","provider":"Chinese AI labs","api_id":"kimi-k2.7-code-highspeed","status":"ga","context_window":262144,"max_output_tokens":0,"input_price":1.9,"output_price":8,"cached_input_price":0.38,"modalities_in":["text"],"capabilities":["coding","high-throughput serving (~180 tok/s, up to 260 on short contexts)"],"flagship":false,"knowledge_cutoff":"","license":"Modified MIT","released":"","best_for":"Same weights as kimi-k2.7-code at exactly 2x the price — only worth it when interactive latency, not cost, is the constraint."},{"slug":"chinese-labs-kimi-k2-6","name":"Kimi K2.6","provider":"Chinese AI labs","api_id":"kimi-k2.6","status":"ga","context_window":262144,"max_output_tokens":0,"input_price":0.97,"output_price":4.02,"cached_input_price":0.16,"modalities_in":["text","image","video"],"capabilities":["vision","thinking and non-thinking modes","agentic tasks","function calling"],"flagship":false,"knowledge_cutoff":"","license":"Modified MIT","released":"2026-02","best_for":"Cheapest multimodal option in the Kimi line — pick it over K2.7 Code when you need image/video input and over K3 when 256k context is enough."},{"slug":"chinese-labs-glm-5-3","name":"GLM-5.3","provider":"Chinese AI labs","api_id":"glm-5.3","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":1.4,"output_price":4.4,"cached_input_price":0.26,"modalities_in":["text"],"capabilities":["always-on reasoning (low/high/max effort)","function calling","context caching","structured output","MCP","OpenAI + Anthropic protocols"],"flagship":false,"knowledge_cutoff":"","license":"GLM-5.3 License (custom; license_name glm-5.3 on Hugging Face)","released":"2026-08-25","best_for":"Best frontier-class price/context ratio here — 1M context at $1.40 in / $4.40 out, but note reasoning cannot be disabled, so budget for thinking tokens on the output side."},{"slug":"chinese-labs-glm-5-3-flash","name":"GLM-5.3-Flash","provider":"Chinese AI labs","api_id":"glm-5.3-flash","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":0.075,"output_price":0.25,"cached_input_price":0.015,"modalities_in":["text","image","video","file"],"capabilities":["native multimodal","reasoning","function calling","context caching"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"2026-08-25","best_for":"The cheapest capable multimodal model in this entire report and plain MIT weights — but the posted rate is a 50%-off promo running to 9 Sep 2026, so model your budget on double it."},{"slug":"chinese-labs-glm-5-2","name":"GLM-5.2","provider":"Chinese AI labs","api_id":"glm-5.2","status":"ga","context_window":1000000,"max_output_tokens":131072,"input_price":1.4,"output_price":4.4,"cached_input_price":0.26,"modalities_in":["text"],"capabilities":["thinking mode (toggleable)","function calling","context caching","structured output","MCP"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"2026-06-16","best_for":"Same base model and same price as GLM-5.3 but under plain MIT and with thinking optionally off — the one to self-host, or to use when you need non-reasoning responses."},{"slug":"chinese-labs-glm-5","name":"GLM-5","provider":"Chinese AI labs","api_id":"glm-5","status":"ga","context_window":200000,"max_output_tokens":131072,"input_price":1,"output_price":3.2,"cached_input_price":0.2,"modalities_in":["text"],"capabilities":["thinking mode","function calling","context caching","structured output"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"2026-02-11","best_for":"Cheaper than GLM-5.2/5.3 if 200k context is enough; otherwise the newer siblings are worth the extra 40 cents per million input."},{"slug":"chinese-labs-glm-4-7","name":"GLM-4.7","provider":"Chinese AI labs","api_id":"glm-4.7","status":"ga","context_window":200000,"max_output_tokens":131072,"input_price":0.6,"output_price":2.2,"cached_input_price":0.11,"modalities_in":["text"],"capabilities":["thinking mode","function calling","context caching","structured output"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"","best_for":"Solid mid-tier coding/agent workhorse at under half GLM-5.2's price; step down to it when you don't need million-token context."},{"slug":"chinese-labs-glm-4-7-flashx","name":"GLM-4.7-FlashX","provider":"Chinese AI labs","api_id":"glm-4.7-flashx","status":"ga","context_window":200000,"max_output_tokens":131072,"input_price":0.07,"output_price":0.4,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["low latency","function calling"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"High-volume classification, routing and extraction where GLM-5.3-Flash's multimodality is unnecessary — but check GLM-5.3-Flash first, it is currently cheaper on input."},{"slug":"chinese-labs-glm-4-7-flash","name":"GLM-4.7-Flash","provider":"Chinese AI labs","api_id":"glm-4.7-flash","status":"ga","context_window":200000,"max_output_tokens":131072,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["free tier"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Listed as free on Z.ai's price page — fine for prototyping, but treat a free SKU as rate-limited and revocable, never as production capacity."},{"slug":"chinese-labs-glm-4-6v","name":"GLM-4.6V","provider":"Chinese AI labs","api_id":"glm-4.6v","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0.3,"output_price":0.9,"cached_input_price":0.05,"modalities_in":["text","image","video","file"],"capabilities":["visual understanding","native function calling","multimodal agents"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Dedicated vision endpoint for document/chart/video understanding when you don't need GLM-5.3-Flash's 1M context; glm-4.6v-flashx ($0.04/$0.40) is the budget variant."},{"slug":"chinese-labs-glm-4-5-air","name":"GLM-4.5-Air","provider":"Chinese AI labs","api_id":"glm-4.5-air","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":0.2,"output_price":1.1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["function calling","agentic tasks"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"","best_for":"Legacy small model kept alive for existing integrations — new builds should start on GLM-4.7-FlashX or GLM-5.3-Flash instead."},{"slug":"chinese-labs-minimax-m3","name":"MiniMax-M3","provider":"Chinese AI labs","api_id":"MiniMax-M3","status":"ga","context_window":1000000,"max_output_tokens":0,"input_price":0.3,"output_price":1.2,"cached_input_price":0.06,"modalities_in":["text","image"],"capabilities":["agentic tool use","coding","1M context","prompt caching","OpenAI-compatible API","Anthropic-compatible API"],"flagship":false,"knowledge_cutoff":"","license":"MiniMax Model License (license_name minimax-community)","released":"2026-06","best_for":"Outstanding cost-per-context: 1M window at $0.30/$1.20 with open weights — just watch the tier break, since anything over 512k input bills at double."},{"slug":"chinese-labs-minimax-m2-7","name":"MiniMax-M2.7","provider":"Chinese AI labs","api_id":"MiniMax-M2.7","status":"ga","context_window":204800,"max_output_tokens":0,"input_price":0.3,"output_price":1.2,"cached_input_price":0.06,"modalities_in":["text"],"capabilities":["agentic tool use","coding","prompt caching"],"flagship":false,"knowledge_cutoff":"","license":"MiniMax Model License (custom; see LICENSE on Hugging Face)","released":"2026-04","best_for":"Same price as M3 with a fifth of the context and no vision — only choose it if you have already validated against these exact weights."},{"slug":"chinese-labs-minimax-m2-7-highspeed","name":"MiniMax-M2.7-highspeed","provider":"Chinese AI labs","api_id":"MiniMax-M2.7-highspeed","status":"ga","context_window":204800,"max_output_tokens":0,"input_price":0.6,"output_price":2.4,"cached_input_price":0.06,"modalities_in":["text"],"capabilities":["low-latency serving","prompt caching"],"flagship":false,"knowledge_cutoff":"","license":"MiniMax Model License (custom; see LICENSE on Hugging Face)","released":"","best_for":"A 2x latency surcharge on identical weights — justify it with a measured p95 requirement, not a hunch."},{"slug":"chinese-labs-seed-2-1-pro","name":"Seed 2.1 Pro","provider":"Chinese AI labs","api_id":"","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":0.8909,"output_price":4.4547,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["reasoning","agentic execution","coding","long video understanding"],"flagship":false,"knowledge_cutoff":"","license":"","released":"2026","best_for":"ByteDance's strongest reasoning/agent model, sold via Volcengine Ark (CN) and BytePlus ModelArk (intl) — but you must pull the rate card from the console yourself, as no fetchable official page publishes it."},{"slug":"chinese-labs-seed-2-1-turbo","name":"Seed 2.1 Turbo","provider":"Chinese AI labs","api_id":"dola-seed-2-1-turbo","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":0.4455,"output_price":2.2274,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["high-throughput serving","multimodal understanding","agentic tasks"],"flagship":false,"knowledge_cutoff":"","license":"","released":"2026","best_for":"The throughput-oriented half of the Seed 2.1 family for consumer-scale products; price unverified from any official page, so confirm before committing."},{"slug":"chinese-labs-hunyuan-a13b","name":"hunyuan-a13b","provider":"Chinese AI labs","api_id":"hunyuan-a13b","status":"ga","context_window":224000,"max_output_tokens":32768,"input_price":0.074,"output_price":0.297,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["hybrid fast/slow reasoning","long-document understanding","math","function calling"],"flagship":false,"knowledge_cutoff":"","license":"Tencent Hunyuan A13B Community License","released":"","best_for":"Absurdly cheap 224k-context reasoning (¥0.5/¥2 per 1M = $0.07/$0.30) — the budget choice for bulk long-document work if you can live with a Chinese-cloud endpoint."},{"slug":"chinese-labs-hunyuan-turbos-latest","name":"hunyuan-turbos-latest","provider":"Chinese AI labs","api_id":"hunyuan-turbos-latest","status":"retired","context_window":-1,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["chat completions","function calling","multi-turn dialogue"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Tencent's default chat endpoint, but it is absent from the published price table (billing is migrating to TokenHub) — get a quote before designing around it."},{"slug":"chinese-labs-hunyuan-vision-1-5-instruct","name":"hunyuan-vision-1.5-instruct","provider":"Chinese AI labs","api_id":"hunyuan-vision-1.5-instruct","status":"ga","context_window":24000,"max_output_tokens":16000,"input_price":0.445,"output_price":1.337,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["image recognition","fast-thinking visual analysis"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Fast image understanding at ¥3/¥9 per 1M; the 24k context is the real limit, so batch pages rather than whole documents."},{"slug":"chinese-labs-hunyuan-t1-vision","name":"hunyuan-t1-vision","provider":"Chinese AI labs","api_id":"hunyuan-t1-vision-20250916","status":"ga","context_window":28000,"max_output_tokens":20000,"input_price":0.445,"output_price":1.337,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["deep-thinking vision","OCR","chart reasoning","visual grounding"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Same price as the instruct vision model but with explicit reasoning — the better pick for OCR-plus-inference tasks like reading charts and solving from images."},{"slug":"chinese-labs-hunyuan-turbos-vision-video","name":"hunyuan-turbos-vision-video","provider":"Chinese AI labs","api_id":"hunyuan-turbos-vision-video","status":"ga","context_window":24000,"max_output_tokens":8000,"input_price":0.445,"output_price":1.337,"cached_input_price":-1,"modalities_in":["text","video"],"capabilities":["video description","video Q&A"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Cheap video captioning and Q&A; the 8k output cap rules out long transcript-style generation."},{"slug":"chinese-labs-hunyuan-role-latest","name":"hunyuan-role-latest","provider":"Chinese AI labs","api_id":"hunyuan-role-latest","status":"ga","context_window":28000,"max_output_tokens":4096,"input_price":0.356,"output_price":1.426,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["role-play","character consistency","emotional dialogue"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Purpose-built for companion/persona apps with strong character consistency — a general model at this price will drift more."},{"slug":"chinese-labs-hunyuan-translation","name":"hunyuan-translation","provider":"Chinese AI labs","api_id":"hunyuan-translation","status":"ga","context_window":4000,"max_output_tokens":4000,"input_price":0.178,"output_price":0.535,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["translation across 33 languages plus 5 minority languages","WMT25 winner in 30 languages"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Benchmark-leading MT at ~$0.18/$0.53 per 1M; the 4k window forces segment-level batching, which is usually what you want for translation anyway."},{"slug":"chinese-labs-hunyuan-translation-lite","name":"hunyuan-translation-lite","provider":"Chinese AI labs","api_id":"hunyuan-translation-lite","status":"ga","context_window":4000,"max_output_tokens":4000,"input_price":0.149,"output_price":0.446,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["lightweight translation, 16+ languages incl. Cantonese and Japanese"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Only ~17% cheaper than full hunyuan-translation with far fewer languages — take the full model unless you are latency-bound."},{"slug":"chinese-labs-hunyuan-hy4-preview","name":"Hunyuan Hy4-preview","provider":"Chinese AI labs","api_id":"","status":"preview","context_window":1000000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","agentic tasks","1M context"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-08","best_for":"Open weights only — Tencent sells no first-party per-token endpoint for it, so budget for GPUs (770B params) or a third-party host; the Apache-2.0 licence makes it the most commercially permissive frontier model here."},{"slug":"chinese-labs-hunyuan-hy3","name":"Hunyuan Hy3","provider":"Chinese AI labs","api_id":"","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","coding","agentic tasks"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"2026-07","best_for":"Apache-2.0, 21B active — the most self-hostable strong model in this report, and the practical Tencent choice when Hy4-preview's 770B is too big for your cluster. No first-party API."},{"slug":"chinese-labs-ernie-5-1","name":"ERNIE 5.1","provider":"Chinese AI labs","api_id":"ernie-5.1","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0.594,"output_price":2.673,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","general chat","function calling"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Baidu's current flagship and a third cheaper than ERNIE 5.0 on input; prices step up to $0.89/$3.27 once input exceeds 32k, so chunk aggressively."},{"slug":"chinese-labs-ernie-5-0","name":"ERNIE 5.0","provider":"Chinese AI labs","api_id":"ernie-5.0","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0.891,"output_price":3.564,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","general chat","thinking variants (ernie-5.0-thinking-latest / -preview / -exp)"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Superseded by ERNIE 5.1 at a lower price — keep it only for pinned behaviour, and note the 32k-128k tier bills $1.49/$5.94."},{"slug":"chinese-labs-ernie-4-5-turbo-128k","name":"ERNIE-4.5-Turbo-128K","provider":"Chinese AI labs","api_id":"ernie-4.5-turbo-128k","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":0.119,"output_price":0.475,"cached_input_price":0.03,"modalities_in":["text"],"capabilities":["fast general chat","context caching","function calling"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Baidu's volume workhorse at ~$0.12/$0.48 per 1M with a 4x cache discount — five times cheaper than ERNIE 5.1 and the right default for high-throughput extraction."},{"slug":"chinese-labs-ernie-4-5-turbo-vl-32k","name":"ERNIE-4.5-Turbo-VL-32K","provider":"Chinese AI labs","api_id":"ernie-4.5-turbo-vl-32k","status":"ga","context_window":32000,"max_output_tokens":0,"input_price":0.446,"output_price":1.337,"cached_input_price":0.111,"modalities_in":["text","image"],"capabilities":["vision-language understanding","context caching"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Baidu's multimodal endpoint; priced ~4x the text Turbo, so route image traffic here and text elsewhere rather than sending everything to VL."},{"slug":"chinese-labs-ernie-4-5-21b-a3b-thinking","name":"ERNIE-4.5-21B-A3B-Thinking","provider":"Chinese AI labs","api_id":"","status":"ga","context_window":131072,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","long context"],"flagship":false,"knowledge_cutoff":"","license":"Apache-2.0","released":"","best_for":"Small Apache-2.0 reasoning model that runs on a single modern GPU; not a line item in Qianfan's price table, so treat it as self-host-only."},{"slug":"chinese-labs-spark-4-0-ultra","name":"Spark 4.0 Ultra","provider":"Chinese AI labs","api_id":"4.0Ultra","status":"ga","context_window":32768,"max_output_tokens":32768,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["general chat","function calling","X1.5 fast-thinking mode","WebSocket API"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"iFlytek's recommended tier and the migration target as Spark Max retires 10 Mar 2026 — but pricing is quote-based, and the WebSocket-only API is a real integration cost versus every OpenAI-compatible rival here."},{"slug":"chinese-labs-spark-pro-128k","name":"Spark Pro-128K","provider":"Chinese AI labs","api_id":"pro-128k","status":"ga","context_window":131072,"max_output_tokens":131072,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["long context","WebSocket API"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"The only Spark tier with a long window; choose it for document work, but confirm the rate with iFlytek sales since no public price exists."},{"slug":"chinese-labs-spark-max-32k","name":"Spark Max-32K","provider":"Chinese AI labs","api_id":"max-32k","status":"ga","context_window":32768,"max_output_tokens":32768,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["general chat","WebSocket API"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Mid tier with the same window as Ultra and less capability — little reason to pick it over 4.0Ultra for new work."},{"slug":"chinese-labs-spark-lite","name":"Spark Lite","provider":"Chinese AI labs","api_id":"lite","status":"ga","context_window":8192,"max_output_tokens":4096,"input_price":0,"output_price":0,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["lightweight chat","WebSocket API"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Entry tier for simple classification; the 8k window and 4k output cap make it unsuitable for anything document-shaped."},{"slug":"chinese-labs-spark-max","name":"Spark Max","provider":"Chinese AI labs","api_id":"generalv3.5","status":"deprecated","context_window":8192,"max_output_tokens":8192,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["general chat","WebSocket API"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Being retired on 10 Mar 2026 — migrate existing integrations to 4.0Ultra now."},{"slug":"other-labs-jamba-large-1-7","name":"Jamba Large 1.7","provider":"Other notable labs","api_id":"jamba-large-1.7","status":"ga","context_window":262144,"max_output_tokens":4096,"input_price":2,"output_price":8,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["long-context","tool-use","grounding","json-mode","multilingual"],"flagship":true,"knowledge_cutoff":"2024-08-22","license":"Jamba Open Model License","released":"2025-07-02","best_for":"Long-document grounded QA where you need a 256K window and citations-faithful answers on a budget; the hybrid SSM design keeps long-context cost far below dense frontier models."},{"slug":"other-labs-jamba-mini-1-7","name":"Jamba Mini 1.7","provider":"Other notable labs","api_id":"jamba-mini-1.7","status":"retired","context_window":256000,"max_output_tokens":4096,"input_price":0.2,"output_price":0.4,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["long-context","tool-use","grounding","multilingual"],"flagship":false,"knowledge_cutoff":"2024-08-22","license":"Jamba Open Model License","released":"2025-07-02","best_for":"High-volume RAG and summarisation over long inputs; one of the cheapest 256K-context commercial endpoints available."},{"slug":"other-labs-jamba2-mini","name":"Jamba2 Mini","provider":"Other notable labs","api_id":"ai21labs/AI21-Jamba2-Mini","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["long-context","tool-use","grounding"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0","released":"2026-02","best_for":"Self-hosted enterprise QA where you want 256K context and Apache-2.0 freedom; answers without the token overhead of a reasoning model. Not on AI21's paid price list."},{"slug":"other-labs-jamba2-3b","name":"Jamba2 3B","provider":"Other notable labs","api_id":"ai21labs/AI21-Jamba2-3B","status":"ga","context_window":256000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["long-context","on-device","rag"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0","released":"2026-02","best_for":"On-device RAG on iOS/Android/desktop when you need an unusually large 256K window from a 3B model."},{"slug":"other-labs-jamba-reasoning-3b","name":"Jamba Reasoning 3B","provider":"Other notable labs","api_id":"ai21labs/AI21-Jamba-Reasoning-3B","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0","released":"","best_for":"Local reasoning at 3B scale; evaluate against Qwen and LFM2.5 thinking models before committing, since AI21 publishes no hosted endpoint for it."},{"slug":"other-labs-reka-core","name":"Reka Core","provider":"Other notable labs","api_id":"reka-core","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":2,"output_price":6,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["multimodal","video-understanding"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Reka's top multimodal tier for video and image Q&A; priced like a mid-tier model but from a small vendor, so validate availability and rate limits first."},{"slug":"other-labs-reka-flash","name":"Reka Flash","provider":"Other notable labs","api_id":"reka-flash","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":0.8,"output_price":2,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["multimodal","video-understanding"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"The default Reka SKU: multimodal at roughly Haiku-class pricing, always publicly available per Reka's docs."},{"slug":"other-labs-reka-edge-reka-edge-2603","name":"Reka Edge (reka-edge-2603)","provider":"Other notable labs","api_id":"reka-edge","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":0.1,"output_price":0.1,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["multimodal","object-detection","tool-use","on-device"],"flagship":false,"knowledge_cutoff":"","license":"reka-edge-2603-license","released":"","best_for":"Cheapest way to bolt image/video understanding onto a high-volume pipeline, and the weights are downloadable if you'd rather run it yourself; check the bespoke licence before commercial use."},{"slug":"other-labs-reka-flash-3-1","name":"Reka Flash 3.1","provider":"Other notable labs","api_id":"RekaAI/reka-flash-3.1","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning"],"flagship":false,"knowledge_cutoff":"","license":"","released":"2025-07","best_for":"A self-hostable 21B reasoning model with a matching RekaQuant 3-bit build for constrained GPUs; note this open release is distinct from the paid reka-flash endpoint."},{"slug":"other-labs-olmo-3-1-32b-think","name":"Olmo 3.1 32B Think","provider":"Other notable labs","api_id":"allenai/Olmo-3.1-32B-Think","status":"ga","context_window":65536,"max_output_tokens":32768,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","long-cot","math","code"],"flagship":false,"knowledge_cutoff":"2024-12","license":"Apache 2.0","released":"2025-12-23","best_for":"The strongest fully-reproducible reasoning model: choose it when you must audit or re-derive the training pipeline, not when you need best-in-class benchmark scores."},{"slug":"other-labs-olmo-3-1-32b-instruct","name":"Olmo 3.1 32B Instruct","provider":"Other notable labs","api_id":"allenai/Olmo-3.1-32B-Instruct","status":"ga","context_window":65536,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["chat","instruction-following"],"flagship":false,"knowledge_cutoff":"2024-12","license":"Apache 2.0","released":"2025-12","best_for":"General chat variant of the fully-open 32B; Ai2 flags it as intended for research and educational use, so read the Responsible Use Guidelines before shipping it in a product."},{"slug":"other-labs-olmo-3-32b-base","name":"Olmo 3 32B Base","provider":"Other notable labs","api_id":"allenai/Olmo-3-32B","status":"ga","context_window":65536,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["base-model","continued-pretraining"],"flagship":false,"knowledge_cutoff":"2024-12","license":"Apache 2.0","released":"2025-11-20","best_for":"The base checkpoint to fine-tune when you need documented provenance for every training token (Dolma 3, ~9.3T tokens) plus intermediate checkpoints."},{"slug":"other-labs-olmo-3-7b-instruct","name":"Olmo 3 7B Instruct","provider":"Other notable labs","api_id":"allenai/Olmo-3-7B-Instruct","status":"ga","context_window":65536,"max_output_tokens":32768,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["chat","instruction-following","math","code"],"flagship":false,"knowledge_cutoff":"2024-12","license":"Apache 2.0","released":"2025-11-20","best_for":"Small fully-open chat model for academic baselines and ablation studies where a licence-clean, data-transparent 7B is the requirement."},{"slug":"other-labs-olmo-3-7b-think","name":"Olmo 3 7B Think","provider":"Other notable labs","api_id":"allenai/Olmo-3-7B-Think","status":"ga","context_window":65536,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","long-cot"],"flagship":false,"knowledge_cutoff":"2024-12","license":"Apache 2.0","released":"2025-11-20","best_for":"Cheap local long-chain-of-thought experiments; pair with the RL-Zero checkpoints if you are studying RL recipes rather than deploying."},{"slug":"other-labs-molmo-2-8b","name":"Molmo 2 8B","provider":"Other notable labs","api_id":"allenai/Molmo2-8B","status":"ga","context_window":16384,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["vision","video-understanding","spatio-temporal-grounding","counting","captioning"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0 (some training data is academic / non-commercial research only)","released":"2025-12-11","best_for":"Best open choice when you need pointing/grounding output — coordinates and timestamps rather than prose — on short video and multi-image inputs; check the non-commercial data caveat first."},{"slug":"other-labs-molmo-2-4b","name":"Molmo 2 4B","provider":"Other notable labs","api_id":"allenai/Molmo2-4B","status":"ga","context_window":16384,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["vision","video-understanding","grounding"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0 (some training data is academic / non-commercial research only)","released":"2025-12-11","best_for":"The efficiency-tuned Molmo 2 for single-GPU video grounding when the 8B is too heavy."},{"slug":"other-labs-molmo-2-o-7b","name":"Molmo 2-O 7B","provider":"Other notable labs","api_id":"allenai/Molmo2-O-7B","status":"ga","context_window":16384,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["vision","video-understanding","grounding"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0 (some training data is academic / non-commercial research only)","released":"2025-12-11","best_for":"The only end-to-end open VLM here whose language backbone is also fully open (Olmo, not Qwen) — pick it when backbone provenance is the point."},{"slug":"other-labs-nvidia-nemotron-3-ultra-550b-a55b","name":"NVIDIA Nemotron 3 Ultra 550B-A55B","provider":"Other notable labs","api_id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","status":"ga","context_window":1000000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","agentic","long-context","tool-use","multilingual"],"flagship":true,"knowledge_cutoff":"2025-09 (pre-training), 2026-05 (post-training)","license":"OpenMDW-1.1","released":"2026-06-04","best_for":"Frontier-scale open weights for on-prem agentic workloads — but it needs 8x GB200/B200 or 16x H100 minimum, so it is a datacentre commitment, not a download."},{"slug":"other-labs-nvidia-nemotron-3-super-120b-a12b","name":"NVIDIA Nemotron 3 Super 120B-A12B","provider":"Other notable labs","api_id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16","status":"ga","context_window":1000000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","agentic","long-context","tool-use","multilingual"],"flagship":false,"knowledge_cutoff":"2025-06 (pre-training), 2026-02 (post-training)","license":"NVIDIA Nemotron Open Model License","released":"2026-03-11","best_for":"The practical sweet spot of the Nemotron line: 12B active params keeps throughput high for high-volume ticket automation and RAG while retaining a 256K default window."},{"slug":"other-labs-nvidia-nemotron-3-5-lightning-30b-a3b","name":"NVIDIA Nemotron 3.5 Lightning 30B-A3B","provider":"Other notable labs","api_id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","status":"ga","context_window":1000000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","toggleable-thinking","long-context","code","tool-use"],"flagship":false,"knowledge_cutoff":"2025-09 (pre-training), 2026-05 (post-training)","license":"OpenMDW-1.1","released":"2026-08-11","best_for":"Newest and most deployable Nemotron: 256K context on a single H100 with only 3B active params, and reasoning can be switched off per-request to cut token spend."},{"slug":"other-labs-nvidia-nemotron-nano-12b-v2-vl","name":"NVIDIA Nemotron Nano 12B v2 VL","provider":"Other notable labs","api_id":"nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image","video"],"capabilities":["vision","document-intelligence","video-understanding","ocr"],"flagship":false,"knowledge_cutoff":"","license":"NVIDIA Open Model License Agreement","released":"2025-10-28","best_for":"Document intelligence — invoices, forms, charts — at up to 4 images of 2048x1536 per request; a strong self-hosted alternative to paid OCR/VLM APIs."},{"slug":"other-labs-nvidia-nemotron-3-nano-4b","name":"NVIDIA Nemotron 3 Nano 4B","provider":"Other notable labs","api_id":"nvidia/NVIDIA-Nemotron-3-Nano-4B","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["on-device","sub-agent"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Local/edge sub-agent tier of the Nemotron family; verify the exact model card before relying on specs, as details were not confirmed on a primary card."},{"slug":"other-labs-granite-4-2-30b","name":"Granite 4.2 30B","provider":"Other notable labs","api_id":"ibm-granite/granite-4.2-30b","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","toggleable-thinking","tool-use","agentic","code","multilingual"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0","released":"2026-08-25","best_for":"Apache-2.0 enterprise workhorse with 128K native context (512K extensible) and three effort levels, so you can dial reasoning cost per request; deploy on vLLM/SGLang rather than expecting an IBM per-token SKU."},{"slug":"other-labs-granite-4-2-8b","name":"Granite 4.2 8B","provider":"Other notable labs","api_id":"ibm-granite/granite-4.2-8b","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","toggleable-thinking","tool-use","agentic","code","multilingual"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0","released":"2026-08-25","best_for":"The best value in the Granite line for self-hosted agents: 128K context and reliable tool calling on a single mid-range GPU, 12 languages, no licence friction."},{"slug":"other-labs-granite-4-2-3b","name":"Granite 4.2 3B","provider":"Other notable labs","api_id":"ibm-granite/granite-4.2-3b","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","toggleable-thinking","tool-use","multilingual"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0","released":"2026-08","best_for":"Smallest Granite 4.2 for CPU or edge deployment where you still want a 128K window; GGUF, nvfp4 and mxfp4 builds ship alongside."},{"slug":"other-labs-lfm2-5-8b-a1b","name":"LFM2.5-8B-A1B","provider":"Other notable labs","api_id":"LiquidAI/LFM2.5-8B-A1B","status":"ga","context_window":128000,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","tool-use","agentic","on-device","multilingual"],"flagship":false,"knowledge_cutoff":"","license":"lfm1.0","released":"2026","best_for":"Largest LFM: strong agentic tool use at 1.5B active params and 18.5K tok/s at high concurrency — but Liquid explicitly says it is weak on heavy coding and knowledge QA without retrieval, so pair it with RAG."},{"slug":"other-labs-lfm2-5-2-6b","name":"LFM2.5-2.6B","provider":"Other notable labs","api_id":"LiquidAI/LFM2.5-2.6B","status":"ga","context_window":131072,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tool-use","instruction-following","on-device","multilingual"],"flagship":false,"knowledge_cutoff":"","license":"lfm1.0","released":"2026-08-04","best_for":"The default on-device model here: 131K context, 16 languages, and competitive with models 4x its size on tool use — just note the bespoke lfm1.0 licence is not Apache."},{"slug":"other-labs-lfm2-5-vl-3b","name":"LFM2.5-VL-3B","provider":"Other notable labs","api_id":"LiquidAI/LFM2.5-VL-3B","status":"ga","context_window":32768,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["vision","on-device","multilingual"],"flagship":false,"knowledge_cutoff":"","license":"lfm1.0","released":"2026-08-12","best_for":"Vision on a laptop or NPU: ~3.3GB memory and 228 tok/s on an M5 Max, at the cost of a comparatively short 32K window."},{"slug":"other-labs-lfm2-5-1-2b-thinking","name":"LFM2.5-1.2B-Thinking","provider":"Other notable labs","api_id":"LiquidAI/LFM2.5-1.2B-Thinking","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["reasoning","on-device"],"flagship":false,"knowledge_cutoff":"","license":"","released":"2026-01-20","best_for":"Reasoning traces at 1.2B for phone-class hardware; confirm licence and context on the model card before deployment."},{"slug":"other-labs-lfm2-24b-a2b","name":"LFM2-24B-A2B","provider":"Other notable labs","api_id":"LiquidAI/LFM2-24B-A2B","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["on-device","moe"],"flagship":false,"knowledge_cutoff":"","license":"","released":"2026-02-24","best_for":"Liquid's larger MoE for workstation-class local inference; superseded in practice by LFM2.5-8B-A1B for agentic work."},{"slug":"other-labs-sonar","name":"Sonar","provider":"Other notable labs","api_id":"sonar","status":"deprecated","context_window":128000,"max_output_tokens":0,"input_price":1,"output_price":1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["web-search","grounding","citations"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Cheapest grounded-answer endpoint for straightforward factual lookups — but budget the $5-$12 per 1K request search fee on top, and migrate to the Agent API before Sonar chat-completions support ends 2026-09-27."},{"slug":"other-labs-sonar-pro","name":"Sonar Pro","provider":"Other notable labs","api_id":"sonar-pro","status":"deprecated","context_window":200000,"max_output_tokens":0,"input_price":3,"output_price":15,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["web-search","grounding","citations","multi-step-qa","long-context"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Use when a query needs follow-ups and breadth — roughly 2x the search results of base Sonar — but the $15/1M output plus $6-$14 per 1K requests makes it expensive for chatty workloads."},{"slug":"other-labs-sonar-reasoning-pro","name":"Sonar Reasoning Pro","provider":"Other notable labs","api_id":"sonar-reasoning-pro","status":"deprecated","context_window":128000,"max_output_tokens":0,"input_price":2,"output_price":8,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["web-search","reasoning","chain-of-thought","citations"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"Multi-step analytical questions over live web data; be ready to strip a leading <think> block before parsing JSON, and add the $6-$14 per 1K request fee to your cost model."},{"slug":"other-labs-sonar-deep-research","name":"Sonar Deep Research","provider":"Other notable labs","api_id":"sonar-deep-research","status":"deprecated","context_window":128000,"max_output_tokens":0,"input_price":2,"output_price":8,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["web-search","deep-research","report-generation","citations"],"flagship":false,"knowledge_cutoff":"","license":"","released":"","best_for":"One-shot long research reports across hundreds of sources; true cost is dominated by the separately-billed citation ($2/1M), reasoning ($3/1M) and search-query ($5/1K) charges, not the headline token price."},{"slug":"other-labs-apriel-1-6-15b-thinker","name":"Apriel 1.6 15B Thinker","provider":"Other notable labs","api_id":"ServiceNow-AI/Apriel-1.6-15b-Thinker","status":"ga","context_window":131072,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text","image"],"capabilities":["reasoning","vision","tool-use","code","enterprise-workflows"],"flagship":false,"knowledge_cutoff":"","license":"MIT","released":"2025-12-09","best_for":"MIT-licensed multimodal reasoner sized for a single GPU; ServiceNow claims ~30% fewer reasoning tokens than Apriel 1.5, which matters more than raw benchmark deltas for enterprise agent cost."},{"slug":"other-labs-arctic-awm-8b","name":"Arctic-AWM-8B","provider":"Other notable labs","api_id":"Snowflake/Arctic-AWM-8B","status":"ga","context_window":-1,"max_output_tokens":0,"input_price":-1,"output_price":-1,"cached_input_price":-1,"modalities_in":["text"],"capabilities":["tool-use","agentic","multi-turn","mcp"],"flagship":false,"knowledge_cutoff":"","license":"Apache 2.0","released":"2026-02-10","best_for":"Narrow but useful: a small model RL-trained specifically for multi-turn MCP tool calling. Also ships at 4B and 14B. Treat it as a research artifact, not a general chat model."}]