{
  "checked": "2026-09-06",
  "source": "Derived from the preserved 419-record reference dataset and archived model pages",
  "count": 149,
  "models": [
    {
      "slug": "openai-gpt-oss-120b",
      "name": "gpt-oss-120b",
      "provider": "OpenAI",
      "providerSlug": "openai",
      "api_id": "gpt-oss-120b",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 131072,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "streaming",
        "structured-output",
        "batch"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-06-01",
      "released": "2025-08-05",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "117B total / 5.1B active MoE",
      "paramsTotalB": 117,
      "paramsActiveB": 5.1,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "Self-hosting a frontier-ish reasoning model on a single H100 with full chain-of-thought visibility and no per-token bill. OpenAI publishes no hosted price for it on the API pricing page, so cost depends on your own or a third party's infrastructure.",
      "notes": [
        "OpenAI publishes no hosted price for it on the API pricing page, so cost depends on your own or a third party's infrastructure."
      ],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/openai/gpt-oss-120b",
          "primary": true
        },
        {
          "label": "OpenAI pricing",
          "url": "https://developers.openai.com/api/docs/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://developers.openai.com/api/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/openai-gpt-oss-120b/",
      "page": "/open-source-large-language-models/openai-gpt-oss-120b/"
    },
    {
      "slug": "openai-gpt-oss-20b",
      "name": "gpt-oss-20b",
      "provider": "OpenAI",
      "providerSlug": "openai",
      "api_id": "gpt-oss-20b",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 131072,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "streaming",
        "structured-output",
        "batch"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-06-01",
      "released": "2025-08-05",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "21B total / 3.6B active MoE",
      "paramsTotalB": 21,
      "paramsActiveB": 3.6,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "On-device or low-latency local inference with tool calling; runs in ~16GB. No OpenAI-hosted price is published on the pricing page.",
      "notes": [
        "No OpenAI-hosted price is published on the pricing page."
      ],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/openai/gpt-oss-20b",
          "primary": true
        },
        {
          "label": "OpenAI pricing",
          "url": "https://developers.openai.com/api/docs/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://developers.openai.com/api/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/openai-gpt-oss-20b/",
      "page": "/open-source-large-language-models/openai-gpt-oss-20b/"
    },
    {
      "slug": "google-deepmind-gemma-4-31b-instruct",
      "name": "Gemma 4 31B Instruct",
      "provider": "Google DeepMind",
      "providerSlug": "google-deepmind",
      "api_id": "gemma-4-31b-it",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 0,
      "input_price": 0,
      "output_price": 0,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026",
      "license": "Gemma 4 license (Gemma Terms of Use)",
      "licenseKind": "custom",
      "parameters": "31B dense",
      "paramsTotalB": 31,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "vision",
      "best_for": "Largest dense open-weight Gemma for self-hosting on server-grade GPUs with a 256K window; hosted access via the Gemini API is free-tier only (the pricing page lists paid-tier rates as 'Not available'), so plan to run it yourself for production.",
      "notes": [
        "At a three-to-one input-to-output ratio, Gemma 4 31B Instruct costs $0.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $0.00 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "ai.google.dev",
          "url": "https://ai.google.dev/gemma/docs/core/gemma_on_gemini_api",
          "primary": true
        },
        {
          "label": "Google DeepMind pricing",
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://ai.google.dev/gemini-api/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/google-deepmind-gemma-4-31b-instruct/",
      "page": "/open-source-large-language-models/google-deepmind-gemma-4-31b-instruct/"
    },
    {
      "slug": "google-deepmind-gemma-4-26b-a4b-instruct",
      "name": "Gemma 4 26B A4B Instruct",
      "provider": "Google DeepMind",
      "providerSlug": "google-deepmind",
      "api_id": "gemma-4-26b-a4b-it",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 0,
      "input_price": 0,
      "output_price": 0,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026",
      "license": "Gemma 4 license (Gemma Terms of Use)",
      "licenseKind": "custom",
      "parameters": "26B total / 4B active MoE",
      "paramsTotalB": 26,
      "paramsActiveB": 4,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "vision",
      "best_for": "Best open-weight throughput-per-dollar in the family — MoE activating only 4B params per token, so it serves near-4B-dense speed at 26B-class quality; free-tier only on the hosted Gemini API, weights on Kaggle/Hugging Face with QAT quantized builds.",
      "notes": [
        "At a three-to-one input-to-output ratio, Gemma 4 26B A4B Instruct costs $0.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $0.00 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "ai.google.dev",
          "url": "https://ai.google.dev/gemma/docs/core/gemma_on_gemini_api",
          "primary": true
        },
        {
          "label": "Google DeepMind pricing",
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://ai.google.dev/gemini-api/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/google-deepmind-gemma-4-26b-a4b-instruct/",
      "page": "/open-source-large-language-models/google-deepmind-gemma-4-26b-a4b-instruct/"
    },
    {
      "slug": "google-deepmind-gemma-4-12b",
      "name": "Gemma 4 12B",
      "provider": "Google DeepMind",
      "providerSlug": "google-deepmind",
      "api_id": "",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 0,
      "input_price": 0,
      "output_price": 0,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026",
      "license": "Gemma 4 license (Gemma Terms of Use)",
      "licenseKind": "custom",
      "parameters": "12B",
      "paramsTotalB": 12,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "Encoder-free multimodal mid-size open model with a 256K window — the sweet spot for single-GPU self-hosting; download-only (not exposed on the hosted Gemini API).",
      "notes": [
        "At a three-to-one input-to-output ratio, Gemma 4 12B costs $0.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $0.00 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "ai.google.dev",
          "url": "https://ai.google.dev/gemma/docs/core",
          "primary": true
        },
        {
          "label": "Google DeepMind pricing",
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://ai.google.dev/gemini-api/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/google-deepmind-gemma-4-12b/",
      "page": "/open-source-large-language-models/google-deepmind-gemma-4-12b/"
    },
    {
      "slug": "google-deepmind-gemma-4-e4b",
      "name": "Gemma 4 E4B",
      "provider": "Google DeepMind",
      "providerSlug": "google-deepmind",
      "api_id": "",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": 0,
      "output_price": 0,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026",
      "license": "Gemma 4 license (Gemma Terms of Use)",
      "licenseKind": "custom",
      "parameters": "4B effective",
      "paramsTotalB": 4,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "vision",
      "best_for": "On-device/edge deployment with a 128K window; download-only, with QAT quantized builds for phone and laptop inference engines.",
      "notes": [
        "At a three-to-one input-to-output ratio, Gemma 4 E4B costs $0.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $0.00 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "ai.google.dev",
          "url": "https://ai.google.dev/gemma/docs/core",
          "primary": true
        },
        {
          "label": "Google DeepMind pricing",
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://ai.google.dev/gemini-api/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/google-deepmind-gemma-4-e4b/",
      "page": "/open-source-large-language-models/google-deepmind-gemma-4-e4b/"
    },
    {
      "slug": "google-deepmind-gemma-4-e2b",
      "name": "Gemma 4 E2B",
      "provider": "Google DeepMind",
      "providerSlug": "google-deepmind",
      "api_id": "",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": 0,
      "output_price": 0,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026",
      "license": "Gemma 4 license (Gemma Terms of Use)",
      "licenseKind": "custom",
      "parameters": "2B effective",
      "paramsTotalB": 2,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "vision",
      "best_for": "Smallest Gemma 4 — ultra-mobile and embedded inference at 128K context; download-only, no hosted endpoint.",
      "notes": [
        "At a three-to-one input-to-output ratio, Gemma 4 E2B costs $0.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $0.00 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "ai.google.dev",
          "url": "https://ai.google.dev/gemma/docs/core",
          "primary": true
        },
        {
          "label": "Google DeepMind pricing",
          "url": "https://ai.google.dev/gemini-api/docs/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://ai.google.dev/gemini-api/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/google-deepmind-gemma-4-e2b/",
      "page": "/open-source-large-language-models/google-deepmind-gemma-4-e2b/"
    },
    {
      "slug": "meta-muse-glimmer-30b",
      "name": "Muse Glimmer 30B",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-models/Muse-Glimmer-30B",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": -1,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "reasoning",
        "tool-calling",
        "image-understanding",
        "local-inference",
        "fine-tuning",
        "quantization",
        "speculative-decoding",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "",
      "license": "Apache License 2.0",
      "licenseKind": "permissive",
      "parameters": "30B dense",
      "paramsTotalB": 30,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "vision",
      "best_for": "The one Meta open-weight model with a genuinely permissive licence (Apache 2.0, no Community-License user-count or naming clauses) — a 30B dense multimodal model distilled from Muse Spark, with 128K default context and quantized GGUF builds that fit 24-32GB VRAM. Weights only: Meta charges nothing and sells no API for it, so both prices are -1 and your cost is your own hardware or a third-party host.",
      "notes": [
        "Weights only: Meta charges nothing and sells no API for it, so both prices are -1 and your cost is your own hardware or a third-party host."
      ],
      "sources": [
        {
          "label": "dev.meta.ai",
          "url": "https://dev.meta.ai/docs/muse-glimmer",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-muse-glimmer-30b/",
      "page": "/open-source-large-language-models/meta-muse-glimmer-30b/"
    },
    {
      "slug": "meta-llama-4-scout",
      "name": "Llama 4 Scout",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
      "status": "ga",
      "flagship": true,
      "context_window": 10000000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "multilingual",
        "image-understanding",
        "early-fusion-multimodal",
        "long-context",
        "fine-tuning",
        "distillation"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-08",
      "released": "2025-04-05",
      "license": "Llama 4 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "17B active / 109B total, 16 experts (MoE)",
      "paramsTotalB": 109,
      "paramsActiveB": 17,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "vision",
      "best_for": "The long-context option in Meta's open-weight line — a 10M-token window and single-H100 deployability at int4 make it the pick for whole-repository or whole-corpus ingestion; Meta sells no API for it, so both prices are -1 and any per-token figure you see belongs to a third-party host.",
      "notes": [],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama4/MODEL_CARD.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-4-scout/",
      "page": "/open-source-large-language-models/meta-llama-4-scout/"
    },
    {
      "slug": "meta-llama-4-maverick",
      "name": "Llama 4 Maverick",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "multilingual",
        "image-understanding",
        "early-fusion-multimodal",
        "long-context",
        "fine-tuning",
        "distillation"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-08",
      "released": "2025-04-05",
      "license": "Llama 4 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "17B active / 400B total, 128 experts (MoE)",
      "paramsTotalB": 400,
      "paramsActiveB": 17,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "Higher-quality sibling to Scout with the same 17B activated cost per token but 400B total weights — better reasoning and image understanding, at the price of multi-GPU hosting and a shorter 1M window. Weights-only from Meta: both prices -1.",
      "notes": [
        "Weights-only from Meta: both prices -1."
      ],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama4/MODEL_CARD.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-4-maverick/",
      "page": "/open-source-large-language-models/meta-llama-4-maverick/"
    },
    {
      "slug": "meta-llama-3-3-70b-instruct",
      "name": "Llama 3.3 70B Instruct",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-3.3-70B-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "multilingual",
        "tool-calling",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-12",
      "released": "2024-12-06",
      "license": "Llama 3.3 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "70B",
      "paramsTotalB": 70,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "The best quality-per-GPU text-only Llama: 405B-class instruction following at 70B serving cost, which is why it remains the most widely hosted Llama despite Llama 4. Text only — reach for Llama 4 if you need vision or a window past 128K. Meta publishes no price; both prices -1.",
      "notes": [
        "Text only — reach for Llama 4 if you need vision or a window past 128K. Meta publishes no price; both prices -1."
      ],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_3/MODEL_CARD.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-3-3-70b-instruct/",
      "page": "/open-source-large-language-models/meta-llama-3-3-70b-instruct/"
    },
    {
      "slug": "meta-llama-3-2-90b-vision-instruct",
      "name": "Llama 3.2 90B Vision Instruct",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-3.2-90B-Vision-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "image-understanding",
        "document-understanding",
        "chart-reasoning",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "license": "Llama 3.2 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "90B (88.8B)",
      "paramsTotalB": 90,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "large",
      "kind": "vision",
      "best_for": "A cross-attention vision adapter bolted onto Llama 3.1 70B — capable at charts, documents and captioning, but image+text is English-only and it is superseded by Llama 4's native early fusion; pick it only when you need a 3.x-lineage vision model. Prices -1: weights only.",
      "notes": [
        "Prices -1: weights only."
      ],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/MODEL_CARD_VISION.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-3-2-90b-vision-instruct/",
      "page": "/open-source-large-language-models/meta-llama-3-2-90b-vision-instruct/"
    },
    {
      "slug": "meta-llama-3-2-11b-vision-instruct",
      "name": "Llama 3.2 11B Vision Instruct",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-3.2-11B-Vision-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "image-understanding",
        "document-understanding",
        "captioning",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "license": "Llama 3.2 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "11B (10.6B)",
      "paramsTotalB": 11,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "Small enough to fine-tune for a single narrow vision task (receipt or form extraction, screenshot classification) on one commodity GPU; not a general-purpose VLM. Weights only, so both prices -1.",
      "notes": [
        "Weights only, so both prices -1."
      ],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/MODEL_CARD_VISION.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-3-2-11b-vision-instruct/",
      "page": "/open-source-large-language-models/meta-llama-3-2-11b-vision-instruct/"
    },
    {
      "slug": "meta-llama-3-2-3b-instruct",
      "name": "Llama 3.2 3B Instruct",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-3.2-3B-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "multilingual",
        "summarization",
        "retrieval",
        "on-device",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-12",
      "released": "2024-10-24",
      "license": "Llama 3.2 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "3B (3.21B)",
      "paramsTotalB": 3,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "On-device and edge summarisation, rewriting and query expansion where latency and privacy beat raw capability; note Meta's own quantized builds drop the context window from 128K to 8K. Prices -1 — weights only.",
      "notes": [
        "Prices -1 — weights only."
      ],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/MODEL_CARD.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-3-2-3b-instruct/",
      "page": "/open-source-large-language-models/meta-llama-3-2-3b-instruct/"
    },
    {
      "slug": "meta-llama-3-2-1b-instruct",
      "name": "Llama 3.2 1B Instruct",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-3.2-1B-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "multilingual",
        "summarization",
        "on-device",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-12",
      "released": "2024-10-24",
      "license": "Llama 3.2 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "1B (1.23B)",
      "paramsTotalB": 1,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "The smallest Llama, for phone- and microcontroller-class deployment or as a speculative-decoding draft model; expect to fine-tune it for one task rather than use it as a general assistant. Weights only: both prices -1.",
      "notes": [
        "Weights only: both prices -1."
      ],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/MODEL_CARD.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-3-2-1b-instruct/",
      "page": "/open-source-large-language-models/meta-llama-3-2-1b-instruct/"
    },
    {
      "slug": "meta-llama-3-1-405b-instruct",
      "name": "Llama 3.1 405B Instruct",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-3.1-405B-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "multilingual",
        "tool-calling",
        "synthetic-data-generation",
        "distillation",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-12",
      "released": "2024-07-23",
      "license": "Llama 3.1 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "405B",
      "paramsTotalB": 405,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Now mainly a teacher model: its licence explicitly permits using outputs to train other models, which is the remaining reason to run something this expensive when Llama 3.3 70B matches it on most chat benchmarks. Meta charges nothing and hosts nothing — both prices -1.",
      "notes": [
        "Meta charges nothing and hosts nothing — both prices -1."
      ],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_1/MODEL_CARD.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-3-1-405b-instruct/",
      "page": "/open-source-large-language-models/meta-llama-3-1-405b-instruct/"
    },
    {
      "slug": "meta-llama-3-1-70b-instruct",
      "name": "Llama 3.1 70B Instruct",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-3.1-70B-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "multilingual",
        "tool-calling",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-12",
      "released": "2024-07-23",
      "license": "Llama 3.1 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "70B",
      "paramsTotalB": 70,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "Superseded at the same size and cost by Llama 3.3 70B — keep it only for pinned reproducibility of existing 3.1 evaluations or fine-tunes. Weights only, so both prices -1.",
      "notes": [
        "Weights only, so both prices -1."
      ],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_1/MODEL_CARD.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-3-1-70b-instruct/",
      "page": "/open-source-large-language-models/meta-llama-3-1-70b-instruct/"
    },
    {
      "slug": "meta-llama-3-1-8b-instruct",
      "name": "Llama 3.1 8B Instruct",
      "provider": "Meta",
      "providerSlug": "meta",
      "api_id": "meta-llama/Llama-3.1-8B-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "multilingual",
        "tool-calling",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-12",
      "released": "2024-07-23",
      "license": "Llama 3.1 Community License Agreement",
      "licenseKind": "custom",
      "parameters": "8B",
      "paramsTotalB": 8,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Still the default single-GPU fine-tuning baseline across the open-source ecosystem because tooling support is universal; choose Llama 3.2 3B instead if you need something smaller and Muse Glimmer 30B if licence permissiveness matters. Prices -1 — weights only.",
      "notes": [
        "Prices -1 — weights only."
      ],
      "sources": [
        {
          "label": "github.com",
          "url": "https://github.com/meta-llama/llama-models/blob/main/models/llama3_1/MODEL_CARD.md",
          "primary": true
        },
        {
          "label": "Meta pricing",
          "url": "https://dev.meta.ai/docs/pricing-rate-limits",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://dev.meta.ai/docs/overview",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/meta-llama-3-1-8b-instruct/",
      "page": "/open-source-large-language-models/meta-llama-3-1-8b-instruct/"
    },
    {
      "slug": "deepseek-deepseek-v4-flash",
      "name": "DeepSeek-V4-Flash",
      "provider": "DeepSeek",
      "providerSlug": "deepseek",
      "api_id": "deepseek-v4-flash",
      "status": "ga",
      "flagship": true,
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "input_price": 0.44,
      "output_price": 1.32,
      "cached_input_price": 0.014,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "prompt-caching",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-07-31",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "304B total MoE (~13B active; V4 preview quoted 284B total / 13B active, architecture unchanged in the 0731 re-post-train)",
      "paramsTotalB": 304,
      "paramsActiveB": 13,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "The default DeepSeek pick: near-Pro reasoning at a third of the price with the same 1M context, ideal for high-volume agentic coding and long-document work — schedule batchy jobs outside 01:00-04:00/06:00-10:00 UTC weekdays and you pay $0.22/$0.66 instead of $0.44/$1.32.",
      "notes": [
        "At a three-to-one input-to-output ratio, DeepSeek-V4-Flash costs $0.66 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $26.27 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "DeepSeek pricing",
          "url": "https://api-docs.deepseek.com/quick_start/pricing/",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://api-docs.deepseek.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/deepseek-deepseek-v4-flash/",
      "page": "/open-source-large-language-models/deepseek-deepseek-v4-flash/"
    },
    {
      "slug": "deepseek-deepseek-v4-pro",
      "name": "DeepSeek-V4-Pro",
      "provider": "DeepSeek",
      "providerSlug": "deepseek",
      "api_id": "deepseek-v4-pro",
      "status": "ga",
      "flagship": true,
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "input_price": 1.32,
      "output_price": 3.96,
      "cached_input_price": 0.044,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "prompt-caching",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-13",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "1.7T total MoE (~49B active; V4 preview quoted 1.6T total / 49B active)",
      "paramsTotalB": 1700,
      "paramsActiveB": 49,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "DeepSeek's frontier tier — worth the 3x premium over Flash only for hard agentic/production coding, deep reasoning and tool-heavy workflows where Flash's quality gap actually shows; note the tighter 500-concurrent limit and use reasoning_effort=max sparingly since thinking tokens bill at the output rate.",
      "notes": [
        "At a three-to-one input-to-output ratio, DeepSeek-V4-Pro costs $1.98 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $78.80 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "DeepSeek pricing",
          "url": "https://api-docs.deepseek.com/quick_start/pricing/",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://api-docs.deepseek.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/deepseek-deepseek-v4-pro/",
      "page": "/open-source-large-language-models/deepseek-deepseek-v4-pro/"
    },
    {
      "slug": "deepseek-deepseek-v4-flash-vision-exp",
      "name": "DeepSeek-V4-Flash-Vision-Exp",
      "provider": "DeepSeek",
      "providerSlug": "deepseek",
      "api_id": "deepseek-v4-flash-vision-exp",
      "status": "preview",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "input_price": 0.44,
      "output_price": 1.32,
      "cached_input_price": 0.014,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "prompt-caching",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-21",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "305B total MoE (V4-Flash architecture plus vision modules, continued-trained)",
      "paramsTotalB": 305,
      "paramsActiveB": null,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "The only DeepSeek model that takes images — use it for screenshot-driven and chart/document-reading agents at identical Flash pricing; it is explicitly experimental and drops FIM completion, so keep text-only traffic on deepseek-v4-flash. Images bill as input tokens (up to 384 tokens each), accepted as base64, public URL (max 32 MiB), or Files API file_id (max 64 MiB, upload is free).",
      "notes": [
        "Images bill as input tokens (up to 384 tokens each), accepted as base64, public URL (max 32 MiB), or Files API file_id (max 64 MiB, upload is free).",
        "At a three-to-one input-to-output ratio, DeepSeek-V4-Flash-Vision-Exp costs $0.66 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $26.27 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "DeepSeek pricing",
          "url": "https://api-docs.deepseek.com/quick_start/pricing/",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://api-docs.deepseek.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/deepseek-deepseek-v4-flash-vision-exp/",
      "page": "/open-source-large-language-models/deepseek-deepseek-v4-flash-vision-exp/"
    },
    {
      "slug": "alibaba-qwen-qwen3-8-2-4t-a95b",
      "name": "Qwen3.8-2.4T-A95B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3.8-2.4t-a95b",
      "status": "ga",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "input_price": 2,
      "output_price": 6,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08",
      "license": "Qwen3.8-Max License (custom, non-Apache)",
      "licenseKind": "permissive",
      "parameters": "2.4T-A95B MoE (512 experts, 11 active/token)",
      "paramsTotalB": 2400,
      "paramsActiveB": 95,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "The published base weights behind Qwen3.8-Max and the largest open-weight model in existence — text-only, thinking-mode-only, 262K native context (YaRN-extensible to ~1.01M). Note it ships under a custom licence, NOT Apache 2.0; the hosted qwen3.8-max adds vision, non-thinking mode and 1M context by default.",
      "notes": [
        "Note it ships under a custom licence, NOT Apache 2.0; the hosted qwen3.8-max adds vision, non-thinking mode and 1M context by default.",
        "At a three-to-one input-to-output ratio, Qwen3.8-2.4T-A95B costs $3.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $119.40 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "alibabacloud.com",
          "url": "https://www.alibabacloud.com/help/en/model-studio/billing-for-model-studio",
          "primary": true
        },
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-8-2-4t-a95b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-8-2-4t-a95b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-8-27b",
      "name": "Qwen3.8-27B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3.8-27b",
      "status": "ga",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "input_price": 0.5,
      "output_price": 3,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-14",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "27B dense",
      "paramsTotalB": 27,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "vision",
      "best_for": "The standout self-host pick — a truly Apache-2.0, 27B dense multimodal model with 262K context and adjustable reasoning effort (xhigh/medium/low) that fits on a single high-memory GPU in BF16 or comfortably in FP8.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3.8-27B costs $1.13 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $44.70 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "alibabacloud.com",
          "url": "https://www.alibabacloud.com/help/en/model-studio/billing-for-model-studio",
          "primary": true
        },
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-8-27b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-8-27b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-6-35b-a3b",
      "name": "Qwen3.6-35B-A3B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3.6-35b-a3b",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 65536,
      "input_price": 0.375,
      "output_price": 2.25,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-04-16",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "35B-A3B MoE",
      "paramsTotalB": 35,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "vision",
      "best_for": "Only 3B active parameters, so it runs fast on modest hardware while punching well above its weight on coding benchmarks — the sweet spot for local agentic coding under Apache 2.0. Thinking is enabled by default on the Qwen3.6 series.",
      "notes": [
        "Thinking is enabled by default on the Qwen3.6 series.",
        "At a three-to-one input-to-output ratio, Qwen3.6-35B-A3B costs $0.84 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $33.52 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-6-35b-a3b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-6-35b-a3b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-6-27b",
      "name": "Qwen3.6-27B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3.6-27b",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 65536,
      "input_price": 0.6,
      "output_price": 3.6,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-04",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "27B dense",
      "paramsTotalB": 27,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "vision",
      "best_for": "Dense 27B for workloads where MoE routing hurts quality consistency; now priced above the newer Qwen3.8-27B ($0.60/$3.60 vs $0.50/$3) — prefer 3.8-27B unless pinned.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3.6-27B costs $1.35 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $53.64 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-6-27b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-6-27b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-5-397b-a17b",
      "name": "Qwen3.5-397B-A17B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3.5-397b-a17b",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 65536,
      "input_price": 0.6,
      "output_price": 3.6,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2026-01",
      "released": "2026-02-16",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "397B-A17B MoE",
      "paramsTotalB": 397,
      "paramsActiveB": 17,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "The largest fully Apache-2.0 Qwen model — the one to self-host when licence purity matters more than raw scale and the custom-licensed Qwen3.8-2.4T is off the table. 262K native, YaRN-extensible to ~1.01M.",
      "notes": [
        "262K native, YaRN-extensible to ~1.01M.",
        "At a three-to-one input-to-output ratio, Qwen3.5-397B-A17B costs $1.35 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $53.64 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-5-397b-a17b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-5-397b-a17b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-5-122b-a10b",
      "name": "Qwen3.5-122B-A10B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3.5-122b-a10b",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 65536,
      "input_price": 0.4,
      "output_price": 3.2,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2026-01",
      "released": "2026-02",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "122B-A10B MoE",
      "paramsTotalB": 122,
      "paramsActiveB": 10,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Mid-size Apache-2.0 MoE that fits a single 8-GPU node in FP8 — a reasonable step down from the 397B when memory is the constraint.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3.5-122B-A10B costs $1.10 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $43.68 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-5-122b-a10b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-5-122b-a10b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-5-35b-a3b",
      "name": "Qwen3.5-35B-A3B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3.5-35b-a3b",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 65536,
      "input_price": 0.25,
      "output_price": 2,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2026-01",
      "released": "2026-02",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "35B-A3B MoE",
      "paramsTotalB": 35,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "Cheapest hosted Qwen3.5 open model and an easy local run at 3B active params; superseded on quality by Qwen3.6-35B-A3B for a small price premium.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3.5-35B-A3B costs $0.69 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $27.30 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-5-35b-a3b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-5-35b-a3b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-5-27b",
      "name": "Qwen3.5-27B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3.5-27b",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 65536,
      "input_price": 0.3,
      "output_price": 2.4,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2026-01",
      "released": "2026-02",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "27B dense",
      "paramsTotalB": 27,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "Dense 27B at the lowest price in the 27B line ($0.30/$2.40) — good value if you don't need the multimodal input that Qwen3.6-27B and Qwen3.8-27B add.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3.5-27B costs $0.82 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $32.76 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-5-27b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-5-27b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-coder-next",
      "name": "Qwen3-Coder-Next",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-coder-next",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 65536,
      "input_price": 0.3,
      "output_price": 1.5,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning",
        "code-execution"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-11",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "",
      "paramsTotalB": null,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "unrecorded",
      "kind": "coding",
      "best_for": "Best coding value in the catalogue — 5x cheaper than qwen3-coder-plus at the base tier and open-weight. Tiers: 0-32K $0.30/$1.50, 32-128K $0.50/$2.50, 128-256K $0.80/$4.",
      "notes": [
        "Tiers: 0-32K $0.30/$1.50, 32-128K $0.50/$2.50, 128-256K $0.80/$4.",
        "At a three-to-one input-to-output ratio, Qwen3-Coder-Next costs $0.60 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $23.85 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-coder-next/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-coder-next/"
    },
    {
      "slug": "alibaba-qwen-qwen3-coder-480b-a35b-instruct",
      "name": "Qwen3-Coder-480B-A35B-Instruct",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-coder-480b-a35b-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 200000,
      "max_output_tokens": 65536,
      "input_price": 1.5,
      "output_price": 7.5,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning",
        "code-execution"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-07",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "480B-A35B MoE",
      "paramsTotalB": 480,
      "paramsActiveB": 35,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "coding",
      "best_for": "The heavyweight open coding model for hard multi-file agentic tasks; expensive and tier-sensitive (0-32K $1.50/$7.50, 32-128K $2.70/$13.50, 128-200K $4.50/$22.50) — benchmark it against qwen3-coder-next before committing.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-Coder-480B-A35B-Instruct costs $3.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $119.25 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-coder-480b-a35b-instruct/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-coder-480b-a35b-instruct/"
    },
    {
      "slug": "alibaba-qwen-qwen3-coder-30b-a3b-instruct",
      "name": "Qwen3-Coder-30B-A3B-Instruct",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-coder-30b-a3b-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 200000,
      "max_output_tokens": 65536,
      "input_price": 0.45,
      "output_price": 2.25,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning",
        "code-execution"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-07",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "30B-A3B MoE",
      "paramsTotalB": 30,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "coding",
      "best_for": "Laptop-to-workstation coding model with 3B active params; the practical choice for local IDE autocomplete and small refactors. Tiers: 0-32K $0.45/$2.25, 32-128K $0.75/$3.75, 128-200K $1.20/$6.",
      "notes": [
        "Tiers: 0-32K $0.45/$2.25, 32-128K $0.75/$3.75, 128-200K $1.20/$6.",
        "At a three-to-one input-to-output ratio, Qwen3-Coder-30B-A3B-Instruct costs $0.90 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $35.77 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-coder-30b-a3b-instruct/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-coder-30b-a3b-instruct/"
    },
    {
      "slug": "alibaba-qwen-qwen3-next-80b-a3b-thinking",
      "name": "Qwen3-Next-80B-A3B-Thinking",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-next-80b-a3b-thinking",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 32768,
      "input_price": 0.15,
      "output_price": 1.2,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-09",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "80B-A3B MoE",
      "paramsTotalB": 80,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "reasoning",
      "best_for": "Very cheap open reasoning at $0.15 input — hybrid-attention architecture with 3B active params makes long-context inference unusually fast per dollar.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-Next-80B-A3B-Thinking costs $0.41 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $16.38 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-next-80b-a3b-thinking/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-next-80b-a3b-thinking/"
    },
    {
      "slug": "alibaba-qwen-qwen3-next-80b-a3b-instruct",
      "name": "Qwen3-Next-80B-A3B-Instruct",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-next-80b-a3b-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 32768,
      "input_price": 0.15,
      "output_price": 1.2,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-09",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "80B-A3B MoE",
      "paramsTotalB": 80,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "Non-thinking sibling for latency-sensitive work where you don't want reasoning tokens billed — same price as the thinking variant, so choose on behaviour not cost.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-Next-80B-A3B-Instruct costs $0.41 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $16.38 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-next-80b-a3b-instruct/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-next-80b-a3b-instruct/"
    },
    {
      "slug": "alibaba-qwen-qwen3-235b-a22b-thinking-2507",
      "name": "Qwen3-235B-A22B-Thinking-2507",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-235b-a22b-thinking-2507",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 32768,
      "input_price": 0.23,
      "output_price": 2.3,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-07",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "235B-A22B MoE",
      "paramsTotalB": 235,
      "paramsActiveB": 22,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "reasoning",
      "best_for": "The 2025 open reasoning workhorse, still one of the cheapest large thinking models at $0.23 input; Qwen3.5-397B beats it but needs far more memory to self-host.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-235B-A22B-Thinking-2507 costs $0.75 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $29.67 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-235b-a22b-thinking-2507/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-235b-a22b-thinking-2507/"
    },
    {
      "slug": "alibaba-qwen-qwen3-235b-a22b-instruct-2507",
      "name": "Qwen3-235B-A22B-Instruct-2507",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-235b-a22b-instruct-2507",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 32768,
      "input_price": 0.23,
      "output_price": 0.92,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-07",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "235B-A22B MoE",
      "paramsTotalB": 235,
      "paramsActiveB": 22,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Excellent bulk-throughput value at $0.23/$0.92 — 2.5x cheaper output than its thinking twin, ideal for summarisation and extraction at scale.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-235B-A22B-Instruct-2507 costs $0.40 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $16.01 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-235b-a22b-instruct-2507/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-235b-a22b-instruct-2507/"
    },
    {
      "slug": "alibaba-qwen-qwen3-30b-a3b-thinking-2507",
      "name": "Qwen3-30B-A3B-Thinking-2507",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-30b-a3b-thinking-2507",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 32768,
      "input_price": 0.2,
      "output_price": 2.4,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-07",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "30B-A3B MoE",
      "paramsTotalB": 30,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "reasoning",
      "best_for": "Small MoE reasoner that self-hosts on a single 48GB card in FP8; output pricing is high relative to size, so prefer local deployment over the API for volume.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-30B-A3B-Thinking-2507 costs $0.75 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $29.76 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-30b-a3b-thinking-2507/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-30b-a3b-thinking-2507/"
    },
    {
      "slug": "alibaba-qwen-qwen3-30b-a3b-instruct-2507",
      "name": "Qwen3-30B-A3B-Instruct-2507",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-30b-a3b-instruct-2507",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 32768,
      "input_price": 0.2,
      "output_price": 0.8,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-07",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "30B-A3B MoE",
      "paramsTotalB": 30,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "Fast, cheap non-thinking small MoE — a solid drop-in for classification, routing and tool-dispatch layers.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-30B-A3B-Instruct-2507 costs $0.35 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $13.92 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-30b-a3b-instruct-2507/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-30b-a3b-instruct-2507/"
    },
    {
      "slug": "alibaba-qwen-qwen3-235b-a22b",
      "name": "Qwen3-235B-A22B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-235b-a22b",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 16384,
      "input_price": 0.7,
      "output_price": 2.8,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-03",
      "released": "2025-04",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "235B-A22B MoE",
      "paramsTotalB": 235,
      "paramsActiveB": 22,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Original hybrid-thinking Qwen3 flagship; output is $2.80 non-thinking / $8.40 thinking. The -2507 refresh is better and 3x cheaper — migrate.",
      "notes": [
        "The -2507 refresh is better and 3x cheaper — migrate.",
        "At a three-to-one input-to-output ratio, Qwen3-235B-A22B costs $1.22 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $48.72 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-235b-a22b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-235b-a22b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-32b",
      "name": "Qwen3-32B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-32b",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 16384,
      "input_price": 0.16,
      "output_price": 0.64,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-03",
      "released": "2025-04",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "32B dense",
      "paramsTotalB": 32,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "The most widely deployed open Qwen3 dense model and a well-supported fine-tuning base; output $0.64 non-thinking. Newer 27B models beat it, but tooling maturity still favours it.",
      "notes": [
        "Newer 27B models beat it, but tooling maturity still favours it.",
        "At a three-to-one input-to-output ratio, Qwen3-32B costs $0.28 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $11.14 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-32b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-32b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-30b-a3b",
      "name": "Qwen3-30B-A3B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-30b-a3b",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 16384,
      "input_price": 0.2,
      "output_price": 0.8,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-03",
      "released": "2025-04",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "30B-A3B MoE",
      "paramsTotalB": 30,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "Original hybrid-mode 30B MoE; output $0.80 non-thinking / $2.40 thinking. Superseded by the -2507 split variants at the same price.",
      "notes": [
        "Superseded by the -2507 split variants at the same price.",
        "At a three-to-one input-to-output ratio, Qwen3-30B-A3B costs $0.35 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $13.92 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-30b-a3b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-30b-a3b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-14b",
      "name": "Qwen3-14B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-14b",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 8192,
      "input_price": 0.35,
      "output_price": 1.4,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-03",
      "released": "2025-04",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "14B dense",
      "paramsTotalB": 14,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Oddly priced above Qwen3-32B on the API ($0.35 vs $0.16) — only worth it as a self-hosted fine-tune target on a single 24-32GB GPU. Output $1.40 non-thinking / $4.20 thinking.",
      "notes": [
        "Output $1.40 non-thinking / $4.20 thinking.",
        "At a three-to-one input-to-output ratio, Qwen3-14B costs $0.61 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $24.36 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-14b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-14b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-8b",
      "name": "Qwen3-8B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-8b",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 8192,
      "input_price": 0.18,
      "output_price": 0.7,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-03",
      "released": "2025-04",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "8B dense",
      "paramsTotalB": 8,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Edge/consumer-GPU tier — runs quantised on 8-12GB VRAM. Output $0.70 non-thinking / $2.10 thinking. Smaller 4B/1.7B/0.6B siblings exist on Hugging Face but are not separately priced in Model Studio.",
      "notes": [
        "Output $0.70 non-thinking / $2.10 thinking. Smaller 4B/1.7B/0.6B siblings exist on Hugging Face but are not separately priced in Model Studio.",
        "At a three-to-one input-to-output ratio, Qwen3-8B costs $0.31 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $12.33 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-8b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-8b/"
    },
    {
      "slug": "alibaba-qwen-qwen3-vl-235b-a22b-thinking",
      "name": "Qwen3-VL-235B-A22B-Thinking",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-vl-235b-a22b-thinking",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 32768,
      "input_price": 0.4,
      "output_price": 4,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-09",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "235B-A22B MoE",
      "paramsTotalB": 235,
      "paramsActiveB": 22,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "reasoning",
      "best_for": "Largest open vision-language reasoner — use for chart/diagram reasoning and GUI-agent work where a small VL model fails; output is 2.5x the instruct variant.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-VL-235B-A22B-Thinking costs $1.30 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $51.60 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-vl-235b-a22b-thinking/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-vl-235b-a22b-thinking/"
    },
    {
      "slug": "alibaba-qwen-qwen3-vl-235b-a22b-instruct",
      "name": "Qwen3-VL-235B-A22B-Instruct",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-vl-235b-a22b-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 32768,
      "input_price": 0.4,
      "output_price": 1.6,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-09",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "235B-A22B MoE",
      "paramsTotalB": 235,
      "paramsActiveB": 22,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "Top open VL model for straightforward description/extraction at $0.40/$1.60 — no reasoning-token overhead.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-VL-235B-A22B-Instruct costs $0.70 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $27.84 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-vl-235b-a22b-instruct/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-vl-235b-a22b-instruct/"
    },
    {
      "slug": "alibaba-qwen-qwen3-vl-32b-thinking",
      "name": "Qwen3-VL-32B-Thinking",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-vl-32b-thinking",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 16384,
      "input_price": 0.16,
      "output_price": 0.64,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-10",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "32B dense",
      "paramsTotalB": 32,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "reasoning",
      "best_for": "Cheapest open VL reasoner on the API ($0.16/$0.64) and dense enough to self-host on two 40GB cards — strong default for visual QA pipelines.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-VL-32B-Thinking costs $0.28 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $11.14 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-vl-32b-thinking/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-vl-32b-thinking/"
    },
    {
      "slug": "alibaba-qwen-qwen3-vl-32b-instruct",
      "name": "Qwen3-VL-32B-Instruct",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-vl-32b-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 16384,
      "input_price": 0.16,
      "output_price": 0.64,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-10",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "32B dense",
      "paramsTotalB": 32,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "vision",
      "best_for": "Same price as the thinking variant with lower latency — pick this for OCR-adjacent and captioning work that needs no deliberation.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-VL-32B-Instruct costs $0.28 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $11.14 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-vl-32b-instruct/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-vl-32b-instruct/"
    },
    {
      "slug": "alibaba-qwen-qwen3-vl-30b-a3b-thinking",
      "name": "Qwen3-VL-30B-A3B-Thinking",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-vl-30b-a3b-thinking",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 16384,
      "input_price": 0.2,
      "output_price": 2.4,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-10",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "30B-A3B MoE",
      "paramsTotalB": 30,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "reasoning",
      "best_for": "Fast MoE visual reasoner (3B active) — good local throughput, but on the API the 32B-thinking model is cheaper on output ($0.64 vs $2.40).",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-VL-30B-A3B-Thinking costs $0.75 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $29.76 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-vl-30b-a3b-thinking/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-vl-30b-a3b-thinking/"
    },
    {
      "slug": "alibaba-qwen-qwen3-vl-30b-a3b-instruct",
      "name": "Qwen3-VL-30B-A3B-Instruct",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-vl-30b-a3b-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 16384,
      "input_price": 0.2,
      "output_price": 0.8,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-10",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "30B-A3B MoE",
      "paramsTotalB": 30,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "vision",
      "best_for": "Best latency-per-dollar open VL model for high-volume image pipelines you host yourself.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-VL-30B-A3B-Instruct costs $0.35 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $13.92 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-vl-30b-a3b-instruct/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-vl-30b-a3b-instruct/"
    },
    {
      "slug": "alibaba-qwen-qwen3-vl-8b-thinking",
      "name": "Qwen3-VL-8B-Thinking",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-vl-8b-thinking",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 8192,
      "input_price": 0.18,
      "output_price": 2.1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-10",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "8B dense",
      "paramsTotalB": 8,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "small",
      "kind": "reasoning",
      "best_for": "Smallest open VL reasoner — the one to fine-tune for a narrow visual domain on a single consumer GPU.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-VL-8B-Thinking costs $0.66 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $26.19 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-vl-8b-thinking/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-vl-8b-thinking/"
    },
    {
      "slug": "alibaba-qwen-qwen3-vl-8b-instruct",
      "name": "Qwen3-VL-8B-Instruct",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen3-vl-8b-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 8192,
      "input_price": 0.18,
      "output_price": 0.7,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "batch",
        "streaming",
        "structured-output",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-10",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "8B dense",
      "paramsTotalB": 8,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "Edge-deployable multimodal model for on-device captioning and screenshot understanding.",
      "notes": [
        "At a three-to-one input-to-output ratio, Qwen3-VL-8B-Instruct costs $0.31 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $12.33 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen3-vl-8b-instruct/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen3-vl-8b-instruct/"
    },
    {
      "slug": "alibaba-qwen-qwen2-5-omni-7b",
      "name": "Qwen2.5-Omni-7B",
      "provider": "Alibaba Qwen",
      "providerSlug": "alibaba-qwen",
      "api_id": "qwen2.5-omni-7b",
      "status": "ga",
      "flagship": false,
      "context_window": 32768,
      "max_output_tokens": 2048,
      "input_price": 0.1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "audio",
        "video"
      ],
      "modalities_out": [
        "text",
        "audio"
      ],
      "capabilities": [
        "vision",
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-03",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "7B dense",
      "paramsTotalB": 7,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "The small open any-to-any model people actually self-host for offline voice assistants; output pricing varies by modality and is not stated as a single figure on the pricing page. Superseded by the Qwen3.5-Omni hosted tier.",
      "notes": [
        "Superseded by the Qwen3.5-Omni hosted tier."
      ],
      "sources": [
        {
          "label": "Alibaba Qwen pricing",
          "url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://www.alibabacloud.com/help/en/model-studio/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/alibaba-qwen-qwen2-5-omni-7b/",
      "page": "/open-source-large-language-models/alibaba-qwen-qwen2-5-omni-7b/"
    },
    {
      "slug": "mistral-mistral-medium-3-5",
      "name": "Mistral Medium 3.5",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "mistral-medium-3-5",
      "status": "ga",
      "flagship": true,
      "context_window": 256000,
      "max_output_tokens": 0,
      "input_price": 1.5,
      "output_price": 7.5,
      "cached_input_price": 0.15,
      "modalities_in": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "prompt-caching",
        "batch",
        "streaming",
        "structured-output",
        "web-search",
        "code-execution"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-04-28",
      "license": "Modified MIT",
      "licenseKind": "permissive",
      "parameters": "128B dense",
      "paramsTotalB": 128,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "Mistral's current flagship for long-horizon agentic work and agentic coding — but note it costs 3x the input and 5x the output of Mistral Large 3, so only reach for it when tool-calling depth or coding quality actually justifies the premium. Aliases: mistral-medium-3, mistral-medium-latest.",
      "notes": [
        "Aliases: mistral-medium-3, mistral-medium-latest.",
        "At a three-to-one input-to-output ratio, Mistral Medium 3.5 costs $3.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $119.25 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/mistral-medium-3-5-26-04",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-mistral-medium-3-5/",
      "page": "/open-source-large-language-models/mistral-mistral-medium-3-5/"
    },
    {
      "slug": "mistral-mistral-large-3",
      "name": "Mistral Large 3",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "mistral-large-2512",
      "status": "ga",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 0,
      "input_price": 0.5,
      "output_price": 1.5,
      "cached_input_price": 0.05,
      "modalities_in": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "prompt-caching",
        "batch",
        "streaming",
        "structured-output",
        "web-search",
        "code-execution"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-12-02",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "675B-A41B MoE",
      "paramsTotalB": 675,
      "paramsActiveB": 41,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "The value pick of the whole lineup: frontier-size open-weight MoE at $0.50/$1.50, cheaper than Medium 3.5 and most rivals' mid-tier models — default choice for general multimodal chat, RAG and bulk generation unless you specifically need Medium 3.5's agentic coding. Alias: mistral-large-latest.",
      "notes": [
        "Alias: mistral-large-latest.",
        "At a three-to-one input-to-output ratio, Mistral Large 3 costs $0.75 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $29.85 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/mistral-large-3-25-12",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-mistral-large-3/",
      "page": "/open-source-large-language-models/mistral-mistral-large-3/"
    },
    {
      "slug": "mistral-mistral-small-4",
      "name": "Mistral Small 4",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "mistral-small-2603",
      "status": "ga",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 0,
      "input_price": 0.15,
      "output_price": 0.6,
      "cached_input_price": 0.015,
      "modalities_in": [
        "text",
        "image",
        "pdf"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "prompt-caching",
        "batch",
        "streaming",
        "structured-output",
        "web-search",
        "code-execution"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-03-16",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "119B-A6.5B MoE",
      "paramsTotalB": 119,
      "paramsActiveB": 6.5,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "vision",
      "best_for": "Best price/performance workhorse — a hybrid instruct+reasoning+coding model with only 6.5B active params, so it is fast and cheap while still handling agents and vision; the model to self-host under Apache 2.0 if you want reasoning without a licence conversation. Alias: mistral-small-latest.",
      "notes": [
        "Alias: mistral-small-latest.",
        "At a three-to-one input-to-output ratio, Mistral Small 4 costs $0.26 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $10.44 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/mistral-small-4-0-26-03",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-mistral-small-4/",
      "page": "/open-source-large-language-models/mistral-mistral-small-4/"
    },
    {
      "slug": "mistral-ministral-3-14b",
      "name": "Ministral 3 14B",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "ministral-14b-2512",
      "status": "ga",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 0,
      "input_price": 0.2,
      "output_price": 0.2,
      "cached_input_price": 0.02,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "prompt-caching",
        "batch",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-12-02",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "14B dense",
      "paramsTotalB": 14,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "Largest edge-class Ministral — flat $0.20 in and out makes it the cheapest option when your workload is output-heavy; separate Base, Instruct and Reasoning weights ship on Hugging Face for local/on-device deployment. Alias: ministral-14b-latest.",
      "notes": [
        "Alias: ministral-14b-latest.",
        "At a three-to-one input-to-output ratio, Ministral 3 14B costs $0.20 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $7.98 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/ministral-3-14b-25-12",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-ministral-3-14b/",
      "page": "/open-source-large-language-models/mistral-ministral-3-14b/"
    },
    {
      "slug": "mistral-ministral-3-8b",
      "name": "Ministral 3 8B",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "ministral-8b-2512",
      "status": "ga",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 0,
      "input_price": 0.15,
      "output_price": 0.15,
      "cached_input_price": 0.015,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "prompt-caching",
        "batch",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-12-02",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "8B dense",
      "paramsTotalB": 8,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "The sweet spot for high-volume classification, extraction and routing where you still want vision and 256K context; also the recommended local model for consumer GPUs. Alias: ministral-8b-latest.",
      "notes": [
        "Alias: ministral-8b-latest.",
        "At a three-to-one input-to-output ratio, Ministral 3 8B costs $0.15 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $5.99 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/ministral-3-8b-25-12",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-ministral-3-8b/",
      "page": "/open-source-large-language-models/mistral-ministral-3-8b/"
    },
    {
      "slug": "mistral-ministral-3-3b",
      "name": "Ministral 3 3B",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "ministral-3b-2512",
      "status": "ga",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 0,
      "input_price": 0.1,
      "output_price": 0.1,
      "cached_input_price": 0.01,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "prompt-caching",
        "batch",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-12-02",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "3B dense",
      "paramsTotalB": 3,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "tiny",
      "kind": "vision",
      "best_for": "Cheapest model Mistral serves and the one to run on-device — pick it for latency-critical edge inference or trivially structured tasks, not for anything needing world knowledge. Alias: ministral-3b-latest.",
      "notes": [
        "Alias: ministral-3b-latest.",
        "At a three-to-one input-to-output ratio, Ministral 3 3B costs $0.10 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $3.99 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/ministral-3-3b-25-12",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-ministral-3-3b/",
      "page": "/open-source-large-language-models/mistral-ministral-3-3b/"
    },
    {
      "slug": "mistral-z-ai-glm-5-2",
      "name": "Z.ai GLM 5.2",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "zai-glm-5-2",
      "status": "preview",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "input_price": 1.4,
      "output_price": 4.4,
      "cached_input_price": 0.14,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "prompt-caching",
        "batch",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-06",
      "license": "Open (third-party, Z.ai — licence set by Z.ai, not Mistral)",
      "licenseKind": "custom",
      "parameters": "",
      "paramsTotalB": null,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "unrecorded",
      "kind": "language",
      "best_for": "Use when you need a 1M-token window on Mistral's EU-hosted infrastructure — it is a third-party Z.ai model served without Mistral modifications, aimed at long-context coding and agentic workflows. Public preview.",
      "notes": [
        "Public preview.",
        "At a three-to-one input-to-output ratio, Z.ai GLM 5.2 costs $2.15 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $85.56 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/zai-glm-5-2",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-z-ai-glm-5-2/",
      "page": "/open-source-large-language-models/mistral-z-ai-glm-5-2/"
    },
    {
      "slug": "mistral-leanstral-1-5",
      "name": "Leanstral 1.5",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "labs-leanstral-1-5",
      "status": "preview",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 128000,
      "input_price": 0,
      "output_price": 0,
      "cached_input_price": 0,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-06-30",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "119B-A6.5B MoE",
      "paramsTotalB": 119,
      "paramsActiveB": 6.5,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "coding",
      "best_for": "Free Labs endpoint for Lean 4 formal proof engineering, autoformalization and automated theorem proving — genuinely $0 while Mistral gathers feedback, but it is a research preview with no availability guarantee, so don't build production on it. (The older labs-leanstral-2603 still shown on the marketing pricing page was retired 30 Jun 2026.)",
      "notes": [
        "(The older labs-leanstral-2603 still shown on the marketing pricing page was retired 30 Jun 2026.)",
        "At a three-to-one input-to-output ratio, Leanstral 1.5 costs $0.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $0.00 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/leanstral-1-5",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-leanstral-1-5/",
      "page": "/open-source-large-language-models/mistral-leanstral-1-5/"
    },
    {
      "slug": "mistral-voxtral-small",
      "name": "Voxtral Small",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "voxtral-small-2507",
      "status": "ga",
      "flagship": false,
      "context_window": 32000,
      "max_output_tokens": 0,
      "input_price": 0.1,
      "output_price": 0.4,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "audio"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "batch",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-07-15",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "24B dense",
      "paramsTotalB": 24,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "audio",
      "best_for": "The one Voxtral that does audio *understanding* rather than plain transcription — chat and Q&A over speech via /v1/chat/completions. Text tokens are $0.10/$0.40 per M; audio input is billed separately at $0.004 per audio minute. Alias: voxtral-small-latest.",
      "notes": [
        "Text tokens are $0.10/$0.40 per M; audio input is billed separately at $0.004 per audio minute. Alias: voxtral-small-latest.",
        "At a three-to-one input-to-output ratio, Voxtral Small costs $0.18 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $6.96 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/voxtral-small-25-07",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-voxtral-small/",
      "page": "/open-source-large-language-models/mistral-voxtral-small/"
    },
    {
      "slug": "mistral-voxtral-mini-transcribe-realtime",
      "name": "Voxtral Mini Transcribe Realtime",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "voxtral-mini-transcribe-realtime-2602",
      "status": "ga",
      "flagship": false,
      "context_window": 0,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "audio"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02-04",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "4B dense",
      "paramsTotalB": 4,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "tiny",
      "kind": "speech",
      "best_for": "Live/streaming transcription at $0.006 per audio minute — double the batch model's rate, so only use it when you actually need low-latency partial results; Apache 2.0 weights (Voxtral-Mini-4B-Realtime-2602) make self-hosting viable. Alias: voxtral-mini-transcribe-realtime-latest.",
      "notes": [
        "Alias: voxtral-mini-transcribe-realtime-latest."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/voxtral-mini-transcribe-realtime-26-02",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-voxtral-mini-transcribe-realtime/",
      "page": "/open-source-large-language-models/mistral-voxtral-mini-transcribe-realtime/"
    },
    {
      "slug": "mistral-voxtral-tts",
      "name": "Voxtral TTS",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "voxtral-mini-tts-2603",
      "status": "ga",
      "flagship": false,
      "context_window": 0,
      "max_output_tokens": 0,
      "input_price": 0,
      "output_price": -1,
      "cached_input_price": 0,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "audio"
      ],
      "capabilities": [
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-03-23",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "4B dense",
      "paramsTotalB": 4,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "tiny",
      "kind": "speech",
      "best_for": "Text-to-speech with zero-shot voice cloning (no transcript needed for the voice prompt), 9 languages and ~90ms time-to-first-audio. Billed per character, not per token: input is free, output is $16 per million characters ($0.016 per 1,000 characters). Weights are CC BY-NC 4.0, so commercial self-hosting is not permitted — use the API. Alias: voxtral-mini-tts-latest.",
      "notes": [
        "Billed per character, not per token: input is free, output is $16 per million characters ($0.016 per 1,000 characters). Weights are CC BY-NC 4.0, so commercial self-hosting is not permitted — use the API. Alias: voxtral-mini-tts-latest."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/voxtral-tts-26-03",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-voxtral-tts/",
      "page": "/open-source-large-language-models/mistral-voxtral-tts/"
    },
    {
      "slug": "mistral-shieldstral-1-0",
      "name": "Shieldstral 1.0",
      "provider": "Mistral AI",
      "providerSlug": "mistral",
      "api_id": "",
      "status": "preview",
      "flagship": false,
      "context_window": 32000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-04",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "3.8B dense",
      "paramsTotalB": 3.8,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "tiny",
      "kind": "safety",
      "best_for": "Compact self-hosted safety classifier: you write policy questions in natural language and it returns yes/no for prompt moderation, response moderation, prompt-response pair classification and refusal detection, over text and images. Weights-only — it is listed in the model catalogue but carries no API model id and appears on neither pricing page, so run it yourself.",
      "notes": [
        "Weights-only — it is listed in the model catalogue but carries no API model id and appears on neither pricing page, so run it yourself."
      ],
      "sources": [
        {
          "label": "docs.mistral.ai",
          "url": "https://docs.mistral.ai/models/shieldstral-1-0",
          "primary": true
        },
        {
          "label": "Mistral AI pricing",
          "url": "https://docs.mistral.ai/inference/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.mistral.ai/getting-started/models/models_overview/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/mistral-shieldstral-1-0/",
      "page": "/open-source-large-language-models/mistral-shieldstral-1-0/"
    },
    {
      "slug": "microsoft-mai-ds-r1",
      "name": "MAI-DS-R1",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "MAI-DS-R1",
      "status": "deprecated",
      "flagship": false,
      "context_window": 163840,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-04",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "671B-A37B MoE",
      "paramsTotalB": 671,
      "paramsActiveB": 37,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "reasoning",
      "best_for": "Microsoft's safety- and unblocking-post-trained variant of DeepSeek-R1 (671B MoE, 37B active, MIT). It has dropped out of the current Foundry catalog alongside the DeepSeek-R1 retirement, though legacy Azure meters still list $1.35/1M in and $5.40/1M out on Global Standard; the weights remain freely downloadable, so self-host rather than plan on the hosted endpoint.",
      "notes": [
        "It has dropped out of the current Foundry catalog alongside the DeepSeek-R1 retirement, though legacy Azure meters still list $1.35/1M in and $5.40/1M out on Global Standard; the weights remain freely downloadable, so self-host rather than plan on the hosted endpoint."
      ],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/microsoft/MAI-DS-R1",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-mai-ds-r1/",
      "page": "/open-source-large-language-models/microsoft-mai-ds-r1/"
    },
    {
      "slug": "microsoft-phi-4",
      "name": "Phi-4",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-4",
      "status": "ga",
      "flagship": true,
      "context_window": 16384,
      "max_output_tokens": 16384,
      "input_price": 0.125,
      "output_price": 0.5,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-06",
      "released": "2024-12-12",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "14.7B",
      "paramsTotalB": 14.7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "The dense 14B workhorse of the family: strongest general Phi quality per dollar for math, code and reasoning when a 16K context is enough — if you need long context, use Phi-4-mini-instruct instead.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-4 costs $0.22 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $8.70 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-4/",
      "page": "/open-source-large-language-models/microsoft-phi-4/"
    },
    {
      "slug": "microsoft-phi-4-mini-instruct",
      "name": "Phi-4-mini-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-4-mini-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 4096,
      "input_price": 0.075,
      "output_price": 0.3,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-06",
      "released": "2025-02",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "3.8B",
      "paramsTotalB": 3.8,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "Cheapest hosted Phi and the default for high-volume classification, extraction and routing across 23 languages with a real 128K window; no native tool calling, so wrap it yourself.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-4-mini-instruct costs $0.13 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $5.22 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-4-mini-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-4-mini-instruct/"
    },
    {
      "slug": "microsoft-phi-4-multimodal-instruct",
      "name": "Phi-4-multimodal-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-4-multimodal-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 4096,
      "input_price": 0.08,
      "output_price": 0.32,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "audio"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-06",
      "released": "2025-02",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "5.6B",
      "paramsTotalB": 5.6,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "The only Phi that takes audio as well as images — good for cheap on-device-class speech understanding, OCR and chart reading in one 5.6B model. Watch the meter: audio input is billed separately at $4.00 per 1M audio tokens, 50x the text input rate.",
      "notes": [
        "Watch the meter: audio input is billed separately at $4.00 per 1M audio tokens, 50x the text input rate.",
        "At a three-to-one input-to-output ratio, Phi-4-multimodal-instruct costs $0.14 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $5.57 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-4-multimodal-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-4-multimodal-instruct/"
    },
    {
      "slug": "microsoft-phi-4-reasoning",
      "name": "Phi-4-reasoning",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-4-reasoning",
      "status": "ga",
      "flagship": false,
      "context_window": 32768,
      "max_output_tokens": 32768,
      "input_price": 0.125,
      "output_price": 0.5,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-03",
      "released": "2025-04-30",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "14.7B",
      "paramsTotalB": 14.7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "reasoning",
      "best_for": "14B SFT-distilled reasoner at Phi-4 prices — the value pick for math and STEM chains of thought when you cannot justify MAI-Thinking-1 or a frontier model; English only.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-4-reasoning costs $0.22 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $8.70 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-4-reasoning/",
      "page": "/open-source-large-language-models/microsoft-phi-4-reasoning/"
    },
    {
      "slug": "microsoft-phi-4-reasoning-plus",
      "name": "Phi-4-reasoning-plus",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-4-reasoning-plus",
      "status": "ga",
      "flagship": false,
      "context_window": 32768,
      "max_output_tokens": 32768,
      "input_price": 0.125,
      "output_price": 0.5,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-03",
      "released": "2025-04-30",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "14.7B",
      "paramsTotalB": 14.7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "reasoning",
      "best_for": "RL-tuned sibling of Phi-4-reasoning at identical token prices — higher accuracy but noticeably longer chains of thought, so it costs more per answer in practice. Still priced and listed in the Foundry catalog, but dropped from the current 'Microsoft models' doc table, so treat its hosted lifetime as shorter than Phi-4-reasoning's.",
      "notes": [
        "Still priced and listed in the Foundry catalog, but dropped from the current 'Microsoft models' doc table, so treat its hosted lifetime as shorter than Phi-4-reasoning's.",
        "At a three-to-one input-to-output ratio, Phi-4-reasoning-plus costs $0.22 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $8.70 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-4-reasoning-plus/",
      "page": "/open-source-large-language-models/microsoft-phi-4-reasoning-plus/"
    },
    {
      "slug": "microsoft-phi-4-mini-reasoning",
      "name": "Phi-4-mini-reasoning",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-4-mini-reasoning",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 128000,
      "input_price": 0.075,
      "output_price": 0.3,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-02",
      "released": "2025-04",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "3.8B",
      "paramsTotalB": 3.8,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "reasoning",
      "best_for": "The cheapest reasoning model Microsoft sells: 3.8B with a 128K window in and out, aimed at edge/embedded math tutoring and step-by-step solvers where output length matters more than breadth.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-4-mini-reasoning costs $0.13 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $5.22 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-4-mini-reasoning/",
      "page": "/open-source-large-language-models/microsoft-phi-4-mini-reasoning/"
    },
    {
      "slug": "microsoft-phi-4-mini-flash-reasoning",
      "name": "Phi-4-mini-flash-reasoning",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-4-mini-flash-reasoning",
      "status": "ga",
      "flagship": false,
      "context_window": 65536,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-02",
      "released": "2025-06",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "3.8B",
      "paramsTotalB": 3.8,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "reasoning",
      "best_for": "Hybrid SambaY/Gated-Memory-Unit architecture giving up to ~10x higher decoding throughput than Phi-4-mini-reasoning on long generations — self-host it (or use Foundry managed compute) when tokens/sec on a single GPU is the binding constraint; there is no per-token Azure meter for it.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/microsoft/Phi-4-mini-flash-reasoning",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-4-mini-flash-reasoning/",
      "page": "/open-source-large-language-models/microsoft-phi-4-mini-flash-reasoning/"
    },
    {
      "slug": "microsoft-phi-4-reasoning-vision-15b",
      "name": "Phi-4-Reasoning-Vision-15B",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-4-Reasoning-Vision-15B",
      "status": "ga",
      "flagship": false,
      "context_window": 16384,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "reasoning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-03-04",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "15B",
      "paramsTotalB": 15,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "reasoning",
      "best_for": "Newest Phi release — a 15B multimodal reasoner with explicit <think> traces for chart/diagram/document math and GUI element localization (ScreenSpot-V2). Managed-compute or self-host only; no serverless per-token price published.",
      "notes": [
        "Managed-compute or self-host only; no serverless per-token price published."
      ],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/microsoft/Phi-4-reasoning-vision-15B",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-4-reasoning-vision-15b/",
      "page": "/open-source-large-language-models/microsoft-phi-4-reasoning-vision-15b/"
    },
    {
      "slug": "microsoft-phi-mini-moe-instruct",
      "name": "Phi-mini-MoE-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "microsoft/Phi-mini-MoE-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 4096,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2025-06-23",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "7.6B-A2.4B MoE",
      "paramsTotalB": 7.6,
      "paramsActiveB": 2.4,
      "architecture": "Mixture of experts",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "SlimMoE compression of Phi-3.5-MoE down to 7.6B total / 2.4B active — near-Phi-3.5-MoE quality at a third of the memory, but a hard 4K context kills it for RAG. Weights only, no hosted endpoint.",
      "notes": [
        "Weights only, no hosted endpoint."
      ],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/microsoft/Phi-mini-MoE-instruct",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-mini-moe-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-mini-moe-instruct/"
    },
    {
      "slug": "microsoft-phi-tiny-moe-instruct",
      "name": "Phi-tiny-MoE-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "microsoft/Phi-tiny-MoE-instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 4096,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2025-06-23",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "3.8B-A1.1B MoE",
      "paramsTotalB": 3.8,
      "paramsActiveB": 1.1,
      "architecture": "Mixture of experts",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "Smallest SlimMoE variant (1.1B active) for CPU and NPU inference where every GB counts; 4K context and an Oct-2023 cutoff make it a component model, not a chatbot. Weights only.",
      "notes": [
        "Weights only."
      ],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/microsoft/Phi-tiny-MoE-instruct",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-tiny-moe-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-tiny-moe-instruct/"
    },
    {
      "slug": "microsoft-phi-ground-any",
      "name": "Phi-Ground-Any",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "microsoft/Phi-Ground-Any",
      "status": "preview",
      "flagship": false,
      "context_window": 0,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-05-07",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "",
      "paramsTotalB": null,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "unrecorded",
      "kind": "vision",
      "best_for": "Research GUI-grounding model (Phi-3-V based) that maps a natural-language instruction to on-screen coordinates — a building block for computer-use agents, not a general chat model; specs are thinly documented and it has no hosted endpoint.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/microsoft/Phi-Ground-Any",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-ground-any/",
      "page": "/open-source-large-language-models/microsoft-phi-ground-any/"
    },
    {
      "slug": "microsoft-phi-3-5-mini-instruct",
      "name": "Phi-3.5-mini-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3.5-mini-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 4096,
      "input_price": 0.13,
      "output_price": 0.52,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2024-08-16",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "3.8B",
      "paramsTotalB": 3.8,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "Still deployable and still metered on Azure, but Microsoft has dropped it from the current Microsoft-models doc table and Phi-4-mini-instruct is both cheaper and better — migrate.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-3.5-mini-instruct costs $0.23 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $9.05 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-5-mini-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-5-mini-instruct/"
    },
    {
      "slug": "microsoft-phi-3-5-moe-instruct",
      "name": "Phi-3.5-MoE-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3.5-MoE-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 4096,
      "input_price": 0.16,
      "output_price": 0.64,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2024-08-17",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "41.9B-A6.6B MoE",
      "paramsTotalB": 41.9,
      "paramsActiveB": 6.6,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "The largest Phi ever shipped (16x3.8B MoE, 6.6B active) — historically interesting and still self-hostable under MIT, but superseded on quality-per-dollar by Phi-4 and removed from the current Foundry model list.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-3.5-MoE-instruct costs $0.28 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $11.14 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-5-moe-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-5-moe-instruct/"
    },
    {
      "slug": "microsoft-phi-3-5-vision-instruct",
      "name": "Phi-3.5-vision-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3.5-vision-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 4096,
      "input_price": 0.13,
      "output_price": 0.52,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2024-08-16",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "4.2B",
      "paramsTotalB": 4.2,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "Multi-image and video-frame reasoning at 4.2B; use Phi-4-multimodal-instruct instead unless you specifically need this checkpoint's multi-frame behaviour.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-3.5-vision-instruct costs $0.23 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $9.05 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-5-vision-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-5-vision-instruct/"
    },
    {
      "slug": "microsoft-phi-3-medium-128k-instruct",
      "name": "Phi-3-medium-128k-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3-medium-128k-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 4096,
      "input_price": 0.17,
      "output_price": 0.68,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2024-05-07",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "14B",
      "paramsTotalB": 14,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Legacy 14B long-context Phi-3; still metered and catalogued but strictly worse and pricier than Phi-4 — kept only for pinned reproducibility.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-3-medium-128k-instruct costs $0.30 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $11.83 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-medium-128k-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-medium-128k-instruct/"
    },
    {
      "slug": "microsoft-phi-3-medium-4k-instruct",
      "name": "Phi-3-medium-4k-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3-medium-4k-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 4096,
      "max_output_tokens": 4096,
      "input_price": 0.17,
      "output_price": 0.68,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2024-05-07",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "14B",
      "paramsTotalB": 14,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Short-context variant of Phi-3-medium; no reason to choose it today over Phi-4 at a lower price with a larger window.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-3-medium-4k-instruct costs $0.30 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $11.83 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-medium-4k-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-medium-4k-instruct/"
    },
    {
      "slug": "microsoft-phi-3-small-128k-instruct",
      "name": "Phi-3-small-128k-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3-small-128k-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 4096,
      "input_price": 0.15,
      "output_price": 0.6,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2024-05-07",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "7B",
      "paramsTotalB": 7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Legacy 7B long-context model; superseded on every axis by Phi-4-mini-instruct at half the price.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-3-small-128k-instruct costs $0.26 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $10.44 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-small-128k-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-small-128k-instruct/"
    },
    {
      "slug": "microsoft-phi-3-small-8k-instruct",
      "name": "Phi-3-small-8k-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3-small-8k-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 8192,
      "max_output_tokens": 4096,
      "input_price": 0.15,
      "output_price": 0.6,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2024-05-07",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "7B",
      "paramsTotalB": 7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Short-context legacy 7B; migrate to Phi-4-mini-instruct. (The Foundry catalog record misreports its window as 131072; the model card and name are authoritative at 8K.)",
      "notes": [
        "(The Foundry catalog record misreports its window as 131072; the model card and name are authoritative at 8K.)",
        "At a three-to-one input-to-output ratio, Phi-3-small-8k-instruct costs $0.26 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $10.44 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-small-8k-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-small-8k-instruct/"
    },
    {
      "slug": "microsoft-phi-3-mini-128k-instruct",
      "name": "Phi-3-mini-128k-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3-mini-128k-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 4096,
      "input_price": 0.13,
      "output_price": 0.52,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2024-04-22",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "3.8B",
      "paramsTotalB": 3.8,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "The model that launched the SLM category; still one of the most downloaded Phi checkpoints for offline/edge use, but on Azure Phi-4-mini-instruct is cheaper and stronger.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-3-mini-128k-instruct costs $0.23 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $9.05 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-mini-128k-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-mini-128k-instruct/"
    },
    {
      "slug": "microsoft-phi-3-mini-4k-instruct",
      "name": "Phi-3-mini-4k-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3-mini-4k-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 4096,
      "max_output_tokens": 4096,
      "input_price": 0.13,
      "output_price": 0.52,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "fine-tuning"
      ],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2024-04-22",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "3.8B",
      "paramsTotalB": 3.8,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "The canonical tiny Phi for phones and NPUs (also shipped as GGUF and ONNX DirectML/CUDA/CPU builds); pick it only for offline deployment where the 4K window is fine.",
      "notes": [
        "At a three-to-one input-to-output ratio, Phi-3-mini-4k-instruct costs $0.23 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $9.05 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "prices.azure.com",
          "url": "https://prices.azure.com/api/retail/prices?currencyCode=USD&amp;$filter=productName%20eq%20%27Azure%20Phi%20Models%27",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-mini-4k-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-mini-4k-instruct/"
    },
    {
      "slug": "microsoft-phi-3-vision-128k-instruct",
      "name": "Phi-3-vision-128k-instruct",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "Phi-3-vision-128k-instruct",
      "status": "deprecated",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 4096,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-03",
      "released": "2024-05-19",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "4.2B",
      "paramsTotalB": 4.2,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "Original Phi vision model, managed-compute/self-host only (no serverless per-token meter); superseded by Phi-3.5-vision-instruct and then Phi-4-multimodal-instruct.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/microsoft/Phi-3-vision-128k-instruct",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-3-vision-128k-instruct/",
      "page": "/open-source-large-language-models/microsoft-phi-3-vision-128k-instruct/"
    },
    {
      "slug": "microsoft-phi-2",
      "name": "phi-2",
      "provider": "Microsoft",
      "providerSlug": "microsoft",
      "api_id": "microsoft/phi-2",
      "status": "deprecated",
      "flagship": false,
      "context_window": 2048,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [],
      "tags": [],
      "knowledge_cutoff": "2023-10",
      "released": "2023-12-13",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "2.7B",
      "paramsTotalB": 2.7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "Base (non-instruct) 2.7B research model, still the most-downloaded Microsoft checkpoint on Hugging Face and a common fine-tuning starting point — but a 2K context and no chat tuning make it unsuitable for production assistants.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/microsoft/phi-2",
          "primary": true
        },
        {
          "label": "Microsoft pricing",
          "url": "https://azure.microsoft.com/en-us/pricing/details/phi-3/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/microsoft-phi-2/",
      "page": "/open-source-large-language-models/microsoft-phi-2/"
    },
    {
      "slug": "cohere-command-a",
      "name": "Command A+",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-a-plus-05-2026",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 64000,
      "input_price": 0,
      "output_price": 0,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "vision",
        "reasoning",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-05-20",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "218B total / 25B active MoE",
      "paramsTotalB": 218,
      "paramsActiveB": 25,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "Cohere's flagship: the only model that folds vision, reasoning, agentic tool use and 48-language coverage into one set of weights, and it runs on 1x B200 or 2x H100 — but the API is free only under trial-grade limits (20 req/min, 1,000 calls/month), so real production means Apache-2.0 self-hosting or a sales contract.",
      "notes": [
        "At a three-to-one input-to-output ratio, Command A+ costs $0.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $0.00 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-a/",
      "page": "/open-source-large-language-models/cohere-command-a/"
    },
    {
      "slug": "cohere-command-a-2",
      "name": "Command A",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-a-03-2025",
      "status": "ga",
      "flagship": true,
      "context_window": 256000,
      "max_output_tokens": 8000,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-03",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "111B",
      "paramsTotalB": 111,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "The long-context (256K) dense workhorse for RAG, tool use and 23-language agents, and the only Command A variant with a real 500 req/min production rate limit — but Cohere pulled its list price off the public pricing page, so budget via sales.",
      "notes": [],
      "sources": [
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-a-2/",
      "page": "/open-source-large-language-models/cohere-command-a-2/"
    },
    {
      "slug": "cohere-command-a-reasoning",
      "name": "Command A Reasoning",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-a-reasoning-08-2025",
      "status": "ga",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 32000,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "reasoning",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-08",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "111B",
      "paramsTotalB": 111,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "large",
      "kind": "reasoning",
      "best_for": "Pick when you need an explicit thinking budget on nuanced multi-step or agentic problems in 23 languages and can host it (4x H100 for production); no public price and production API access is contact-sales only.",
      "notes": [],
      "sources": [
        {
          "label": "docs.cohere.com",
          "url": "https://docs.cohere.com/docs/rate-limits",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-a-reasoning/",
      "page": "/open-source-large-language-models/cohere-command-a-reasoning/"
    },
    {
      "slug": "cohere-command-a-translate",
      "name": "Command A Translate",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-a-translate-08-2025",
      "status": "ga",
      "flagship": false,
      "context_window": 8000,
      "max_output_tokens": 8000,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-08",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "111B",
      "paramsTotalB": 111,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "A dedicated 23-language machine-translation model for regulated shops that must translate sensitive documents inside their own perimeter; the 8K input cap means you chunk long documents yourself.",
      "notes": [],
      "sources": [
        {
          "label": "docs.cohere.com",
          "url": "https://docs.cohere.com/docs/rate-limits",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-a-translate/",
      "page": "/open-source-large-language-models/cohere-command-a-translate/"
    },
    {
      "slug": "cohere-command-a-vision",
      "name": "Command A Vision",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-a-vision-07-2025",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 8000,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-07",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "111B",
      "paramsTotalB": 111,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "large",
      "kind": "vision",
      "best_for": "Document/chart/OCR understanding with up to 20 images per request — but it does NOT support tool use, so for agentic multimodal work go to Command A+ instead.",
      "notes": [],
      "sources": [
        {
          "label": "docs.cohere.com",
          "url": "https://docs.cohere.com/docs/rate-limits",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-a-vision/",
      "page": "/open-source-large-language-models/cohere-command-a-vision/"
    },
    {
      "slug": "cohere-command-r-08-2024",
      "name": "Command R+ (08-2024)",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-r-plus-08-2024",
      "status": "ga",
      "flagship": true,
      "context_window": 128000,
      "max_output_tokens": 4000,
      "input_price": 2.5,
      "output_price": 10,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2024-08",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "104B",
      "paramsTotalB": 104,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "Still Live with a 500 req/min production limit, but at $2.50/$10.00 it is priced under the 'existing customers' FAQ and comprehensively beaten by Command A on quality, context and throughput — migrate rather than start here.",
      "notes": [
        "At a three-to-one input-to-output ratio, Command R+ (08-2024) costs $4.38 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $174.00 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-r-08-2024/",
      "page": "/open-source-large-language-models/cohere-command-r-08-2024/"
    },
    {
      "slug": "cohere-command-r-08-2024-2",
      "name": "Command R (08-2024)",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-r-08-2024",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 4000,
      "input_price": 0.15,
      "output_price": 0.6,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2024-08",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "32B",
      "paramsTotalB": 32,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "The cheapest Cohere model with a published price, a real 500 req/min production limit and full tool use — the practical default for high-volume RAG and single-step tool calling when you want a price you can actually see.",
      "notes": [
        "At a three-to-one input-to-output ratio, Command R (08-2024) costs $0.26 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $10.44 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-r-08-2024-2/",
      "page": "/open-source-large-language-models/cohere-command-r-08-2024-2/"
    },
    {
      "slug": "cohere-command-r7b",
      "name": "Command R7B",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-r7b-12-2024",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 4000,
      "input_price": 0.0375,
      "output_price": 0.15,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2024-12",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "8B",
      "paramsTotalB": 8,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Cohere's cheapest served model by a wide margin ($0.0375/1M in) with a 128K window — right for latency-sensitive chatbots, classification and on-device/consumer-GPU deployment where 4K output is enough.",
      "notes": [
        "At a three-to-one input-to-output ratio, Command R7B costs $0.07 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $2.61 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-r7b/",
      "page": "/open-source-large-language-models/cohere-command-r7b/"
    },
    {
      "slug": "cohere-command-r-03-2024",
      "name": "Command R (03-2024)",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-r-03-2024",
      "status": "deprecated",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 4000,
      "input_price": 0.5,
      "output_price": 1.5,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2024-03",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "35B",
      "paramsTotalB": 35,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "Deprecated 2025-09-15 and 3.3x the price of command-r-08-2024 for worse results; the alias `command-r` also points here. No reason to choose it — migrate to command-r-08-2024.",
      "notes": [
        "No reason to choose it — migrate to command-r-08-2024.",
        "At a three-to-one input-to-output ratio, Command R (03-2024) costs $0.75 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $29.85 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-r-03-2024/",
      "page": "/open-source-large-language-models/cohere-command-r-03-2024/"
    },
    {
      "slug": "cohere-command-r-04-2024",
      "name": "Command R+ (04-2024)",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "command-r-plus-04-2024",
      "status": "deprecated",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 4000,
      "input_price": 3,
      "output_price": 15,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "streaming",
        "structured-output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2024-04",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "104B",
      "paramsTotalB": 104,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "Deprecated 2025-09-15 and the most expensive model Cohere still serves; the alias `command-r-plus` resolves here. Move to command-r-plus-08-2024 for a 17%/33% price cut, or Command A.",
      "notes": [
        "Move to command-r-plus-08-2024 for a 17%/33% price cut, or Command A.",
        "At a three-to-one input-to-output ratio, Command R+ (04-2024) costs $6.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $238.50 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-command-r-04-2024/",
      "page": "/open-source-large-language-models/cohere-command-r-04-2024/"
    },
    {
      "slug": "cohere-north-mini-code-1-0",
      "name": "North Mini Code 1.0",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "",
      "status": "ga",
      "flagship": false,
      "context_window": 0,
      "max_output_tokens": 0,
      "input_price": 0,
      "output_price": 0,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tools",
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-06",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "30B total / 3B active MoE",
      "paramsTotalB": 30,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "coding",
      "best_for": "Cohere's first agentic coding model: Apache-2.0, free API key plus free weights, and a 3B active footprint that runs locally — aimed at repo-level SWE-agent/OpenCode style harnesses and terminal agents. Context window is not published; exact API model string is not documented (docs page: /docs/north-mini-code-1.0).",
      "notes": [
        "Context window is not published; exact API model string is not documented (docs page: /docs/north-mini-code-1.0).",
        "At a three-to-one input-to-output ratio, North Mini Code 1.0 costs $0.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $0.00 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-north-mini-code-1-0/",
      "page": "/open-source-large-language-models/cohere-north-mini-code-1-0/"
    },
    {
      "slug": "cohere-cohere-transcribe",
      "name": "Cohere Transcribe",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "cohere-transcribe-03-2026",
      "status": "ga",
      "flagship": false,
      "context_window": 0,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "audio"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-03",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "2B",
      "paramsTotalB": 2,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "speech",
      "best_for": "Open (Apache-2.0) 2B Conformer ASR covering 14 languages with a real-time factor up to 3x faster than similar-size models; no timestamps, no diarization and no language auto-detect, so pin the language. Free on the API but capped at 5 req/min and 25MB per file; production is per-instance on Model Vault from $3.75/hr.",
      "notes": [
        "Free on the API but capped at 5 req/min and 25MB per file; production is per-instance on Model Vault from $3.75/hr."
      ],
      "sources": [
        {
          "label": "docs.cohere.com",
          "url": "https://docs.cohere.com/docs/transcribe",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-cohere-transcribe/",
      "page": "/open-source-large-language-models/cohere-cohere-transcribe/"
    },
    {
      "slug": "cohere-cohere-transcribe-arabic",
      "name": "Cohere Transcribe Arabic",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "cohere-transcribe-arabic-07-2026",
      "status": "ga",
      "flagship": false,
      "context_window": 0,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "audio"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-07",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "2B",
      "paramsTotalB": 2,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "speech",
      "best_for": "Arabic-specialised fine-tune of Cohere Transcribe — use it over the base model for any Arabic audio; same 25MB file cap, and it is not yet on Bedrock/Azure/Oracle. Per-token price not applicable; Model Vault instance pricing applies.",
      "notes": [
        "Per-token price not applicable; Model Vault instance pricing applies."
      ],
      "sources": [
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-cohere-transcribe-arabic/",
      "page": "/open-source-large-language-models/cohere-cohere-transcribe-arabic/"
    },
    {
      "slug": "cohere-aya-expanse-32b",
      "name": "Aya Expanse 32B",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "c4ai-aya-expanse-32b",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 4000,
      "input_price": 0.5,
      "output_price": 1.5,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2024-10",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "32B",
      "paramsTotalB": 32,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "Research-grade 23-language model with a 128K window at a flat $0.50/$1.50 — the cheapest way to test Cohere-family multilingual quality, but CC-BY-NC weights and a research positioning make it a poor commercial default versus Command R.",
      "notes": [
        "At a three-to-one input-to-output ratio, Aya Expanse 32B costs $0.75 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $29.85 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-aya-expanse-32b/",
      "page": "/open-source-large-language-models/cohere-aya-expanse-32b/"
    },
    {
      "slug": "cohere-aya-vision-32b",
      "name": "Aya Vision 32B",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "c4ai-aya-vision-32b",
      "status": "ga",
      "flagship": false,
      "context_window": 16000,
      "max_output_tokens": 4000,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-03",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "32B",
      "paramsTotalB": 32,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "medium",
      "kind": "vision",
      "best_for": "Open multilingual vision-language research model across 23 languages; the pricing FAQ covers only Aya Expanse, so its API price is unstated — for commercial multimodal work use Command A Vision or Command A+.",
      "notes": [],
      "sources": [
        {
          "label": "docs.cohere.com",
          "url": "https://docs.cohere.com/docs/aya-vision",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-aya-vision-32b/",
      "page": "/open-source-large-language-models/cohere-aya-vision-32b/"
    },
    {
      "slug": "cohere-tiny-aya-global",
      "name": "Tiny Aya Global",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "tiny-aya-global",
      "status": "ga",
      "flagship": false,
      "context_window": 8000,
      "max_output_tokens": 8000,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "3.35B",
      "paramsTotalB": 3.35,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "The balanced 3.35B/70-language Tiny Aya variant — the one to start with when you want broad low-resource language coverage on small hardware (GGUF builds published). No public API price.",
      "notes": [
        "No public API price."
      ],
      "sources": [
        {
          "label": "docs.cohere.com",
          "url": "https://docs.cohere.com/docs/tiny-aya",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-tiny-aya-global/",
      "page": "/open-source-large-language-models/cohere-tiny-aya-global/"
    },
    {
      "slug": "cohere-tiny-aya-earth",
      "name": "Tiny Aya Earth",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "tiny-aya-earth",
      "status": "ga",
      "flagship": false,
      "context_window": 8000,
      "max_output_tokens": 8000,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "3.35B",
      "paramsTotalB": 3.35,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "Region-specialised Tiny Aya tuned for West Asian and African languages; choose it over Tiny Aya Global only when your traffic is concentrated in that region. No public API price.",
      "notes": [
        "No public API price."
      ],
      "sources": [
        {
          "label": "docs.cohere.com",
          "url": "https://docs.cohere.com/docs/tiny-aya",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-tiny-aya-earth/",
      "page": "/open-source-large-language-models/cohere-tiny-aya-earth/"
    },
    {
      "slug": "cohere-tiny-aya-fire",
      "name": "Tiny Aya Fire",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "tiny-aya-fire",
      "status": "ga",
      "flagship": false,
      "context_window": 8000,
      "max_output_tokens": 8000,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "3.35B",
      "paramsTotalB": 3.35,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "Region-specialised Tiny Aya tuned for South Asian languages; worth the swap from Tiny Aya Global for Indic-heavy workloads. No public API price.",
      "notes": [
        "No public API price."
      ],
      "sources": [
        {
          "label": "docs.cohere.com",
          "url": "https://docs.cohere.com/docs/tiny-aya",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-tiny-aya-fire/",
      "page": "/open-source-large-language-models/cohere-tiny-aya-fire/"
    },
    {
      "slug": "cohere-tiny-aya-water",
      "name": "Tiny Aya Water",
      "provider": "Cohere",
      "providerSlug": "cohere",
      "api_id": "tiny-aya-water",
      "status": "ga",
      "flagship": false,
      "context_window": 8000,
      "max_output_tokens": 8000,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "streaming"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02",
      "license": "CC-BY-NC-4.0",
      "licenseKind": "non-commercial",
      "parameters": "3.35B",
      "paramsTotalB": 3.35,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "Region-specialised Tiny Aya tuned for European and Asia-Pacific languages; the pick for EU/APAC-focused small-model deployments. No public API price.",
      "notes": [
        "No public API price."
      ],
      "sources": [
        {
          "label": "docs.cohere.com",
          "url": "https://docs.cohere.com/docs/tiny-aya",
          "primary": true
        },
        {
          "label": "Cohere pricing",
          "url": "https://cohere.com/pricing",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.cohere.com/docs/models",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/cohere-tiny-aya-water/",
      "page": "/open-source-large-language-models/cohere-tiny-aya-water/"
    },
    {
      "slug": "chinese-labs-kimi-k3",
      "name": "Kimi K3",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "kimi-k3",
      "status": "ga",
      "flagship": true,
      "context_window": 1048576,
      "max_output_tokens": -1,
      "input_price": 2.98,
      "output_price": 14.9,
      "cached_input_price": 0.3,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "agentic tool use",
        "vision",
        "function calling",
        "context caching",
        "OpenAI-compatible API",
        "Anthropic-compatible API"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-06",
      "license": "Kimi K3 License (custom, license_name kimi-k3 on Hugging Face)",
      "licenseKind": "custom",
      "parameters": "2.8T total / 104B active MoE (896 experts, 16 active)",
      "paramsTotalB": 2800,
      "paramsActiveB": 104,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "Frontier long-context agent work where you can hit cache: the 10x cache-hit discount ($0.30 vs $3.00) matters far more than the headline rate, and $15 output makes it expensive for chatty workloads.",
      "notes": [
        "At a three-to-one input-to-output ratio, Kimi K3 costs $5.96 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $236.91 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "platform.kimi.com",
          "url": "https://platform.kimi.com/docs/pricing/chat-k3",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-kimi-k3/",
      "page": "/open-source-large-language-models/chinese-labs-kimi-k3/"
    },
    {
      "slug": "chinese-labs-kimi-k2-7-code",
      "name": "Kimi K2.7 Code",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "kimi-k2.7-code",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 0,
      "input_price": 0.95,
      "output_price": 4,
      "cached_input_price": 0.19,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "coding",
        "long-horizon agentic coding",
        "function calling",
        "context caching"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-06",
      "license": "Modified MIT",
      "licenseKind": "permissive",
      "parameters": "1T total / 32B active MoE",
      "paramsTotalB": 1000,
      "paramsActiveB": 32,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "coding",
      "best_for": "The value pick for autonomous coding agents — roughly a third of K3's input cost and a quarter of its output cost, with 256k context and weights you can self-host under Modified MIT.",
      "notes": [
        "At a three-to-one input-to-output ratio, Kimi K2.7 Code costs $1.71 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $68.10 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "platform.kimi.ai",
          "url": "https://platform.kimi.ai/docs/pricing/chat-k27-code.md",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-kimi-k2-7-code/",
      "page": "/open-source-large-language-models/chinese-labs-kimi-k2-7-code/"
    },
    {
      "slug": "chinese-labs-kimi-k2-7-code-highspeed",
      "name": "Kimi K2.7 Code Highspeed",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "kimi-k2.7-code-highspeed",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 0,
      "input_price": 1.9,
      "output_price": 8,
      "cached_input_price": 0.38,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "coding",
        "high-throughput serving (~180 tok/s, up to 260 on short contexts)"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "",
      "license": "Modified MIT",
      "licenseKind": "permissive",
      "parameters": "1T total / 32B active MoE",
      "paramsTotalB": 1000,
      "paramsActiveB": 32,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "coding",
      "best_for": "Same weights as kimi-k2.7-code at exactly 2x the price — only worth it when interactive latency, not cost, is the constraint.",
      "notes": [
        "At a three-to-one input-to-output ratio, Kimi K2.7 Code Highspeed costs $3.42 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $136.20 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "platform.kimi.ai",
          "url": "https://platform.kimi.ai/docs/pricing/chat-k27-code.md",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-kimi-k2-7-code-highspeed/",
      "page": "/open-source-large-language-models/chinese-labs-kimi-k2-7-code-highspeed/"
    },
    {
      "slug": "chinese-labs-kimi-k2-6",
      "name": "Kimi K2.6",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "kimi-k2.6",
      "status": "ga",
      "flagship": false,
      "context_window": 262144,
      "max_output_tokens": 0,
      "input_price": 0.97,
      "output_price": 4.02,
      "cached_input_price": 0.16,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "thinking and non-thinking modes",
        "agentic tasks",
        "function calling"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02",
      "license": "Modified MIT",
      "licenseKind": "permissive",
      "parameters": "1T total / 32B active MoE + 400M MoonViT vision encoder",
      "paramsTotalB": 1000,
      "paramsActiveB": 32,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "Cheapest multimodal option in the Kimi line — pick it over K2.7 Code when you need image/video input and over K3 when 256k context is enough.",
      "notes": [
        "At a three-to-one input-to-output ratio, Kimi K2.6 costs $1.73 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $68.90 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "platform.kimi.com",
          "url": "https://platform.kimi.com/docs/pricing/chat-k26.md",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-kimi-k2-6/",
      "page": "/open-source-large-language-models/chinese-labs-kimi-k2-6/"
    },
    {
      "slug": "chinese-labs-glm-5-3",
      "name": "GLM-5.3",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "glm-5.3",
      "status": "ga",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "input_price": 1.4,
      "output_price": 4.4,
      "cached_input_price": 0.26,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "always-on reasoning (low/high/max effort)",
        "function calling",
        "context caching",
        "structured output",
        "MCP",
        "OpenAI + Anthropic protocols"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-25",
      "license": "GLM-5.3 License (custom; license_name glm-5.3 on Hugging Face)",
      "licenseKind": "custom",
      "parameters": "753B",
      "paramsTotalB": 753,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Best frontier-class price/context ratio here — 1M context at $1.40 in / $4.40 out, but note reasoning cannot be disabled, so budget for thinking tokens on the output side.",
      "notes": [
        "At a three-to-one input-to-output ratio, GLM-5.3 costs $2.15 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $85.56 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.z.ai",
          "url": "https://docs.z.ai/guides/overview/pricing",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-glm-5-3/",
      "page": "/open-source-large-language-models/chinese-labs-glm-5-3/"
    },
    {
      "slug": "chinese-labs-glm-5-3-flash",
      "name": "GLM-5.3-Flash",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "glm-5.3-flash",
      "status": "ga",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "input_price": 0.075,
      "output_price": 0.25,
      "cached_input_price": 0.015,
      "modalities_in": [
        "text",
        "image",
        "video",
        "file"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "native multimodal",
        "reasoning",
        "function calling",
        "context caching"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-25",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "320B total / 18B active MoE (sparse + linear hybrid attention)",
      "paramsTotalB": 320,
      "paramsActiveB": 18,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "The cheapest capable multimodal model in this entire report and plain MIT weights — but the posted rate is a 50%-off promo running to 9 Sep 2026, so model your budget on double it.",
      "notes": [
        "At a three-to-one input-to-output ratio, GLM-5.3-Flash costs $0.12 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $4.72 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.z.ai",
          "url": "https://docs.z.ai/guides/overview/pricing",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-glm-5-3-flash/",
      "page": "/open-source-large-language-models/chinese-labs-glm-5-3-flash/"
    },
    {
      "slug": "chinese-labs-glm-5-2",
      "name": "GLM-5.2",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "glm-5.2",
      "status": "ga",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 131072,
      "input_price": 1.4,
      "output_price": 4.4,
      "cached_input_price": 0.26,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "thinking mode (toggleable)",
        "function calling",
        "context caching",
        "structured output",
        "MCP"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-06-16",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "753B",
      "paramsTotalB": 753,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Same base model and same price as GLM-5.3 but under plain MIT and with thinking optionally off — the one to self-host, or to use when you need non-reasoning responses.",
      "notes": [
        "At a three-to-one input-to-output ratio, GLM-5.2 costs $2.15 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $85.56 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.z.ai",
          "url": "https://docs.z.ai/guides/overview/pricing",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-glm-5-2/",
      "page": "/open-source-large-language-models/chinese-labs-glm-5-2/"
    },
    {
      "slug": "chinese-labs-glm-5",
      "name": "GLM-5",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "glm-5",
      "status": "ga",
      "flagship": false,
      "context_window": 200000,
      "max_output_tokens": 131072,
      "input_price": 1,
      "output_price": 3.2,
      "cached_input_price": 0.2,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "thinking mode",
        "function calling",
        "context caching",
        "structured output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02-11",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "754B",
      "paramsTotalB": 754,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Cheaper than GLM-5.2/5.3 if 200k context is enough; otherwise the newer siblings are worth the extra 40 cents per million input.",
      "notes": [
        "At a three-to-one input-to-output ratio, GLM-5 costs $1.55 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $61.68 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.z.ai",
          "url": "https://docs.z.ai/guides/overview/pricing",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-glm-5/",
      "page": "/open-source-large-language-models/chinese-labs-glm-5/"
    },
    {
      "slug": "chinese-labs-glm-4-7",
      "name": "GLM-4.7",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "glm-4.7",
      "status": "ga",
      "flagship": false,
      "context_window": 200000,
      "max_output_tokens": 131072,
      "input_price": 0.6,
      "output_price": 2.2,
      "cached_input_price": 0.11,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "thinking mode",
        "function calling",
        "context caching",
        "structured output"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "",
      "paramsTotalB": null,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "unrecorded",
      "kind": "language",
      "best_for": "Solid mid-tier coding/agent workhorse at under half GLM-5.2's price; step down to it when you don't need million-token context.",
      "notes": [
        "At a three-to-one input-to-output ratio, GLM-4.7 costs $1.00 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $39.78 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.z.ai",
          "url": "https://docs.z.ai/guides/overview/pricing",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-glm-4-7/",
      "page": "/open-source-large-language-models/chinese-labs-glm-4-7/"
    },
    {
      "slug": "chinese-labs-glm-4-5-air",
      "name": "GLM-4.5-Air",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "glm-4.5-air",
      "status": "ga",
      "flagship": false,
      "context_window": -1,
      "max_output_tokens": 0,
      "input_price": 0.2,
      "output_price": 1.1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "function calling",
        "agentic tasks"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "",
      "paramsTotalB": null,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "unrecorded",
      "kind": "language",
      "best_for": "Legacy small model kept alive for existing integrations — new builds should start on GLM-4.7-FlashX or GLM-5.3-Flash instead.",
      "notes": [
        "At a three-to-one input-to-output ratio, GLM-4.5-Air costs $0.43 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $16.89 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.z.ai",
          "url": "https://docs.z.ai/guides/overview/pricing",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-glm-4-5-air/",
      "page": "/open-source-large-language-models/chinese-labs-glm-4-5-air/"
    },
    {
      "slug": "chinese-labs-minimax-m3",
      "name": "MiniMax-M3",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "MiniMax-M3",
      "status": "ga",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 0,
      "input_price": 0.3,
      "output_price": 1.2,
      "cached_input_price": 0.06,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "agentic tool use",
        "coding",
        "1M context",
        "prompt caching",
        "OpenAI-compatible API",
        "Anthropic-compatible API"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-06",
      "license": "MiniMax Model License (license_name minimax-community)",
      "licenseKind": "custom",
      "parameters": "427B total / ~23B active MoE",
      "paramsTotalB": 427,
      "paramsActiveB": 23,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "vision",
      "best_for": "Outstanding cost-per-context: 1M window at $0.30/$1.20 with open weights — just watch the tier break, since anything over 512k input bills at double.",
      "notes": [
        "At a three-to-one input-to-output ratio, MiniMax-M3 costs $0.52 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $20.88 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "platform.minimax.io",
          "url": "https://platform.minimax.io/docs/guides/pricing-paygo",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-minimax-m3/",
      "page": "/open-source-large-language-models/chinese-labs-minimax-m3/"
    },
    {
      "slug": "chinese-labs-minimax-m2-7",
      "name": "MiniMax-M2.7",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "MiniMax-M2.7",
      "status": "ga",
      "flagship": false,
      "context_window": 204800,
      "max_output_tokens": 0,
      "input_price": 0.3,
      "output_price": 1.2,
      "cached_input_price": 0.06,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "agentic tool use",
        "coding",
        "prompt caching"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-04",
      "license": "MiniMax Model License (custom; see LICENSE on Hugging Face)",
      "licenseKind": "custom",
      "parameters": "229B",
      "paramsTotalB": 229,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Same price as M3 with a fifth of the context and no vision — only choose it if you have already validated against these exact weights.",
      "notes": [
        "At a three-to-one input-to-output ratio, MiniMax-M2.7 costs $0.52 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $20.88 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "platform.minimax.io",
          "url": "https://platform.minimax.io/docs/guides/pricing-paygo",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-minimax-m2-7/",
      "page": "/open-source-large-language-models/chinese-labs-minimax-m2-7/"
    },
    {
      "slug": "chinese-labs-minimax-m2-7-highspeed",
      "name": "MiniMax-M2.7-highspeed",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "MiniMax-M2.7-highspeed",
      "status": "ga",
      "flagship": false,
      "context_window": 204800,
      "max_output_tokens": 0,
      "input_price": 0.6,
      "output_price": 2.4,
      "cached_input_price": 0.06,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "low-latency serving",
        "prompt caching"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "",
      "license": "MiniMax Model License (custom; see LICENSE on Hugging Face)",
      "licenseKind": "custom",
      "parameters": "229B",
      "paramsTotalB": 229,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "A 2x latency surcharge on identical weights — justify it with a measured p95 requirement, not a hunch.",
      "notes": [
        "At a three-to-one input-to-output ratio, MiniMax-M2.7-highspeed costs $1.05 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $41.76 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "platform.minimax.io",
          "url": "https://platform.minimax.io/docs/guides/pricing-paygo",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-minimax-m2-7-highspeed/",
      "page": "/open-source-large-language-models/chinese-labs-minimax-m2-7-highspeed/"
    },
    {
      "slug": "chinese-labs-hunyuan-a13b",
      "name": "hunyuan-a13b",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "hunyuan-a13b",
      "status": "ga",
      "flagship": false,
      "context_window": 224000,
      "max_output_tokens": 32768,
      "input_price": 0.074,
      "output_price": 0.297,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "hybrid fast/slow reasoning",
        "long-document understanding",
        "math",
        "function calling"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "",
      "license": "Tencent Hunyuan A13B Community License",
      "licenseKind": "custom",
      "parameters": "80B total / 13B active MoE",
      "paramsTotalB": 80,
      "paramsActiveB": 13,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "Absurdly cheap 224k-context reasoning (¥0.5/¥2 per 1M = $0.07/$0.30) — the budget choice for bulk long-document work if you can live with a Chinese-cloud endpoint.",
      "notes": [
        "At a three-to-one input-to-output ratio, hunyuan-a13b costs $0.13 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $5.16 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "cloud.tencent.com",
          "url": "https://cloud.tencent.com/document/product/1729/97731",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-hunyuan-a13b/",
      "page": "/open-source-large-language-models/chinese-labs-hunyuan-a13b/"
    },
    {
      "slug": "chinese-labs-hunyuan-hy4-preview",
      "name": "Hunyuan Hy4-preview",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "",
      "status": "preview",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "agentic tasks",
        "1M context"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "770B total / 49B active MoE",
      "paramsTotalB": 770,
      "paramsActiveB": 49,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Open weights only — Tencent sells no first-party per-token endpoint for it, so budget for GPUs (770B params) or a third-party host; the Apache-2.0 licence makes it the most commercially permissive frontier model here.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/tencent/Hy4-preview",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-hunyuan-hy4-preview/",
      "page": "/open-source-large-language-models/chinese-labs-hunyuan-hy4-preview/"
    },
    {
      "slug": "chinese-labs-hunyuan-hy3",
      "name": "Hunyuan Hy3",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "",
      "status": "ga",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "coding",
        "agentic tasks"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-07",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "295B total / 21B active MoE (192 experts, top-8)",
      "paramsTotalB": 295,
      "paramsActiveB": 21,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Apache-2.0, 21B active — the most self-hostable strong model in this report, and the practical Tencent choice when Hy4-preview's 770B is too big for your cluster. No first-party API.",
      "notes": [
        "No first-party API."
      ],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/tencent/Hy3",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-hunyuan-hy3/",
      "page": "/open-source-large-language-models/chinese-labs-hunyuan-hy3/"
    },
    {
      "slug": "chinese-labs-ernie-4-5-21b-a3b-thinking",
      "name": "ERNIE-4.5-21B-A3B-Thinking",
      "provider": "Chinese AI labs",
      "providerSlug": "chinese-labs",
      "api_id": "",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "long context"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "",
      "license": "Apache-2.0",
      "licenseKind": "permissive",
      "parameters": "21B total / 3B active MoE (64 text experts, 6 active + 2 shared, 28 layers)",
      "paramsTotalB": 21,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "reasoning",
      "best_for": "Small Apache-2.0 reasoning model that runs on a single modern GPU; not a line item in Qianfan's price table, so treat it as self-host-only.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/baidu/ERNIE-4.5-21B-A3B-Thinking",
          "primary": true
        },
        {
          "label": "Chinese AI labs pricing",
          "url": "https://platform.kimi.ai/docs/pricing/chat · https://docs.z.ai/guides/overview/pricing · https://platform.minimax.io/docs/guides/pricing-paygo · https://cloud.tencent.com/document/product/1729/97731 · https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://platform.kimi.ai/docs · https://docs.z.ai/ · https://platform.minimaxi.com/docs · https://docs.byteplus.com/en/docs/ModelArk · https://cloud.tencent.com/document/product/1729 · https://cloud.baidu.com/doc/qianfan · https://www.xfyun.cn/doc/spark/Web.html",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/chinese-labs-ernie-4-5-21b-a3b-thinking/",
      "page": "/open-source-large-language-models/chinese-labs-ernie-4-5-21b-a3b-thinking/"
    },
    {
      "slug": "other-labs-jamba-large-1-7",
      "name": "Jamba Large 1.7",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "jamba-large-1.7",
      "status": "ga",
      "flagship": true,
      "context_window": 262144,
      "max_output_tokens": 4096,
      "input_price": 2,
      "output_price": 8,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "long-context",
        "tool-use",
        "grounding",
        "json-mode",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-08-22",
      "released": "2025-07-02",
      "license": "Jamba Open Model License",
      "licenseKind": "custom",
      "parameters": "398B total / 94B active",
      "paramsTotalB": 398,
      "paramsActiveB": 94,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Long-document grounded QA where you need a 256K window and citations-faithful answers on a budget; the hybrid SSM design keeps long-context cost far below dense frontier models.",
      "notes": [
        "At a three-to-one input-to-output ratio, Jamba Large 1.7 costs $3.50 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $139.20 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-jamba-large-1-7/",
      "page": "/open-source-large-language-models/other-labs-jamba-large-1-7/"
    },
    {
      "slug": "other-labs-jamba-mini-1-7",
      "name": "Jamba Mini 1.7",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "jamba-mini-1.7",
      "status": "retired",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 4096,
      "input_price": 0.2,
      "output_price": 0.4,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "long-context",
        "tool-use",
        "grounding",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-08-22",
      "released": "2025-07-02",
      "license": "Jamba Open Model License",
      "licenseKind": "custom",
      "parameters": "52B total / 12B active",
      "paramsTotalB": 52,
      "paramsActiveB": 12,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "High-volume RAG and summarisation over long inputs; one of the cheapest 256K-context commercial endpoints available.",
      "notes": [
        "At a three-to-one input-to-output ratio, Jamba Mini 1.7 costs $0.25 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $9.96 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-jamba-mini-1-7/",
      "page": "/open-source-large-language-models/other-labs-jamba-mini-1-7/"
    },
    {
      "slug": "other-labs-jamba2-mini",
      "name": "Jamba2 Mini",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "ai21labs/AI21-Jamba2-Mini",
      "status": "ga",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "long-context",
        "tool-use",
        "grounding"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "52B total / 12B active",
      "paramsTotalB": 52,
      "paramsActiveB": 12,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "Self-hosted enterprise QA where you want 256K context and Apache-2.0 freedom; answers without the token overhead of a reasoning model. Not on AI21's paid price list.",
      "notes": [
        "Not on AI21's paid price list."
      ],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/ai21labs/AI21-Jamba2-Mini",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-jamba2-mini/",
      "page": "/open-source-large-language-models/other-labs-jamba2-mini/"
    },
    {
      "slug": "other-labs-jamba2-3b",
      "name": "Jamba2 3B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "ai21labs/AI21-Jamba2-3B",
      "status": "ga",
      "flagship": false,
      "context_window": 256000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "long-context",
        "on-device",
        "rag"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "3B",
      "paramsTotalB": 3,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "On-device RAG on iOS/Android/desktop when you need an unusually large 256K window from a 3B model.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/ai21labs/AI21-Jamba2-3B",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-jamba2-3b/",
      "page": "/open-source-large-language-models/other-labs-jamba2-3b/"
    },
    {
      "slug": "other-labs-jamba-reasoning-3b",
      "name": "Jamba Reasoning 3B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "ai21labs/AI21-Jamba-Reasoning-3B",
      "status": "ga",
      "flagship": false,
      "context_window": -1,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "3B",
      "paramsTotalB": 3,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "reasoning",
      "best_for": "Local reasoning at 3B scale; evaluate against Qwen and LFM2.5 thinking models before committing, since AI21 publishes no hosted endpoint for it.",
      "notes": [],
      "sources": [
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": true
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-jamba-reasoning-3b/",
      "page": "/open-source-large-language-models/other-labs-jamba-reasoning-3b/"
    },
    {
      "slug": "other-labs-reka-edge-reka-edge-2603",
      "name": "Reka Edge (reka-edge-2603)",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "reka-edge",
      "status": "ga",
      "flagship": false,
      "context_window": -1,
      "max_output_tokens": 0,
      "input_price": 0.1,
      "output_price": 0.1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "multimodal",
        "object-detection",
        "tool-use",
        "on-device"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "",
      "license": "reka-edge-2603-license",
      "licenseKind": "custom",
      "parameters": "7B",
      "paramsTotalB": 7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "Cheapest way to bolt image/video understanding onto a high-volume pipeline, and the weights are downloadable if you'd rather run it yourself; check the bespoke licence before commercial use.",
      "notes": [
        "At a three-to-one input-to-output ratio, Reka Edge (reka-edge-2603) costs $0.10 per million tokens blended. A workload of one million input and 330,000 output tokens per day would run about $3.99 per month at list price, before caching or batch discounts."
      ],
      "sources": [
        {
          "label": "docs.reka.ai",
          "url": "https://docs.reka.ai/pricing.md",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-reka-edge-reka-edge-2603/",
      "page": "/open-source-large-language-models/other-labs-reka-edge-reka-edge-2603/"
    },
    {
      "slug": "other-labs-olmo-3-1-32b-think",
      "name": "Olmo 3.1 32B Think",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "allenai/Olmo-3.1-32B-Think",
      "status": "ga",
      "flagship": false,
      "context_window": 65536,
      "max_output_tokens": 32768,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "long-cot",
        "math",
        "code"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-12",
      "released": "2025-12-23",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "32B",
      "paramsTotalB": 32,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "medium",
      "kind": "reasoning",
      "best_for": "The strongest fully-reproducible reasoning model: choose it when you must audit or re-derive the training pipeline, not when you need best-in-class benchmark scores.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/allenai/Olmo-3.1-32B-Think",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-olmo-3-1-32b-think/",
      "page": "/open-source-large-language-models/other-labs-olmo-3-1-32b-think/"
    },
    {
      "slug": "other-labs-olmo-3-1-32b-instruct",
      "name": "Olmo 3.1 32B Instruct",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "allenai/Olmo-3.1-32B-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 65536,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "instruction-following"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-12",
      "released": "2025-12",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "32B",
      "paramsTotalB": 32,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "General chat variant of the fully-open 32B; Ai2 flags it as intended for research and educational use, so read the Responsible Use Guidelines before shipping it in a product.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/allenai/Olmo-3.1-32B-Instruct",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-olmo-3-1-32b-instruct/",
      "page": "/open-source-large-language-models/other-labs-olmo-3-1-32b-instruct/"
    },
    {
      "slug": "other-labs-olmo-3-32b-base",
      "name": "Olmo 3 32B Base",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "allenai/Olmo-3-32B",
      "status": "ga",
      "flagship": false,
      "context_window": 65536,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "base-model",
        "continued-pretraining"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-12",
      "released": "2025-11-20",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "32B",
      "paramsTotalB": 32,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "The base checkpoint to fine-tune when you need documented provenance for every training token (Dolma 3, ~9.3T tokens) plus intermediate checkpoints.",
      "notes": [],
      "sources": [
        {
          "label": "allenai.org",
          "url": "https://allenai.org/blog/olmo3",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-olmo-3-32b-base/",
      "page": "/open-source-large-language-models/other-labs-olmo-3-32b-base/"
    },
    {
      "slug": "other-labs-olmo-3-7b-instruct",
      "name": "Olmo 3 7B Instruct",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "allenai/Olmo-3-7B-Instruct",
      "status": "ga",
      "flagship": false,
      "context_window": 65536,
      "max_output_tokens": 32768,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "chat",
        "instruction-following",
        "math",
        "code"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-12",
      "released": "2025-11-20",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "7B",
      "paramsTotalB": 7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Small fully-open chat model for academic baselines and ablation studies where a licence-clean, data-transparent 7B is the requirement.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/allenai/Olmo-3-7B-Instruct",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-olmo-3-7b-instruct/",
      "page": "/open-source-large-language-models/other-labs-olmo-3-7b-instruct/"
    },
    {
      "slug": "other-labs-olmo-3-7b-think",
      "name": "Olmo 3 7B Think",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "allenai/Olmo-3-7B-Think",
      "status": "ga",
      "flagship": false,
      "context_window": 65536,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "long-cot"
      ],
      "tags": [],
      "knowledge_cutoff": "2024-12",
      "released": "2025-11-20",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "7B",
      "paramsTotalB": 7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "reasoning",
      "best_for": "Cheap local long-chain-of-thought experiments; pair with the RL-Zero checkpoints if you are studying RL recipes rather than deploying.",
      "notes": [],
      "sources": [
        {
          "label": "allenai.org",
          "url": "https://allenai.org/blog/olmo3",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-olmo-3-7b-think/",
      "page": "/open-source-large-language-models/other-labs-olmo-3-7b-think/"
    },
    {
      "slug": "other-labs-molmo-2-8b",
      "name": "Molmo 2 8B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "allenai/Molmo2-8B",
      "status": "ga",
      "flagship": false,
      "context_window": 16384,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "video-understanding",
        "spatio-temporal-grounding",
        "counting",
        "captioning"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-12-11",
      "license": "Apache 2.0 (some training data is academic / non-commercial research only)",
      "licenseKind": "permissive",
      "parameters": "9B (Qwen3-8B backbone, SigLIP 2 vision)",
      "paramsTotalB": 9,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "Best open choice when you need pointing/grounding output — coordinates and timestamps rather than prose — on short video and multi-image inputs; check the non-commercial data caveat first.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/allenai/Molmo2-8B",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-molmo-2-8b/",
      "page": "/open-source-large-language-models/other-labs-molmo-2-8b/"
    },
    {
      "slug": "other-labs-molmo-2-4b",
      "name": "Molmo 2 4B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "allenai/Molmo2-4B",
      "status": "ga",
      "flagship": false,
      "context_window": 16384,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "video-understanding",
        "grounding"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-12-11",
      "license": "Apache 2.0 (some training data is academic / non-commercial research only)",
      "licenseKind": "permissive",
      "parameters": "4B (Qwen3 backbone)",
      "paramsTotalB": 4,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "vision",
      "best_for": "The efficiency-tuned Molmo 2 for single-GPU video grounding when the 8B is too heavy.",
      "notes": [],
      "sources": [
        {
          "label": "allenai.org",
          "url": "https://allenai.org/blog/molmo2",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-molmo-2-4b/",
      "page": "/open-source-large-language-models/other-labs-molmo-2-4b/"
    },
    {
      "slug": "other-labs-molmo-2-o-7b",
      "name": "Molmo 2-O 7B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "allenai/Molmo2-O-7B",
      "status": "ga",
      "flagship": false,
      "context_window": 16384,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "video-understanding",
        "grounding"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-12-11",
      "license": "Apache 2.0 (some training data is academic / non-commercial research only)",
      "licenseKind": "permissive",
      "parameters": "7B (Olmo backbone)",
      "paramsTotalB": 7,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "The only end-to-end open VLM here whose language backbone is also fully open (Olmo, not Qwen) — pick it when backbone provenance is the point.",
      "notes": [],
      "sources": [
        {
          "label": "allenai.org",
          "url": "https://allenai.org/blog/molmo2",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-molmo-2-o-7b/",
      "page": "/open-source-large-language-models/other-labs-molmo-2-o-7b/"
    },
    {
      "slug": "other-labs-nvidia-nemotron-3-ultra-550b-a55b",
      "name": "NVIDIA Nemotron 3 Ultra 550B-A55B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
      "status": "ga",
      "flagship": true,
      "context_window": 1000000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "agentic",
        "long-context",
        "tool-use",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-09 (pre-training), 2026-05 (post-training)",
      "released": "2026-06-04",
      "license": "OpenMDW-1.1",
      "licenseKind": "permissive",
      "parameters": "550B total / 55B active (LatentMoE, Mamba-2 + MoE)",
      "paramsTotalB": 550,
      "paramsActiveB": 55,
      "architecture": "Mixture of experts",
      "sizeBand": "frontier",
      "kind": "language",
      "best_for": "Frontier-scale open weights for on-prem agentic workloads — but it needs 8x GB200/B200 or 16x H100 minimum, so it is a datacentre commitment, not a download.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-nvidia-nemotron-3-ultra-550b-a55b/",
      "page": "/open-source-large-language-models/other-labs-nvidia-nemotron-3-ultra-550b-a55b/"
    },
    {
      "slug": "other-labs-nvidia-nemotron-3-super-120b-a12b",
      "name": "NVIDIA Nemotron 3 Super 120B-A12B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
      "status": "ga",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "agentic",
        "long-context",
        "tool-use",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-06 (pre-training), 2026-02 (post-training)",
      "released": "2026-03-11",
      "license": "NVIDIA Nemotron Open Model License",
      "licenseKind": "custom",
      "parameters": "120B total / 12B active (LatentMoE, Mamba-2 + MoE)",
      "paramsTotalB": 120,
      "paramsActiveB": 12,
      "architecture": "Mixture of experts",
      "sizeBand": "large",
      "kind": "language",
      "best_for": "The practical sweet spot of the Nemotron line: 12B active params keeps throughput high for high-volume ticket automation and RAG while retaining a 256K default window.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-nvidia-nemotron-3-super-120b-a12b/",
      "page": "/open-source-large-language-models/other-labs-nvidia-nemotron-3-super-120b-a12b/"
    },
    {
      "slug": "other-labs-nvidia-nemotron-3-5-lightning-30b-a3b",
      "name": "NVIDIA Nemotron 3.5 Lightning 30B-A3B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
      "status": "ga",
      "flagship": false,
      "context_window": 1000000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "toggleable-thinking",
        "long-context",
        "code",
        "tool-use"
      ],
      "tags": [],
      "knowledge_cutoff": "2025-09 (pre-training), 2026-05 (post-training)",
      "released": "2026-08-11",
      "license": "OpenMDW-1.1",
      "licenseKind": "permissive",
      "parameters": "30B total / 3B active (hybrid Mamba-2 + MoE)",
      "paramsTotalB": 30,
      "paramsActiveB": 3,
      "architecture": "Mixture of experts",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "Newest and most deployable Nemotron: 256K context on a single H100 with only 3B active params, and reasoning can be switched off per-request to cut token spend.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-nvidia-nemotron-3-5-lightning-30b-a3b/",
      "page": "/open-source-large-language-models/other-labs-nvidia-nemotron-3-5-lightning-30b-a3b/"
    },
    {
      "slug": "other-labs-nvidia-nemotron-nano-12b-v2-vl",
      "name": "NVIDIA Nemotron Nano 12B v2 VL",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image",
        "video"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "document-intelligence",
        "video-understanding",
        "ocr"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-10-28",
      "license": "NVIDIA Open Model License Agreement",
      "licenseKind": "custom",
      "parameters": "12.6B (CRadioV2-H vision encoder)",
      "paramsTotalB": 12.6,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "vision",
      "best_for": "Document intelligence — invoices, forms, charts — at up to 4 images of 2048x1536 per request; a strong self-hosted alternative to paid OCR/VLM APIs.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-nvidia-nemotron-nano-12b-v2-vl/",
      "page": "/open-source-large-language-models/other-labs-nvidia-nemotron-nano-12b-v2-vl/"
    },
    {
      "slug": "other-labs-granite-4-2-30b",
      "name": "Granite 4.2 30B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "ibm-granite/granite-4.2-30b",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "toggleable-thinking",
        "tool-use",
        "agentic",
        "code",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-25",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "30B dense",
      "paramsTotalB": 30,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "medium",
      "kind": "language",
      "best_for": "Apache-2.0 enterprise workhorse with 128K native context (512K extensible) and three effort levels, so you can dial reasoning cost per request; deploy on vLLM/SGLang rather than expecting an IBM per-token SKU.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/ibm-granite/granite-4.2-30b",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-granite-4-2-30b/",
      "page": "/open-source-large-language-models/other-labs-granite-4-2-30b/"
    },
    {
      "slug": "other-labs-granite-4-2-8b",
      "name": "Granite 4.2 8B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "ibm-granite/granite-4.2-8b",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "toggleable-thinking",
        "tool-use",
        "agentic",
        "code",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-25",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "8B dense (GQA, 32 heads, RoPE)",
      "paramsTotalB": 8,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "The best value in the Granite line for self-hosted agents: 128K context and reliable tool calling on a single mid-range GPU, 12 languages, no licence friction.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/ibm-granite/granite-4.2-8b",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-granite-4-2-8b/",
      "page": "/open-source-large-language-models/other-labs-granite-4-2-8b/"
    },
    {
      "slug": "other-labs-granite-4-2-3b",
      "name": "Granite 4.2 3B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "ibm-granite/granite-4.2-3b",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "toggleable-thinking",
        "tool-use",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "3B dense",
      "paramsTotalB": 3,
      "paramsActiveB": null,
      "architecture": "Dense",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "Smallest Granite 4.2 for CPU or edge deployment where you still want a 128K window; GGUF, nvfp4 and mxfp4 builds ship alongside.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/ibm-granite",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-granite-4-2-3b/",
      "page": "/open-source-large-language-models/other-labs-granite-4-2-3b/"
    },
    {
      "slug": "other-labs-lfm2-5-8b-a1b",
      "name": "LFM2.5-8B-A1B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "LiquidAI/LFM2.5-8B-A1B",
      "status": "ga",
      "flagship": false,
      "context_window": 128000,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "tool-use",
        "agentic",
        "on-device",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026",
      "license": "lfm1.0",
      "licenseKind": "custom",
      "parameters": "8.3B total / 1.5B active (18 double-gated conv + 6 GQA layers)",
      "paramsTotalB": 8.3,
      "paramsActiveB": 1.5,
      "architecture": "Mixture of experts",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Largest LFM: strong agentic tool use at 1.5B active params and 18.5K tok/s at high concurrency — but Liquid explicitly says it is weak on heavy coding and knowledge QA without retrieval, so pair it with RAG.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/LiquidAI/LFM2.5-8B-A1B",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-lfm2-5-8b-a1b/",
      "page": "/open-source-large-language-models/other-labs-lfm2-5-8b-a1b/"
    },
    {
      "slug": "other-labs-lfm2-5-2-6b",
      "name": "LFM2.5-2.6B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "LiquidAI/LFM2.5-2.6B",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tool-use",
        "instruction-following",
        "on-device",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-04",
      "license": "lfm1.0",
      "licenseKind": "custom",
      "parameters": "2.69B",
      "paramsTotalB": 2.69,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "language",
      "best_for": "The default on-device model here: 131K context, 16 languages, and competitive with models 4x its size on tool use — just note the bespoke lfm1.0 licence is not Apache.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/LiquidAI/LFM2.5-2.6B",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-lfm2-5-2-6b/",
      "page": "/open-source-large-language-models/other-labs-lfm2-5-2-6b/"
    },
    {
      "slug": "other-labs-lfm2-5-vl-3b",
      "name": "LFM2.5-VL-3B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "LiquidAI/LFM2.5-VL-3B",
      "status": "ga",
      "flagship": false,
      "context_window": 32768,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "vision",
        "on-device",
        "multilingual"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-08-12",
      "license": "lfm1.0",
      "licenseKind": "custom",
      "parameters": "3.1B (LFM2.5-2.6B backbone + SigLIP2 NaFlex 400M)",
      "paramsTotalB": 3.1,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "tiny",
      "kind": "vision",
      "best_for": "Vision on a laptop or NPU: ~3.3GB memory and 228 tok/s on an M5 Max, at the cost of a comparatively short 32K window.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/LiquidAI/LFM2.5-VL-3B",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-lfm2-5-vl-3b/",
      "page": "/open-source-large-language-models/other-labs-lfm2-5-vl-3b/"
    },
    {
      "slug": "other-labs-apriel-1-6-15b-thinker",
      "name": "Apriel 1.6 15B Thinker",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "ServiceNow-AI/Apriel-1.6-15b-Thinker",
      "status": "ga",
      "flagship": false,
      "context_window": 131072,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text",
        "image"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "reasoning",
        "vision",
        "tool-use",
        "code",
        "enterprise-workflows"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2025-12-09",
      "license": "MIT",
      "licenseKind": "permissive",
      "parameters": "15B",
      "paramsTotalB": 15,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "reasoning",
      "best_for": "MIT-licensed multimodal reasoner sized for a single GPU; ServiceNow claims ~30% fewer reasoning tokens than Apriel 1.5, which matters more than raw benchmark deltas for enterprise agent cost.",
      "notes": [],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/ServiceNow-AI/Apriel-1.6-15b-Thinker",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-apriel-1-6-15b-thinker/",
      "page": "/open-source-large-language-models/other-labs-apriel-1-6-15b-thinker/"
    },
    {
      "slug": "other-labs-arctic-awm-8b",
      "name": "Arctic-AWM-8B",
      "provider": "Other notable labs",
      "providerSlug": "other-labs",
      "api_id": "Snowflake/Arctic-AWM-8B",
      "status": "ga",
      "flagship": false,
      "context_window": -1,
      "max_output_tokens": 0,
      "input_price": -1,
      "output_price": -1,
      "cached_input_price": -1,
      "modalities_in": [
        "text"
      ],
      "modalities_out": [
        "text"
      ],
      "capabilities": [
        "tool-use",
        "agentic",
        "multi-turn",
        "mcp"
      ],
      "tags": [],
      "knowledge_cutoff": "",
      "released": "2026-02-10",
      "license": "Apache 2.0",
      "licenseKind": "permissive",
      "parameters": "8B (Qwen3 base, agentic RL)",
      "paramsTotalB": 8,
      "paramsActiveB": null,
      "architecture": "",
      "sizeBand": "small",
      "kind": "language",
      "best_for": "Narrow but useful: a small model RL-trained specifically for multi-turn MCP tool calling. Also ships at 4B and 14B. Treat it as a research artifact, not a general chat model.",
      "notes": [
        "Also ships at 4B and 14B. Treat it as a research artifact, not a general chat model."
      ],
      "sources": [
        {
          "label": "huggingface.co",
          "url": "https://huggingface.co/Snowflake/Arctic-AWM-8B",
          "primary": true
        },
        {
          "label": "Other notable labs pricing",
          "url": "https://www.ai21.com/pricing/",
          "primary": false
        },
        {
          "label": "API docs",
          "url": "https://docs.ai21.com/",
          "primary": false
        }
      ],
      "checked": "2026-09-06",
      "reference": "/reference/models/other-labs-arctic-awm-8b/",
      "page": "/open-source-large-language-models/other-labs-arctic-awm-8b/"
    }
  ]
}
