{
  "version": 1,
  "updated": "2026-07-22",
  "count": 195,
  "docs": "https://aimodelwatch.dev/api",
  "source": "Compiled from official provider documentation. Each model carries its own source_url.",
  "license": "MIT — attribution appreciated: AI Model Watch (https://aimodelwatch.dev)",
  "models": [
    {
      "id": "claude-fable-5",
      "name": "Claude Fable 5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 10,
      "price_output_per_mtok": 50,
      "price_cached_input_per_mtok": 1,
      "knowledge_cutoff": null,
      "released": "2026-06-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-fable-5",
      "notes": "Anthropic's most capable widely released model. GA on Claude API, Claude Platform on AWS, Amazon Bedrock, Google Cloud, and Microsoft Foundry beginning June 9, 2026. Adaptive thinking always on; no extended thinking. 5m cache write $12.50/MTok, 1h cache write $20/MTok (cached_input field here = Cache Hits & Refreshes $1/MTok). Batch: $5 in / $25 out per MTok. 1M context at standard pricing. Uses the Opus 4.7-generation tokenizer (~30-35% more tokens for the same text vs pre-4.7 models). Dateless model ID is a pinned snapshot. Knowledge cutoff not published on overview page (null).",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-mythos-5",
      "name": "Claude Mythos 5",
      "provider": "Anthropic",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 10,
      "price_output_per_mtok": 50,
      "price_cached_input_per_mtok": 1,
      "knowledge_cutoff": null,
      "released": "2026-06-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-mythos-5",
      "notes": "Limited availability (not GA) via Project Glasswing to approved customers, beginning June 9, 2026. Successor to Claude Mythos Preview. Adaptive thinking always on; no extended thinking. Pricing identical to Fable 5: $10 in / $50 out; 5m cache $12.50, 1h cache $20, cache hits $1/MTok; Batch $5/$25. 1M context at standard pricing. Status set to 'preview' to reflect limited/invite-only availability. Knowledge cutoff not published (null).",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-mythos-preview",
      "name": "Claude Mythos Preview",
      "provider": "Anthropic",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-06-30",
      "replacement": "claude-mythos-5",
      "api_string": "claude-mythos-preview",
      "notes": "Invitation-only research preview model for defensive cybersecurity workflows under Project Glasswing (no self-serve sign-up). Will be RETIRED on June 30, 2026; recommended migration to Claude Mythos 5 (claude-mythos-5). Standalone per-model pricing not published on the pricing page (Long context note groups it with the 1M-context models at standard pricing, but no $ figures listed) -> prices null. Included in 1M-token context group. Max output / knowledge cutoff not published (null). As of 2026-06-28 the doc lists it as still functional pending the June 30, 2026 retirement; marked 'deprecated'",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 25,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2026-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-opus-4-8",
      "notes": "Most capable Opus-tier model (complex reasoning, long-horizon agentic coding). API ID == alias 'claude-opus-4-8' (dateless pinned snapshot). Adaptive thinking; no extended thinking. effort defaults to 'high'. 5m cache $6.25/MTok, 1h cache $10/MTok, cache hits $0.50/MTok. Batch: $2.50 in / $12.50 out. 1M context at standard pricing; 200k on Microsoft Foundry. Supports up to 300k output via Batch API beta header. Fast mode (preview): $10 in / $50 out per MTok. Reliable knowledge cutoff Jan 2026; training data cutoff Jan 2026. Tentative retirement not sooner than May 28, 2027. Explicit release da",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-opus-4-7",
      "name": "Claude Opus 4.7",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 25,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2026-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-opus-4-7",
      "notes": "Listed under 'Legacy models' on overview but Active in deprecations table. Dateless pinned snapshot ID == alias. Adaptive thinking; no extended thinking. Introduced the new tokenizer (~30-35% more tokens for same text). temperature/top_p/top_k deprecated (400 error on non-default values) on Opus 4.7 and later. 5m cache $6.25, 1h cache $10, cache hits $0.50/MTok. Batch $2.50/$12.50. 1M context standard pricing. Fast mode (preview): $30 in / $150 out per MTok. Reliable knowledge cutoff Jan 2026; training cutoff Jan 2026. Tentative retirement not sooner than April 16, 2027. Status 'ga' per Active",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-opus-4-6",
      "name": "Claude Opus 4.6",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 25,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2025-05",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-opus-4-6",
      "notes": "Listed under 'Legacy models' on overview but Active in deprecations table. Dateless pinned snapshot ID == alias. Extended thinking: Yes; adaptive thinking: Yes. 5m cache $6.25, 1h cache $10, cache hits $0.50/MTok. Batch $2.50/$12.50. 1M context standard pricing. Fast mode (preview): $30 in / $150 out per MTok. Reliable knowledge cutoff May 2025; training data cutoff Aug 2025. Tentative retirement not sooner than February 5, 2027. Status 'ga' per Active state. Release date not explicitly published (null).",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-opus-4-5",
      "name": "Claude Opus 4.5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 25,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2025-05",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-opus-4-5-20251101",
      "notes": "API ID claude-opus-4-5-20251101 (alias claude-opus-4-5). Listed under 'Legacy models' on overview but Active in deprecations table. Extended thinking: Yes; adaptive thinking: No. 200k context (not 1M). Max output 64k. 5m cache $6.25, 1h cache $10, cache hits $0.50/MTok. Batch $2.50/$12.50. Reliable knowledge cutoff May 2025; training data cutoff Aug 2025. Snapshot date 20251101 implies a Nov 2025 release. Tentative retirement not sooner than November 24, 2026.",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-opus-4-1",
      "name": "Claude Opus 4.1",
      "provider": "Anthropic",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": 32000,
      "price_input_per_mtok": 15,
      "price_output_per_mtok": 75,
      "price_cached_input_per_mtok": 1.5,
      "knowledge_cutoff": "2025-01",
      "released": null,
      "deprecated_on": "2026-06-05",
      "retires_on": "2026-08-05",
      "replacement": "claude-opus-4-8",
      "api_string": "claude-opus-4-1-20250805",
      "notes": "DEPRECATED June 5, 2026; RETIRES August 5, 2026. Recommended replacement claude-opus-4-8. API ID claude-opus-4-1-20250805 (alias claude-opus-4-1). Extended thinking: Yes; adaptive thinking: No. 200k context, max output 32k. Pricing $15 in / $75 out; 5m cache $18.75, 1h cache $30, cache hits $1.50/MTok. Batch $7.50/$37.50. Reliable knowledge cutoff Jan 2025; training data cutoff Mar 2025. Snapshot date 20250805 implies an Aug 5, 2025 release.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-opus-4",
      "name": "Claude Opus 4",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": null,
      "price_input_per_mtok": 15,
      "price_output_per_mtok": 75,
      "price_cached_input_per_mtok": 1.5,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-14",
      "retires_on": "2026-06-15",
      "replacement": "claude-opus-4-8",
      "api_string": "claude-opus-4-20250514",
      "notes": "RETIRED June 15, 2026 on Anthropic-operated platforms (deprecated April 14, 2026); pricing page notes still available on Google Cloud. Recommended replacement claude-opus-4-8. API ID claude-opus-4-20250514. Pricing $15 in / $75 out; 5m cache $18.75, 1h cache $30, cache hits $1.50/MTok; Batch $7.50/$37.50. Snapshot date 20250514 implies a May 14, 2025 release. context_window 200k reflects the Opus 4-class family norm but is NOT on current official pages (not directly corroborated); max output left null.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": "2026-01",
      "released": "2026-06-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-sonnet-5",
      "notes": "Best combination of speed and intelligence; Anthropic's most agentic Sonnet, performance close to Opus 4.8 at lower cost. Dateless pinned snapshot ID == alias claude-sonnet-5. Extended thinking: No; adaptive thinking: Yes (effort defaults to 'high' on the Claude API and Claude Code). 1M context at standard pricing; max output 128k (300k via Batch API beta header output-300k-2026-03-24). Prices shown are STANDARD ($3 in / $15 out, 5m cache $3.75, 1h cache $6, cache hit $0.30, Batch $1.50/$7.50). INTRODUCTORY pricing of $2 in / $10 out per MTok is in effect through August 31, 2026 (5m cache $2.50, 1h cache $4, cache hit $0.20, Batch $1/$5), reverting to standard on September 1, 2026 — WATCH: if standard is not yet in effect, current live cost is the intro rate. Uses the newer tokenizer (~30% more tokens for the same text). Reliable knowledge cutoff Jan 2026; training data cutoff Jan 2026. Supersedes Claude Sonnet 4.6 (now a legacy model). Released 2026-06-30 (system card date).",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-sonnet-4-6",
      "name": "Claude Sonnet 4.6",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": "2025-08",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-sonnet-4-6",
      "notes": "Best combination of speed and intelligence. Dateless pinned snapshot ID == alias claude-sonnet-4-6. Extended thinking: Yes; adaptive thinking: Yes. 1M context, max output 128k (300k via Batch API beta header). 5m cache $3.75, 1h cache $6, cache hits $0.30/MTok. Batch $1.50/$7.50. Reliable knowledge cutoff Aug 2025; training data cutoff Jan 2026. Tentative retirement not sooner than February 17, 2027. Release date not explicitly published (null). Now a LEGACY model (still generally available, not deprecated) — superseded by Claude Sonnet 5 as of 2026-06-30; consider migrating for improved performance.",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-sonnet-4-5",
      "name": "Claude Sonnet 4.5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": "2025-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-sonnet-4-5-20250929",
      "notes": "API ID claude-sonnet-4-5-20250929 (alias claude-sonnet-4-5). Listed under 'Legacy models' on overview but Active in deprecations table. Extended thinking: Yes; adaptive thinking: No. 200k context, max output 64k. 5m cache $3.75, 1h cache $6, cache hits $0.30/MTok. Batch $1.50/$7.50. Reliable knowledge cutoff Jan 2025; training data cutoff Jul 2025. Snapshot 20250929 implies Sep 29, 2025 release. Tentative retirement not sooner than September 29, 2026.",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-sonnet-4",
      "name": "Claude Sonnet 4",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-14",
      "retires_on": "2026-06-15",
      "replacement": "claude-sonnet-4-6",
      "api_string": "claude-sonnet-4-20250514",
      "notes": "RETIRED June 15, 2026 on Anthropic-operated platforms (deprecated April 14, 2026); pricing page notes still available on Bedrock and Google Cloud. Recommended replacement claude-sonnet-4-6. API ID claude-sonnet-4-20250514. Pricing $3 in / $15 out; 5m cache $3.75, 1h cache $6, cache hits $0.30/MTok; Batch $1.50/$7.50. Snapshot 20250514 implies May 14, 2025 release. Context window and max output not in current overview tables -> null (not corroborated on current pages).",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": 0.1,
      "knowledge_cutoff": "2025-02",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-haiku-4-5-20251001",
      "notes": "Fastest model with near-frontier intelligence. API ID claude-haiku-4-5-20251001 (alias claude-haiku-4-5). Extended thinking: Yes; adaptive thinking: No. 200k context, max output 64k. 5m cache $1.25, 1h cache $2, cache hits $0.10/MTok. Batch $0.50/$2.50. Reliable knowledge cutoff Feb 2025; training data cutoff Jul 2025. Snapshot 20251001 implies Oct 1, 2025 release. Tentative retirement not sooner than October 15, 2026.",
      "source_url": "https://platform.claude.com/docs/en/docs/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-haiku-3-5",
      "name": "Claude Haiku 3.5",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.8,
      "price_output_per_mtok": 4,
      "price_cached_input_per_mtok": 0.08,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-12-19",
      "retires_on": "2026-02-19",
      "replacement": "claude-haiku-4-5-20251001",
      "api_string": "claude-3-5-haiku-20241022",
      "notes": "RETIRED February 19, 2026 on Anthropic-operated platforms (deprecated December 19, 2025); pricing page notes still available on Bedrock and Google Cloud. Recommended replacement claude-haiku-4-5-20251001. API ID claude-3-5-haiku-20241022. Pricing $0.80 in / $4 out; 5m cache $1, 1h cache $1.60, cache hits $0.08/MTok; Batch $0.40/$2. Snapshot 20241022 implies Oct 22, 2024 release. Context/max-output not in current overview tables -> null.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-3-7-sonnet",
      "name": "Claude Sonnet 3.7",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-10-28",
      "retires_on": "2026-02-19",
      "replacement": "claude-sonnet-4-6",
      "api_string": "claude-3-7-sonnet-20250219",
      "notes": "RETIRED February 19, 2026 (deprecated October 28, 2025). Recommended replacement claude-sonnet-4-6. API ID claude-3-7-sonnet-20250219. No longer listed in the current pricing table -> per-model prices null (not corroborated on current pricing page). Snapshot 20250219 implies Feb 19, 2025 release.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "claude-3-haiku",
      "name": "Claude Haiku 3",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-02-19",
      "retires_on": "2026-04-20",
      "replacement": "claude-haiku-4-5-20251001",
      "api_string": "claude-3-haiku-20240307",
      "notes": "RETIRED April 20, 2026 (deprecated February 19, 2026). Recommended replacement claude-haiku-4-5-20251001. API ID claude-3-haiku-20240307. Not listed in the current pricing table -> per-model prices null. Snapshot 20240307 implies Mar 7, 2024 release.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-6-sol",
      "name": "GPT-5.6 Sol",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 30,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2026-02-16",
      "released": "2026-07-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.6-sol",
      "notes": "Current OpenAI flagship — the top 'Sol' tier of the GPT-5.6 family (Sol/Terra/Luna), GA on the API + Codex 2026-07-09 (preview from 2026-06-26). Text+image input, text output, configurable reasoning effort. Priced identically to GPT-5.5 ($5/$0.50/$30) but stronger on coding/knowledge-work/cyber/science per OpenAI. New caching model: explicit cache breakpoints, 30-min min cache life, cache writes billed 1.25x uncached input, reads keep the 90% discount. Standard tier; batch ~50%.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-6-terra",
      "name": "GPT-5.6 Terra",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.25,
      "knowledge_cutoff": "2026-02-16",
      "released": "2026-07-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.6-terra",
      "notes": "Balanced 'Terra' tier of the GPT-5.6 family — GPT-5.5-class quality at lower cost, positioned for everyday work. GA on the API + Codex 2026-07-09. Text+image input, text output. Pricing from developers.openai.com/api/docs/pricing (standard tier); batch ~50%.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-6-luna",
      "name": "GPT-5.6 Luna",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 6,
      "price_cached_input_per_mtok": 0.1,
      "knowledge_cutoff": "2026-02-16",
      "released": "2026-07-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.6-luna",
      "notes": "Fast, cost-efficient 'Luna' tier of the GPT-5.6 family — built for high-volume tasks. GA on the API + Codex 2026-07-09. Text+image input, text output, full 1.05M context. Pricing from developers.openai.com/api/docs/pricing (standard tier); batch ~50%.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-5",
      "name": "GPT-5.5",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 30,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2025-12-01",
      "released": "2026-04-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.5",
      "notes": "Prior flagship, still GA — superseded as OpenAI's top tier by GPT-5.6 Sol (2026-07-09) at the same $5/$0.50/$30 price. Snapshot alias gpt-5.5-2026-04-23 (snapshot date taken as release date). Text input / text output, configurable reasoning effort. Pricing from developers.openai.com/api/docs/pricing (standard tier); batch tier is ~50% ($2.50/$0.25/$15.00).",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.5",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-5-pro",
      "name": "GPT-5.5 Pro",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 30,
      "price_output_per_mtok": 180,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2025-12-01",
      "released": "2026-04-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.5-pro",
      "notes": "Snapshot alias gpt-5.5-pro-2026-04-23. Responses API only (incl. Batch); long-running, background mode recommended. Reasoning effort medium/high/xhigh. No cached-input price published. Regional data-residency endpoints add ~10% surcharge.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.5-pro",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-4",
      "name": "GPT-5.4",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.25,
      "knowledge_cutoff": "2025-08-31",
      "released": "2026-03-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.4",
      "notes": "Snapshot alias gpt-5.4-2026-03-05 (snapshot date taken as release date; not explicitly stated on doc page). Reasoning model with configurable effort.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.4",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-4-mini",
      "name": "GPT-5.4 mini",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 0.75,
      "price_output_per_mtok": 4.5,
      "price_cached_input_per_mtok": 0.075,
      "knowledge_cutoff": "2025-08-31",
      "released": "2026-03-17",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.4-mini",
      "notes": "Snapshot alias gpt-5.4-mini-2026-03-17 (snapshot date taken as release date; not explicitly stated). Positioned for coding, computer use, subagents. Note: context window 400K (smaller than full gpt-5.4's 1.05M).",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.4-mini",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-4-nano",
      "name": "GPT-5.4 nano",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": 1.25,
      "price_cached_input_per_mtok": 0.02,
      "knowledge_cutoff": "2025-08-31",
      "released": "2026-03-17",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.4-nano",
      "notes": "Snapshot alias gpt-5.4-nano-2026-03-17. Cheapest GPT-5.4-class model for high-volume classification/extraction/ranking/subagents. Reasoning effort none(default)/low/medium/high/xhigh.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.4-nano",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-4-pro",
      "name": "GPT-5.4 Pro",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 30,
      "price_output_per_mtok": 180,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2025-08-31",
      "released": "2026-03-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.4-pro",
      "notes": "Snapshot alias gpt-5.4-pro-2026-03-05. Responses API only. Reasoning effort medium/high/xhigh. Doc notes potential 2x/1.5x multipliers for sessions exceeding 272K input tokens. No cached-input price published.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.4-pro",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-3-codex",
      "name": "GPT-5.3-Codex",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 1.75,
      "price_output_per_mtok": 14,
      "price_cached_input_per_mtok": 0.175,
      "knowledge_cutoff": "2025-08-31",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.3-codex",
      "notes": "Agentic coding model optimized for Codex. Reasoning effort low/medium/high/xhigh. Release date not stated on doc page. Cached-input price ($0.175) from pricing page.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.3-codex",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-realtime-2",
      "name": "GPT-Realtime-2",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "audio",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": 32000,
      "price_input_per_mtok": 4,
      "price_output_per_mtok": 24,
      "price_cached_input_per_mtok": 0.4,
      "knowledge_cutoff": "2024-09-30",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-realtime-2",
      "notes": "Speech-to-speech realtime model. Input: text/audio/image; Output: text/audio. Prices listed are the TEXT-token rates ($4 in / $24 out / $0.40 cached). Audio tokens are billed separately at $32 input / $64 output per 1M; image input $5/1M. Max output 32K tokens.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-realtime-2",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-image-2",
      "name": "GPT Image 2",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 8,
      "price_output_per_mtok": 30,
      "price_cached_input_per_mtok": 2,
      "knowledge_cutoff": null,
      "released": "2026-04-21",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-image-2",
      "notes": "Current flagship image-gen model (v1/images/generations). Snapshot gpt-image-2-2026-04-21. Input text+image, output image. Prices are per-1M-token for image input ($8, cached $2) and image output ($30). Designated replacement for dall-e-2/dall-e-3 (retired May 12, 2026).",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-image-2",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-image-1-5",
      "name": "GPT Image 1.5",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 8,
      "price_output_per_mtok": 32,
      "price_cached_input_per_mtok": 2,
      "knowledge_cutoff": null,
      "released": "2025-12-16",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "gpt-image-2",
      "api_string": "gpt-image-1.5",
      "notes": "Previous-generation image model, still available. Snapshot gpt-image-1.5-2025-12-16 (the dated snapshot is marked Deprecated on the doc page, though the base alias remains listed). Image output token price $32/1M.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-image-1.5",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-image-1-mini",
      "name": "GPT Image 1 mini",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": 0.25,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-image-1-mini",
      "notes": "Low-cost image model. Pricing per 1M tokens: image input $2.50 (cached $0.25), output $8.00. Spec page not separately retrieved; data from pricing page.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-4o-transcribe",
      "name": "GPT-4o Transcribe",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "audio",
        "text"
      ],
      "context_window": 16000,
      "max_output_tokens": 2000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-06-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-4o-transcribe",
      "notes": "Speech-to-text. Input audio+text, output text. ~$0.006/min. Token prices shown are text-token equivalents; audio input billed differently. Still listed as current on models index.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-4o-transcribe",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-4o-mini-transcribe",
      "name": "GPT-4o mini Transcribe",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "audio",
        "text"
      ],
      "context_window": 16000,
      "max_output_tokens": 2000,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-06-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-4o-mini-transcribe",
      "notes": "Cheaper transcription variant (~$0.003/min). Context window/max output assumed same as gpt-4o-transcribe (16K/2K) but not separately confirmed on its own spec page; price from pricing page.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "text-embedding-3-large",
      "name": "text-embedding-3-large",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8191,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.13,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-01-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "text-embedding-3-large",
      "notes": "Most capable embedding model. Output up to 3072 dimensions (reducible via dimensions param). Max input 8191 tokens (8191/3072 corroborated by OpenAI new-embedding-models announcement, not the model spec page). Embeddings have no output-token billing; $0.13/1M input.",
      "source_url": "https://developers.openai.com/api/docs/models/text-embedding-3-large",
      "open_weight": false,
      "embedding_dimensions": 3072
    },
    {
      "id": "text-embedding-3-small",
      "name": "text-embedding-3-small",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8191,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.02,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-01-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "text-embedding-3-small",
      "notes": "Cost-efficient embedding model. Default 1536 dimensions, max input 8191 tokens (corroborated by OpenAI new-embedding-models announcement). $0.02/1M input; no output-token billing.",
      "source_url": "https://developers.openai.com/api/docs/models/text-embedding-3-small",
      "open_weight": false,
      "embedding_dimensions": 1536
    },
    {
      "id": "gpt-5",
      "name": "GPT-5",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-09-30",
      "released": "2025-08-07",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.5",
      "api_string": "gpt-5-2025-08-07",
      "notes": "Previous flagship. Deprecation announced 2026-06-11; snapshot gpt-5-2025-08-07 shuts down 2026-12-11, replaced by gpt-5.5. Current per-token price not captured (not on the current pricing summary). Release date = snapshot date.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-mini",
      "name": "GPT-5 mini",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-08-07",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.4-mini",
      "api_string": "gpt-5-mini-2025-08-07",
      "notes": "Snapshot gpt-5-mini-2025-08-07 shuts down 2026-12-11, replaced by gpt-5.4-mini. Context/max-output assumed same as gpt-5 family (400K/128K); price not captured.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-nano",
      "name": "GPT-5 nano",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-08-07",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.4-nano",
      "api_string": "gpt-5-nano-2025-08-07",
      "notes": "Snapshot gpt-5-nano-2025-08-07 shuts down 2026-12-11, replaced by gpt-5.4-nano. Context/max-output assumed same as gpt-5 family; price not captured.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-pro",
      "name": "GPT-5 Pro",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-10-06",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.5-pro",
      "api_string": "gpt-5-pro-2025-10-06",
      "notes": "Snapshot gpt-5-pro-2025-10-06 shuts down 2026-12-11, replaced by gpt-5.5-pro. Price not captured.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "o3",
      "name": "o3",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": 100000,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2024-06-01",
      "released": "2025-04-16",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.5",
      "api_string": "o3-2025-04-16",
      "notes": "o-series reasoning model. Snapshot o3-2025-04-16 shuts down 2026-12-11, replaced by gpt-5.5. Pricing still on spec page: $2 in / $0.50 cached / $8 out per 1M.",
      "source_url": "https://developers.openai.com/api/docs/models/o3",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "o3-pro",
      "name": "o3-pro",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": 100000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-06-01",
      "released": "2025-06-10",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.5-pro",
      "api_string": "o3-pro-2025-06-10",
      "notes": "Higher-compute o3 variant. Snapshot o3-pro-2025-06-10 shuts down 2026-12-11, replaced by gpt-5.5-pro. Context/max-output assumed same as o3 (200K/100K); price not captured on current pricing summary.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "o3-deep-research",
      "name": "o3-deep-research",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": 100000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 20,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-22",
      "retires_on": "2026-07-23",
      "replacement": "gpt-5.5-pro",
      "api_string": "o3-deep-research",
      "notes": "Deep-research model. Deprecation announced 2026-04-22; shuts down 2026-07-23, replaced by gpt-5.5-pro. Pricing shown is BATCH tier ($5 in / $20 out per 1M). Context/max-output assumed o3-class (200K/100K).",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "o4-mini-deep-research",
      "name": "o4-mini-deep-research",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": 100000,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "o4-mini-deep-research",
      "notes": "Still listed on pricing page. Prices shown are BATCH tier ($1 in / $4 out per 1M). Context/max-output assumed o4-mini-class (200K/100K), not separately confirmed. Status GA but uncertain; not in current deprecation list.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "computer-use-preview",
      "name": "computer-use-preview",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.5,
      "price_output_per_mtok": 6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-22",
      "retires_on": "2026-07-23",
      "replacement": "gpt-5.4-mini",
      "api_string": "computer-use-preview",
      "notes": "Computer-use agent model. Deprecation announced 2026-04-22; shuts down 2026-07-23, replaced by gpt-5.4-mini. Pricing shown is BATCH tier ($1.50 in / $6 out per 1M). Context/max-output not published.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gpt-5-codex",
      "name": "GPT-5-Codex",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-22",
      "retires_on": "2026-07-23",
      "replacement": "gpt-5.5",
      "api_string": "gpt-5-codex",
      "notes": "Earlier Codex model. Deprecation announced 2026-04-22; shuts down 2026-07-23, replaced by gpt-5.5. Also covers gpt-5.1-codex* per the deprecation table. Context/max-output assumed gpt-5-class; price not captured.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "sora-2",
      "name": "Sora 2",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "video"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-03-24",
      "retires_on": "2026-09-24",
      "replacement": null,
      "api_string": "sora-2",
      "notes": "Video generation (Videos API). Priced per second, not per token: sora-2 $0.10/s (720p, standard) / $0.05/s batch. Videos API + sora-2* announced for discontinuation 2026-03-24; shuts down 2026-09-24 with no replacement (discontinued). Token-price fields N/A (per-second billing).",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "sora-2-pro",
      "name": "Sora 2 Pro",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "video"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-03-24",
      "retires_on": "2026-09-24",
      "replacement": null,
      "api_string": "sora-2-pro",
      "notes": "Higher-quality video model. Per-second pricing: $0.30/s (720p) up to $0.70/s (1080p) standard; batch half. Discontinued with Videos API, shuts down 2026-09-24, no replacement. Token-price fields N/A.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "dall-e-3",
      "name": "DALL-E 3",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2023-11-06",
      "deprecated_on": "2025-11-14",
      "retires_on": "2026-05-12",
      "replacement": "gpt-image-2",
      "api_string": "dall-e-3",
      "notes": "Legacy image model. Priced per-image (not per token). Deprecation announced 2025-11-14; retired 2026-05-12, replaced by gpt-image-2. As of 2026-06-28 this is past its shutdown date (effectively retired).",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "dall-e-2",
      "name": "DALL-E 2",
      "provider": "OpenAI",
      "status": "retired",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-11-14",
      "retires_on": "2026-05-12",
      "replacement": "gpt-image-2",
      "api_string": "dall-e-2",
      "notes": "Legacy image model, per-image pricing. Retired 2026-05-12 (past shutdown date as of 2026-06-28), replaced by gpt-image-2.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-3-1-pro-preview",
      "name": "Gemini 3.1 Pro Preview",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 12,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": "2025-01",
      "released": "2026-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.1-pro-preview",
      "notes": "Output modality: text only. Tiered pricing by prompt size: input $2.00 (<=200k tokens) / $4.00 (>200k); output $12.00 (<=200k) / $18.00 (>200k); cached input $0.20 (<=200k) / $0.40 (>200k) per 1M tokens. The values shown here are the <=200k tier. gemini-3-pro-preview alias now points to this model (the original gemini-3-pro-preview was shut down 2026-03-09). Preview/not-stable; may change. Source pages also include ai.google.dev/gemini-api/docs/pricing and /deprecations.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-pro-preview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-3-5-flash",
      "name": "Gemini 3.5 Flash",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 1.5,
      "price_output_per_mtok": 9,
      "price_cached_input_per_mtok": 0.15,
      "knowledge_cutoff": "2025-01",
      "released": "2026-05-19",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.5-flash",
      "notes": "GA/stable. Output token limit reported as 'up to 65,000' on the what's-new page; model spec convention is 65,536 - treated as 65536. The what's-new page describes additional output capabilities (images, audio, structured outputs) but the spec table for the core text model lists text output. Listed as the recommended replacement for gemini-2.5-flash and the gemini-2.0-flash line. Context-caching also has a per-hour storage fee not captured here. What's-new page last updated 2026-06-24 UTC.",
      "source_url": "https://ai.google.dev/gemini-api/docs/whats-new-gemini-3.5",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-3-flash-preview",
      "name": "Gemini 3 Flash Preview",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": 0.05,
      "knowledge_cutoff": "2025-01",
      "released": "2025-12",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3-flash-preview",
      "notes": "Output: text only. Pricing differs for audio input: input $0.50 (text/image/video) / $1.00 (audio); cached input $0.05 (text/image/video) / $0.10 (audio) per 1M tokens. Values shown are the text/image/video tier. Preview model; superseded in the lineup by the GA Gemini 3.5 Flash but still listed separately and not on the deprecation schedule as of 2026-06-28.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-3-flash-preview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-3-1-flash-lite",
      "name": "Gemini 3.1 Flash-Lite",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.25,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": 0.025,
      "knowledge_cutoff": "2025-01",
      "released": "2026-05-07",
      "deprecated_on": null,
      "retires_on": "2027-05-07",
      "replacement": null,
      "api_string": "gemini-3.1-flash-lite",
      "notes": "Listed as Stable on the models overview. Output: text only. Audio input priced higher: input $0.25 (text/image/video) / $0.50 (audio); cached input $0.025 (text/image/video) / $0.05 (audio) per 1M tokens. Deprecations page lists release 2026-05-07 and a shutdown date of 2027-05-07 with no replacement yet named (standard ~1yr stable lifecycle). It is the recommended replacement for gemini-2.5-flash-lite and the gemini-2.0-flash-lite line.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-2-5-pro",
      "name": "Gemini 2.5 Pro",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": 0.125,
      "knowledge_cutoff": "2025-01",
      "released": "2025-06-17",
      "deprecated_on": null,
      "retires_on": "2026-10-16",
      "replacement": "gemini-3.1-pro-preview",
      "api_string": "gemini-2.5-pro",
      "notes": "Output: text only. Tiered pricing: input $1.25 (<=200k tokens) / $2.50 (>200k); output $10.00 (<=200k) / $15.00 (>200k); cached input $0.125 (<=200k) / $0.25 (>200k) per 1M tokens (values shown are <=200k tier). Deprecations page: released 2025-06-17, shutdown 2026-10-16, replacement gemini-3.1-pro-preview. Still callable as of 2026-06-28 but on the deprecation schedule. Latest update June 2025.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-pro",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-2-5-flash",
      "name": "Gemini 2.5 Flash",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.03,
      "knowledge_cutoff": "2025-01",
      "released": "2025-06-17",
      "deprecated_on": null,
      "retires_on": "2026-10-16",
      "replacement": "gemini-3.5-flash",
      "api_string": "gemini-2.5-flash",
      "notes": "Output: text only. Audio input priced higher: input $0.30 (text/image/video) / $1.00 (audio); cached input $0.03 (text/image/video) / $0.10 (audio) per 1M tokens (values shown are text/image/video tier). Deprecations page: released 2025-06-17, shutdown 2026-10-16, replacement gemini-3.5-flash. Still callable as of 2026-06-28. Latest update June 2025.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-flash",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-2-5-flash-lite",
      "name": "Gemini 2.5 Flash-Lite",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": 0.01,
      "knowledge_cutoff": "2025-01",
      "released": "2025-07-22",
      "deprecated_on": null,
      "retires_on": "2026-10-16",
      "replacement": "gemini-3.1-flash-lite",
      "api_string": "gemini-2.5-flash-lite",
      "notes": "Output: text only. Audio input priced higher: input $0.10 (text/image/video) / $0.30 (audio); cached input $0.01 (text/image/video) / $0.03 (audio) per 1M tokens (values shown are text/image/video tier). Deprecations page: released 2025-07-22, shutdown 2026-10-16, replacement gemini-3.1-flash-lite. Still callable as of 2026-06-28. Latest update July 2025.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-flash-lite",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-2-0-flash",
      "name": "Gemini 2.0 Flash",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 1048576,
      "max_output_tokens": 8192,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": 0.025,
      "knowledge_cutoff": "2024-08",
      "released": "2025-02-05",
      "deprecated_on": null,
      "retires_on": "2026-06-01",
      "replacement": "gemini-3.5-flash",
      "api_string": "gemini-2.0-flash",
      "notes": "RETIRED/shut down 2026-06-01 (model page shows 'Gemini 2.0 Flash is deprecated and has been shut down June 1, 2026'). As of today 2026-06-28 it is no longer callable. Output: text only (8,192 max output tokens). Audio input priced higher: input $0.10 (text/image/video) / $0.70 (audio); cached input $0.025 (text/image/video) / $0.175 (audio) per 1M tokens. Knowledge cutoff August 2024. gemini-2.0-flash-001 alias retired same day. Replacement gemini-3.5-flash. Pricing retained on page for reference.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.0-flash",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-2-0-flash-lite",
      "name": "Gemini 2.0 Flash-Lite",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 1048576,
      "max_output_tokens": 8192,
      "price_input_per_mtok": 0.075,
      "price_output_per_mtok": 0.3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-08",
      "released": "2025-02-25",
      "deprecated_on": null,
      "retires_on": "2026-06-01",
      "replacement": "gemini-3.1-flash-lite",
      "api_string": "gemini-2.0-flash-lite",
      "notes": "RETIRED/shut down 2026-06-01 (deprecations page, alongside gemini-2.0-flash-lite-001). No longer callable as of 2026-06-28. Context caching not offered (cached input null). Max output 8,192 tokens. Knowledge cutoff not separately confirmed on a dedicated spec page; inferred August 2024 same as gemini-2.0-flash family - LOW confidence, treat as approximate. Modalities inferred from the 2.0 family - moderate confidence. Replacement gemini-3.1-flash-lite. Released 2025-02-25 per deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-embedding-2",
      "name": "Gemini Embedding 2",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-embedding-2",
      "notes": "Multimodal embedding model (stable/GA). Output is a text embedding vector, not tokens, so there is no output-token price. Input token limit 8,192; output dimension size flexible 128-3072 (recommended 768 / 1536 / 3072). The $0.20/1M figure is the TEXT input rate; the pricing page bills other input types separately - image $0.45, audio $6.50, video $12.00 per 1M (batch is half of each). Docs list 'latest update April 2026' rather than a release date, so released is null rather than guessed.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-embedding-2",
      "open_weight": false,
      "embedding_dimensions": 3072
    },
    {
      "id": "gemini-embedding-001",
      "name": "Gemini Embedding 001",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 2048,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-embedding-001",
      "notes": "Text-only embedding model (stable/GA). Input token limit 2,048; output dimension size flexible 128-3072 (recommended 768 / 1536 / 3072). Embeddings have no output-token billing; $0.15/1M input, $0.075/1M batch. Google's embeddings docs list it alongside gemini-embedding-2 as currently available and state no deprecation or retirement date, so lifecycle fields stay null (never infer a sunset the provider has not declared). Docs list 'last updated June 2025' rather than a release date, so released is null.",
      "source_url": "https://ai.google.dev/gemini-api/docs/embeddings",
      "open_weight": false,
      "embedding_dimensions": 3072
    },
    {
      "id": "gemini-2-5-computer-use-preview",
      "name": "Gemini 2.5 Computer Use",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-2.5-computer-use-preview-10-2025",
      "notes": "Specialist UI-automation model that emits browser/computer actions from a screenshot - not a general chat model. Preview. Input image+text, output text; 128,000 input / 64,000 output tokens. Priced on the same two-tier scale as gemini-2.5-pro: $1.25/M in and $10.00/M out at <=200K prompt tokens, rising to $2.50/M and $15.00/M above it - the base tier is stored here, matching the gemini-2-5-pro row's convention. Google's model page notes that Gemini 3 Pro and Flash have built-in computer use without a separate model, but it declares NO deprecation or retirement date for this model, so lifecycle fields stay null. Cached-input rate not published for this model. Spec table latest update October 2025 (page revised 2026-04-28), which is not a release date, so released is null.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-computer-use-preview-10-2025",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "gemini-robotics-er-1-6-preview",
      "name": "Gemini Robotics-ER 1.6",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 131072,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2025-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-robotics-er-1.6-preview",
      "notes": "Embodied-reasoning vision-language model for robotics (spatial understanding, pointing, trajectory planning) - a specialist, not a general chat model. Preview. Inputs text/image/video/audio, output text; 131,072 input / 65,536 output tokens. $1.00/M in and $5.00/M out standard, $0.50/$2.50 batch. Supports function calling, structured output, thinking and caching, but no separate cached-input rate is published. Spec table latest update December 2025, which is not a release date, so released is null.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-robotics-er-1.6-preview",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-5",
      "name": "Grok 4.5",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 500000,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 6,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": null,
      "released": "2026-07-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.5",
      "notes": "New flagship launched 2026-07-08 (public access 2026-07-09); xAI's smartest model for chat, coding, agentic and knowledge work per docs.x.ai. Aliases: grok-4.5-latest, grok-build-latest. Modality 'text, image -> text' (vision) per official model page. Cached input $0.30/M per the official model page (2026-07-22); it read $0.50/M on 2026-07-12, so this may be a cached-tier reduction, but the change is not independently timestamped by xAI and is recorded here as a value correction, not a changelog price-change event. TIERED pricing: figures shown are the base tier for prompts <=200K tokens; prompts >200K are billed 2x ($4/M input, $0.60/M cached, $12/M output). 500K context window (smaller than Grok 4.3's 1M). Max output tokens and knowledge cutoff not published on the docs page.",
      "source_url": "https://docs.x.ai/developers/models/grok-4.5",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-3",
      "name": "Grok 4.3",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": "2026-04-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.3",
      "notes": "Current flagship/recommended model for chat and coding per docs.x.ai/developers/models. Aliases: grok-4.3-latest, grok-latest. Modality 'text, image -> text' per official model page. Reasoning model (supports function calling, structured outputs, reasoning). Cached input $0.20/M confirmed on official model page. Max output tokens not published on the docs page (third-party sources describe 'no output token limit' but this is not officially stated). Knowledge cutoff not explicitly stated on the 4.x model page; the main models page only states the Nov 2024 cutoff for 'Grok 3 and Grok 4'. Release",
      "source_url": "https://docs.x.ai/developers/models/grok-4.3",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-20-0309-reasoning",
      "name": "Grok 4.20 (0309) Reasoning",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.20-0309-reasoning",
      "notes": "Reasoning-optimized variant. Aliases include grok-4.20-reasoning-latest, grok-4.20, grok-4.20-reasoning. Modality 'text, image -> text' per official model page. Pricing and cached-input ($0.20/M) confirmed on official model page. Max output tokens, knowledge cutoff and release date not published on docs. Note: docs warn no logprobs support for grok-4.20+.",
      "source_url": "https://docs.x.ai/developers/models/grok-4.20-0309-reasoning",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-20-0309-non-reasoning",
      "name": "Grok 4.20 (0309) Non-Reasoning",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.20-0309-non-reasoning",
      "notes": "Non-reasoning (latency-sensitive) variant; reasoning disabled, function calling and structured outputs supported. Aliases include grok-4.20-non-reasoning-latest. Modality 'text, image -> text'. Pricing and cached input ($0.20/M) confirmed on official model page. Max output tokens, knowledge cutoff and release date not published on docs.",
      "source_url": "https://docs.x.ai/developers/models/grok-4.20-0309-non-reasoning",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-20-multi-agent-0309",
      "name": "Grok 4.20 Multi-Agent (0309)",
      "provider": "xAI",
      "status": "beta",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.20-multi-agent-0309",
      "notes": "Marked Beta on the official model page; designed for multi-agent orchestration. Aliases include grok-4.20-multi-agent, grok-4.20-multi-agent-latest. Modality 'text, image -> text'. Lower rate limits than the standard 4.20 SKUs (9 req/s, 2.5M tokens/min vs 37 req/s, 10M tokens/min). Pricing and cached input ($0.20/M) confirmed on official model page. Official docs list a 1M context window; one third-party source claims 2M but this is NOT corroborated by docs.x.ai, so 1M is used.",
      "source_url": "https://docs.x.ai/developers/models/grok-4.20-multi-agent-0309",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-build-0-1",
      "name": "Grok Build 0.1",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-build-0.1",
      "notes": "Coding-focused model. Aliases: grok-code-fast-1, grok-code-fast, grok-code-fast-1-0825 (the retired grok-code-fast-1 slug now redirects here). Modality 'text, image -> text'. 256k context window. Pricing $1.00/$2.00 per M with $0.20/M cached input, confirmed on official model page. Max output tokens, knowledge cutoff and release date not published on docs; a third-party source cites a 2026-05-29 launch (uncorroborated by official docs).",
      "source_url": "https://docs.x.ai/developers/models/grok-build-0.1",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-imagine-image-quality",
      "name": "Grok Imagine Image (Quality)",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-imagine-image-quality",
      "notes": "Image-generation model priced per image ($0.05/image), not per token, so token price fields are null. This is the redirect/replacement target for the retired grok-imagine-image-pro (retired 2026-05-15). Standard tier grok-imagine-image is $0.02/image.",
      "source_url": "https://docs.x.ai/developers/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-imagine-image",
      "name": "Grok Imagine Image",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-imagine-image",
      "notes": "Standard image-generation model priced per image ($0.02/image), not per token. Token price fields null by design.",
      "source_url": "https://docs.x.ai/developers/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-imagine-video-1-5",
      "name": "Grok Imagine Video 1.5",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-imagine-video-1.5",
      "notes": "Video-generation model priced per second of output ($0.080/sec), not per token. Token price fields null by design.",
      "source_url": "https://docs.x.ai/developers/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-imagine-video",
      "name": "Grok Imagine Video",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-imagine-video",
      "notes": "Video-generation model priced per second ($0.050/sec), not per token. Token price fields null by design.",
      "source_url": "https://docs.x.ai/developers/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-0709",
      "name": "Grok 4 (0709)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-11",
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-0709",
      "notes": "Retired from the xAI API on 2026-05-15 12:00 PM PT per official migration page. Requests to this slug now auto-redirect to grok-4.3 with 'low' reasoning effort; the slug still resolves so existing code does not break. Knowledge cutoff Nov 2024 per docs note covering Grok 4. Per-token pricing no longer published on docs (was historically $3/$15 per M per third-party trackers); left null as not officially current.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-3",
      "name": "Grok 3",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "text-out"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-11",
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-3",
      "notes": "Retired from the xAI API on 2026-05-15 12:00 PM PT per official migration page; redirects to grok-4.3 with 'none' reasoning effort, slug still resolves. ~131K context window per third-party sources (not re-confirmed on current docs). Knowledge cutoff Nov 2024 per docs. Per-token pricing no longer published on docs; left null.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-fast-reasoning",
      "name": "Grok 4 Fast (Reasoning)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-fast-reasoning",
      "notes": "Retired from xAI API on 2026-05-15 12:00 PM PT; redirects to grok-4.3 with 'low' reasoning effort, slug still resolves. Specs/pricing no longer on docs; left null. (Oracle OCI mirror lists deprecated 2026-05-15, retires 2026-08-15 on their platform, but the xAI-native retirement date is 2026-05-15.)",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-fast-non-reasoning",
      "name": "Grok 4 Fast (Non-Reasoning)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-fast-non-reasoning",
      "notes": "Retired from xAI API on 2026-05-15 12:00 PM PT; redirects to grok-4.3 with 'none' reasoning effort, slug still resolves. Specs/pricing no longer on docs; left null.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-1-fast-reasoning",
      "name": "Grok 4.1 Fast (Reasoning)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-1-fast-reasoning",
      "notes": "Retired from xAI API on 2026-05-15 12:00 PM PT; redirects to grok-4.3 with 'low' reasoning effort, slug still resolves. Specs/pricing no longer on docs; left null. (Was ~$0.20/$0.50 per M per third-party trackers before retirement; not officially current.)",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-4-1-fast-non-reasoning",
      "name": "Grok 4.1 Fast (Non-Reasoning)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-1-fast-non-reasoning",
      "notes": "Retired from xAI API on 2026-05-15 12:00 PM PT; redirects to grok-4.3 with 'none' reasoning effort, slug still resolves. Specs/pricing no longer on docs; left null.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-code-fast-1",
      "name": "Grok Code Fast 1",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-build-0.1",
      "api_string": "grok-code-fast-1",
      "notes": "Retired as a standalone slug on 2026-05-15 12:00 PM PT; redirects to grok-build-0.1. Note grok-code-fast-1 also persists as an ALIAS of grok-build-0.1 on the current model page, so the name still resolves. 256k context window inferred from grok-build-0.1 (same model lineage).",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "grok-imagine-image-pro",
      "name": "Grok Imagine Image Pro",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-imagine-image-quality",
      "api_string": "grok-imagine-image-pro",
      "notes": "Image-generation model retired on 2026-05-15 12:00 PM PT; redirects to grok-imagine-image-quality. Priced per image, not per token, so token fields null.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "llama-4-maverick",
      "name": "Llama 4 Maverick (17B-128E Instruct)",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": 0.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-08",
      "released": "2025-04-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
      "notes": "Open-weight, natively multimodal MoE: 17B active / 400B total params, 128 experts. License: Llama 4 Community License Agreement (commercial use permitted for orgs with <700M MAU). Meta is the model owner; no first-party Meta API pricing for self-host. Official Llama API model ID is 'Llama-4-Maverick-17B-128E-Instruct-FP8' and the Llama API serves it at a 128k context window (developer.meta.com/Llama API docs), whereas the open weights support up to 1M tokens. Hosted price shown is OpenRouter slug 'meta-llama/llama-4-maverick' = $0.15 in / $0.60 out per 1M (OpenRouter page accessed 2026-06-28).",
      "source_url": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-4-scout",
      "name": "Llama 4 Scout (17B-16E Instruct)",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 10000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-08",
      "released": "2025-04-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-4-Scout-17B-16E-Instruct-FP8",
      "notes": "Open-weight, natively multimodal MoE: 17B active / 109B total params, 16 experts; fits on a single H100. License: Llama 4 Community License Agreement. Open weights support up to 10M-token context; the official Llama API serves it at 128k (model ID 'Llama-4-Scout-17B-16E-Instruct-FP8'). Hosted price is OpenRouter slug 'meta-llama/llama-4-scout' = $0.10 in / $0.30 out per 1M (page accessed 2026-06-28); Together AI reported ~$0.08 in / $0.30 out per 1M. No cached-input discount published. Same 12 supported languages as Maverick.",
      "source_url": "https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-4-behemoth",
      "name": "Llama 4 Behemoth (preview)",
      "provider": "Meta",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": null,
      "notes": "Announced as a preview/teacher model (~288B active params, 16 experts, ~2T total) used to distill Scout and Maverick. As of 2026-06-28 it is NOT released as open weights and has no public API string, pricing, context window, or confirmed release date. Included only because it is a currently-announced member of the Llama 4 herd. All numeric fields null (unconfirmed).",
      "source_url": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-3-3-70b-instruct",
      "name": "Llama 3.3 70B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.32,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-12-06",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.3-70B-Instruct",
      "notes": "Open-weight, text-only, 70B dense (GQA), 128k context. License: Llama 3.3 Community License Agreement. Also a first-party Llama API model ID 'Llama-3.3-70B-Instruct'. Hosted price = OpenRouter slug 'meta-llama/llama-3.3-70b-instruct' $0.10 in / $0.32 out per 1M (accessed 2026-06-28); Together AI lists ~$0.88-$1.04 per 1M flat (Together pricing page, accessed 2026-06-28). Free tier also exists on OpenRouter. Langs: English, German, French, Italian, Portuguese, Hindi, Spanish, Thai.",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-3-3-8b-instruct",
      "name": "Llama 3.3 8B Instruct (Llama API)",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "Llama-3.3-8B-Instruct",
      "notes": "Listed on the official Meta Llama API models page as a lightweight, ultra-fast text-only variant with 128k context (model ID 'Llama-3.3-8B-Instruct'). NOTE/UNCERTAINTY: there is no corresponding standalone 'Llama-3.3-8B' open-weight checkpoint on Meta's Hugging Face org (the 8B open weight in this generation is Llama-3.1-8B-Instruct); this ID appears specific to the hosted Llama API. Pricing, exact release date, and knowledge cutoff not published on the API models page (null). Not separately priced on OpenRouter/Together under this exact name.",
      "source_url": "https://llama.developer.meta.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-3-1-405b-instruct",
      "name": "Llama 3.1 405B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-07-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.1-405B-Instruct",
      "notes": "Open-weight, text-only, 405B dense, 128k context. License: Llama 3.1 Community License. Released 2024-07-23 alongside 8B/70B. Hosted price NOT reliably extractable in this pass: OpenRouter slug is 'meta-llama/llama-3.1-405b-instruct' (131k ctx) but the per-token figures did not render; Together AI reportedly does NOT serve 405B on its serverless API (their 405B model page states it is not available serverless), and third-party reports put serverless 405B around $3.00-$3.50 per 1M elsewhere (unverified). Prices left null rather than guessed. Source URL is the Llama 3.1 collection card (lists 8B",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-3-1-70b-instruct",
      "name": "Llama 3.1 70B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-07-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "llama-3-3-70b-instruct",
      "api_string": "meta-llama/Llama-3.1-70B-Instruct",
      "notes": "Open-weight, text-only, 70B dense, 128k context. License: Llama 3.1 Community License. Largely superseded by Llama 3.3 70B Instruct (same size, improved quality, Dec 2024) which Meta positions as the recommended 70B; not formally 'deprecated/retired' (open weights remain downloadable), so 'replacement' is advisory, not an enforced retirement. retires_on field repurposed here to point to the recommended successor id (no official retirement date exists - treat as null date). Per-token hosted price not separately captured this pass (null).",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-3-1-8b-instruct",
      "name": "Llama 3.1 8B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.02,
      "price_output_per_mtok": 0.03,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-07-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.1-8B-Instruct",
      "notes": "Open-weight, text-only, 8B dense, 128k context. License: Llama 3.1 Community License. Hosted price = OpenRouter slug 'meta-llama/llama-3.1-8b-instruct' $0.02 in / $0.03 out per 1M (page accessed 2026-06-28; 131k ctx). Together AI reports ~$0.18 per 1M flat. Free tier also available. Langs: English, German, French, Italian, Portuguese, Hindi, Spanish, Thai.",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-3-2-90b-vision-instruct",
      "name": "Llama 3.2 90B Vision Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.2-90B-Vision-Instruct",
      "notes": "Open-weight multimodal (text+image), ~88.8B params, 128k context. License: Llama 3.2 Community License. Source is the 3.2 11B Vision card which documents the 90B sibling (same architecture/128k ctx/Dec-2023 cutoff). Image+text tasks are English-primary; text-only adds German, French, Italian, Portuguese, Hindi, Spanish, Thai. Per-token hosted price not separately captured this pass (null). Dedicated card: huggingface.co/meta-llama/Llama-3.2-90B-Vision-Instruct.",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-3-2-11b-vision-instruct",
      "name": "Llama 3.2 11B Vision Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.2-11B-Vision-Instruct",
      "notes": "Open-weight multimodal (text+image), 10.6B params, 128k context. License: Llama 3.2 Community License. Image+text tasks English-primary; text-only adds German, French, Italian, Portuguese, Hindi, Spanish, Thai. Per-token hosted price not separately captured this pass (null).",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-3-2-3b-instruct",
      "name": "Llama 3.2 3B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.2-3B-Instruct",
      "notes": "Open-weight, text-only lightweight/on-device model, 128k context. License: Llama 3.2 Community License. Referenced on the 3.2 collection/11B card and has a dedicated card at huggingface.co/meta-llama/Llama-3.2-3B-Instruct. Free tier exists on OpenRouter (slug 'meta-llama/llama-3.2-3b-instruct'); paid per-token figures not captured this pass (null).",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "llama-3-2-1b-instruct",
      "name": "Llama 3.2 1B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.2-1B-Instruct",
      "notes": "Open-weight, text-only smallest on-device model, 128k context. License: Llama 3.2 Community License. Dedicated card at huggingface.co/meta-llama/Llama-3.2-1B-Instruct. Per-token hosted price not captured this pass (null).",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "mistral-large-3",
      "name": "Mistral Large 3",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "mistral-large-2512",
      "notes": "Current flagship (v25.12). 'latest' alias mistral-large-latest -> mistral-large-2512. Open-weight, sparse MoE 41B active / 675B total. 256k context per official model page. Pricing from mistral.ai/pricing (page fetched 2026-06-28): $0.5/M in, $1.5/M out. Max output and knowledge cutoff not published on official pages.",
      "source_url": "https://docs.mistral.ai/models/mistral-large-3-25-12",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "mistral-medium-3-5",
      "name": "Mistral Medium 3.5",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.5,
      "price_output_per_mtok": 7.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "mistral-medium-latest",
      "notes": "Current frontier multimodal model (v26.04), agentic/coding focus. Dated alias resolves to mistral-medium-26.04 (full dated string not explicitly published; mistral-medium-2508 is the now-deprecated 3.1). Pricing $1.5/M in, $7.5/M out (mistral.ai/pricing, 2026-06-28). Listed because it is the named replacement for several deprecated models (Large 2.1, Pixtral Large). Context window not published on official overview.",
      "source_url": "https://mistral.ai/pricing/",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "mistral-small-4",
      "name": "Mistral Small 4",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": 0.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-03-16",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "mistral-small-2603",
      "notes": "Current Small (v26.03). 'latest' alias mistral-small-latest -> mistral-small-2603. Hybrid instruct/reasoning/coding, 119B params (6.5B active), 256k context. Open-weight (Open v26.03). Pricing $0.15/M in, $0.6/M out (mistral.ai/pricing, 2026-06-28).",
      "source_url": "https://docs.mistral.ai/models/model-cards/mistral-small-4-0-26-03",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "mistral-small-3-2",
      "name": "Mistral Small 3.2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-30",
      "retires_on": "2026-07-31",
      "replacement": "mistral-small-2603",
      "api_string": "mistral-small-2506",
      "notes": "v25.06. Deprecated 2026-04-30, retirement 2026-07-31, replaced by Mistral Small 4. Pricing no longer listed on current pricing page.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "codestral",
      "name": "Codestral (v25.08)",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "code"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 0.9,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "codestral-2508",
      "notes": "Current Codestral (Premier v25.08). 'latest' alias codestral-latest -> codestral-2508. Code completion / FIM, 128k context. Pricing $0.3/M in, $0.9/M out (mistral.ai/pricing, 2026-06-28). Release date 'July 30, 2025' per model page.",
      "source_url": "https://docs.mistral.ai/models/codestral-25-08",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "codestral-2501",
      "name": "Codestral (v25.01)",
      "provider": "Mistral",
      "status": "retired",
      "modality": [
        "text",
        "code"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-11-06",
      "retires_on": "2025-11-30",
      "replacement": "codestral-2508",
      "api_string": "codestral-2501",
      "notes": "Deprecated 2025-11-06, retired 2025-11-30, replaced by Codestral v25.08. An earlier codestral-2405 alias also exists in the legacy list.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "codestral-embed",
      "name": "Codestral Embed",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "code"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "codestral-embed-2505",
      "notes": "Premier code-embedding model (v25.05). Embeddings only (input-priced): $0.15/M input tokens (mistral.ai/pricing, 2026-06-28). Max input 8,192 tokens (contextLength '8k' in the open-source docs schema codestral-embed-25-05.ts). Output dimension configurable via output_dimension: default 1536, max 3072 (docs.mistral.ai/capabilities/embeddings/code_embeddings). releaseDate 2025-05-28 per the docs schema.",
      "source_url": "https://docs.mistral.ai/models/codestral-embed-25-05",
      "open_weight": false,
      "embedding_dimensions": 1536
    },
    {
      "id": "pixtral-large",
      "name": "Pixtral Large",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-11-18",
      "deprecated_on": "2026-02-27",
      "retires_on": "2026-05-31",
      "replacement": "mistral-medium-latest",
      "api_string": "pixtral-large-2411",
      "notes": "First frontier-class multimodal model (v24.11), 128k context. Deprecated 2026-02-27, retirement 2026-05-31, replaced by Mistral Medium 3.5. As of 2026-06-28 it is past its retirement date and no longer on the pricing page; classified deprecated/retiring per official legacy table. 'latest' alias pixtral-large-latest historically mapped here.",
      "source_url": "https://docs.mistral.ai/models/pixtral-large-24-11",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "pixtral-12b",
      "name": "Pixtral 12B",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-09",
      "deprecated_on": "2025-12-02",
      "retires_on": "2025-12-31",
      "replacement": "ministral-3-14b-latest",
      "api_string": "pixtral-12b-2409",
      "notes": "Open-weight 12B vision model (v24.09). Deprecated 2025-12-02, retired 2025-12-31, replaced by Ministral 3 14B. Past retirement as of 2026-06-28.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "ministral-3-14b",
      "name": "Ministral 3 14B",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": 0.2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "ministral-3-14b-latest",
      "notes": "Ministral 3 family (v25.12), open-weight Apache 2.0, text+vision, 40+ languages. Pricing $0.2/M in & out (mistral.ai/pricing, 2026-06-28, listed as 'Ministral 14B'). Context window not published on official overview (Large 3 / Small 4 are 256k; Ministral 3 not explicitly stated). Dated alias expected ministral-3-14b-2512.",
      "source_url": "https://mistral.ai/news/mistral-3/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "ministral-3-8b",
      "name": "Ministral 3 8B",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": 0.15,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "ministral-3-8b-latest",
      "notes": "Ministral 3 family (v25.12), open-weight Apache 2.0, text+vision, edge deployment. Official model page states 256k context. Pricing $0.15/M in & out (mistral.ai/pricing 'Ministral 8B', 2026-06-28). Dated alias on page shown as ministral-8b-2512 / ministral-3-8b-2512.",
      "source_url": "https://docs.mistral.ai/models/ministral-3-8b-25-12",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "ministral-3-3b",
      "name": "Ministral 3 3B",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.1,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "ministral-3-3b-latest",
      "notes": "Smallest Ministral 3 (v25.12), open-weight Apache 2.0, text+vision. Pricing $0.10/M in & out per the official API pricing page (mistral.ai/pricing/api), verified 2026-07-09. CORRECTION: previously stored as $0.04/M from a weak general-pricing 'cheapest tier' search corroboration — the authoritative per-model API table lists $0.10/M in & out. Context window not published on official overview. Dated alias expected ministral-3-3b-2512.",
      "source_url": "https://mistral.ai/pricing/api/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "ministral-8b-2410",
      "name": "Ministral 8B (2410)",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-10",
      "deprecated_on": "2025-12-02",
      "retires_on": "2025-12-31",
      "replacement": "ministral-3-8b-latest",
      "api_string": "ministral-8b-2410",
      "notes": "Original Ministral 8B (v24.10). Deprecated 2025-12-02, retired 2025-12-31, replaced by Ministral 3 8B. Past retirement as of 2026-06-28.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "ministral-3b-2410",
      "name": "Ministral 3B (2410)",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-10",
      "deprecated_on": "2025-12-02",
      "retires_on": "2025-12-31",
      "replacement": "ministral-3-3b-latest",
      "api_string": "ministral-3b-2410",
      "notes": "Original Ministral 3B (v24.10). Deprecated 2025-12-02, retired 2025-12-31, replaced by Ministral 3 3B. Past retirement as of 2026-06-28.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "mistral-large-2-1",
      "name": "Mistral Large 2.1",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-11",
      "deprecated_on": "2026-02-27",
      "retires_on": "2026-05-31",
      "replacement": "mistral-medium-latest",
      "api_string": "mistral-large-2411",
      "notes": "v24.11. Deprecated 2026-02-27, retirement 2026-05-31, replaced by Mistral Medium 3.5. Past retirement as of 2026-06-28.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "mistral-large-2-0",
      "name": "Mistral Large 2.0",
      "provider": "Mistral",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-07",
      "deprecated_on": "2024-11-30",
      "retires_on": "2025-03-30",
      "replacement": "mistral-large-2512",
      "api_string": "mistral-large-2407",
      "notes": "v24.07. Deprecated 2024-11-30, retired 2025-03-30, replacement Mistral Large 3.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "mistral-medium-3-1",
      "name": "Mistral Medium 3.1",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-08",
      "deprecated_on": "2026-05-22",
      "retires_on": "2026-08-31",
      "replacement": "mistral-medium-latest",
      "api_string": "mistral-medium-2508",
      "notes": "v25.08. Deprecated 2026-05-22, retirement 2026-08-31, replaced by Mistral Medium 3.5. Still in deprecation window as of 2026-06-28.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "mistral-nemo",
      "name": "Mistral NeMo",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-07",
      "deprecated_on": "2026-05-22",
      "retires_on": "2026-07-31",
      "replacement": "ministral-3-8b-latest",
      "api_string": "open-mistral-nemo-2407",
      "notes": "Open-weight 12B (v24.07, with NVIDIA). Deprecated 2026-05-22, retirement 2026-07-31, replaced by Ministral 3 8B. Still in deprecation window as of 2026-06-28; pricing no longer on current page.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "mixtral-8x22b",
      "name": "Mixtral 8x22B",
      "provider": "Mistral",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 64000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-04",
      "deprecated_on": "2024-11-30",
      "retires_on": "2025-03-30",
      "replacement": "mistral-small-2603",
      "api_string": "open-mixtral-8x22b",
      "notes": "Open-weight MoE. Deprecated 2024-11-30, retired 2025-03-30, replacement Mistral Small 4. Context 64k commonly cited but not confirmed on the legacy table; treat as approximate.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "mixtral-8x7b",
      "name": "Mixtral 8x7B",
      "provider": "Mistral",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 32000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2023-12",
      "deprecated_on": "2024-11-30",
      "retires_on": "2025-03-30",
      "replacement": "mistral-small-2603",
      "api_string": "open-mixtral-8x7b",
      "notes": "Open-weight MoE. Deprecated 2024-11-30, retired 2025-03-30, replacement Mistral Small 4. Context 32k commonly cited but not confirmed on the legacy table; treat as approximate.",
      "source_url": "https://docs.mistral.ai/getting-started/models/models_overview/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "magistral-medium-1-2",
      "name": "Magistral Medium 1.2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-18",
      "deprecated_on": "2026-05-22",
      "retires_on": "2026-07-31",
      "replacement": "mistral-medium-3-5",
      "api_string": "magistral-medium-2509",
      "notes": "Mistral's frontier multimodal reasoning model (v25.09); alias magistral-medium-latest -> magistral-medium-2509. Premier tier: API-only, no published weights. Text + image in, reasoning + text out. DEPRECATED 2026-05-22, RETIRES 2026-07-31 -> Mistral Medium 3.5, as Mistral folds its specialist reasoning line back into its general-purpose models. Lifecycle, price and context from Mistral's official docs model schema; $2/M in, $5/M out corroborated by the per-model table at mistral.ai/pricing/api, which still lists it as a current offering (the docs schema is the more precise surface for lifecycle). Max output and knowledge cutoff not published -> null.",
      "source_url": "https://docs.mistral.ai/models/magistral-medium-1-2-25-09",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "magistral-small-1-2",
      "name": "Magistral Small 1.2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-18",
      "deprecated_on": "2026-04-30",
      "retires_on": "2026-07-31",
      "replacement": "mistral-small-4",
      "api_string": "magistral-small-2509",
      "notes": "Mistral's small multimodal reasoning model (v25.09); alias magistral-small-latest -> magistral-small-2509. Open weights: Apache 2.0, 24B dense (huggingface.co/mistralai/Magistral-Small-2509). Text + image in, reasoning + text out. DEPRECATED 2026-04-30, RETIRES 2026-07-31 -> Mistral Small 4. Lifecycle, price and context from Mistral's official docs model schema; $0.5/M in, $1.50/M out corroborated by mistral.ai/pricing/api. Max output and knowledge cutoff not published -> null.",
      "source_url": "https://docs.mistral.ai/models/magistral-small-1-2-25-09",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "devstral-2",
      "name": "Devstral 2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-09",
      "deprecated_on": "2026-05-22",
      "retires_on": "2026-07-31",
      "replacement": "mistral-medium-3-5",
      "api_string": "devstral-2512",
      "notes": "Mistral's frontier agentic-coding model for software-engineering tasks (v25.12); aliases devstral-latest and devstral-medium-latest -> devstral-2512. Open weights: Modified MIT, 123B (huggingface.co/mistralai/Devstral-2-123B-Instruct-2512). Text in, text out. DEPRECATED 2026-05-22, RETIRES 2026-07-31 -> Mistral Medium 3.5, as Mistral folds its specialist coding line back into its general-purpose models. Lifecycle, price and context from Mistral's official docs model schema; $0.40/M in, $2/M out corroborated by mistral.ai/pricing/api, which still lists it as a current offering. Max output and knowledge cutoff not published -> null.",
      "source_url": "https://docs.mistral.ai/models/devstral-2-25-12",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "devstral-small-2",
      "name": "Devstral Small 2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-09",
      "deprecated_on": "2026-02-27",
      "retires_on": "2026-03-31",
      "replacement": "mistral-medium-3-5",
      "api_string": "labs-devstral-small-2512",
      "notes": "Small agentic-coding model (v25.12), Labs tier; alias devstral-small-latest -> labs-devstral-small-2512. Open weights: Apache 2.0, 24B. Text + image in, text out. DEPRECATED 2026-02-27 with a stated retirement of 2026-03-31 -> Mistral Medium 3.5. That date has passed, but Mistral still lists the model as deprecated rather than retired and still prices it publicly, so we record the status Mistral states rather than inferring a retirement it has not declared. Price and context from the official docs model schema, corroborated by mistral.ai/pricing/api. Max output and knowledge cutoff not published -> null.",
      "source_url": "https://docs.mistral.ai/models/devstral-small-2-25-12",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "labs-leanstral-1-5",
      "name": "Leanstral 1.5",
      "provider": "Mistral",
      "status": "preview",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 0,
      "price_output_per_mtok": 0,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-06-30",
      "deprecated_on": null,
      "retires_on": "2026-09-30",
      "replacement": null,
      "api_string": "labs-leanstral-1-5",
      "notes": "Specialist model for Lean 4 formal proof engineering, automated theorem proving and autoformalization -> not a general chat model. Labs tier, sparse MoE 119B total / 6.5B active. FREE during the Labs preview: pricing.free=true, $0/M in and out per the official docs model schema. Open weights: Apache 2.0 (huggingface.co/mistralai/Leanstral-1.5-119B-A6B). Text + image in, text out; 256K context, 128K max output. Mistral marks it status Active in its experimental Labs tier with a stated retirement of 2026-09-30 -> we normalize Labs/experimental to preview (it is not a GA production model), which also keeps the catalog's first $0 model out of the GA 'cheapest' rankings where a free specialist would rank misleadingly. Knowledge cutoff not published -> null.",
      "source_url": "https://mistral.ai/news/leanstral-1-5/",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "deepseek-v4-flash",
      "name": "DeepSeek-V4-Flash",
      "provider": "DeepSeek",
      "status": "preview",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "price_input_per_mtok": 0.14,
      "price_output_per_mtok": 0.28,
      "price_cached_input_per_mtok": 0.0028,
      "knowledge_cutoff": null,
      "released": "2026-04-24",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "deepseek-v4-flash",
      "notes": "Smaller/cheaper V4 model (~284B total / ~13B active params per authoritative third-party reports). Context length 1M, max output 384K tokens. Input price $0.14/M cache-miss, $0.0028/M cache-hit; output $0.28/M (USD). Supports dual modes (Thinking / Non-Thinking), JSON output, tool calls, chat prefix completion; FIM completion is non-thinking-mode only. Concurrency limit 2500. The legacy aliases deepseek-chat and deepseek-reasoner currently route to this model (non-thinking / thinking respectively). Part of the 'DeepSeek V4 Preview' generation (released 2026-04-24), hence status=preview. Knowle",
      "source_url": "https://api-docs.deepseek.com/quick_start/pricing",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "deepseek-v4-pro",
      "name": "DeepSeek-V4-Pro",
      "provider": "DeepSeek",
      "status": "preview",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "price_input_per_mtok": 0.435,
      "price_output_per_mtok": 0.87,
      "price_cached_input_per_mtok": 0.003625,
      "knowledge_cutoff": null,
      "released": "2026-04-24",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "deepseek-v4-pro",
      "notes": "Larger/most-capable V4 model (~1.6T total / ~49B active params per HuggingFace model card + authoritative third-party reports; MIT License; mixed FP4/FP8). Context length 1M, max output 384K tokens. Input price $0.435/M cache-miss, $0.003625/M cache-hit; output $0.87/M (USD). Supports three reasoning-effort modes (non-think / think high / think max), JSON output, tool calls; FIM completion non-thinking-mode only. Concurrency limit 500. Part of the 'DeepSeek V4 Preview' generation (released 2026-04-24), hence status=preview. Knowledge cutoff NOT officially published by DeepSeek -> left null. Pr",
      "source_url": "https://api-docs.deepseek.com/quick_start/pricing",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "deepseek-chat",
      "name": "deepseek-chat (legacy alias)",
      "provider": "DeepSeek",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "price_input_per_mtok": 0.14,
      "price_output_per_mtok": 0.28,
      "price_cached_input_per_mtok": 0.0028,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-07-24",
      "replacement": "deepseek-v4-flash",
      "api_string": "deepseek-chat",
      "notes": "Legacy model-name alias, NOT a separate model. Official docs: 'deepseek-chat & deepseek-reasoner will be fully retired and inaccessible after Jul 24th, 2026, 15:59 (UTC Time). (Currently routing to deepseek-v4-flash non-thinking/thinking).' deepseek-chat = NON-thinking mode of deepseek-v4-flash. Exact retirement timestamp: 2026-07-24 15:59 UTC. Replacement: use explicit name deepseek-v4-flash (non-thinking). Pricing shown matches deepseek-v4-flash since it routes there. This alias historically mapped to DeepSeek-V3-series non-thinking; as of the V4 Preview it routes to V4-Flash.",
      "source_url": "https://api-docs.deepseek.com/news/news260424",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "deepseek-reasoner",
      "name": "deepseek-reasoner (legacy alias)",
      "provider": "DeepSeek",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "price_input_per_mtok": 0.14,
      "price_output_per_mtok": 0.28,
      "price_cached_input_per_mtok": 0.0028,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-07-24",
      "replacement": "deepseek-v4-flash",
      "api_string": "deepseek-reasoner",
      "notes": "Legacy model-name alias, NOT a separate model. Official docs: 'deepseek-chat & deepseek-reasoner will be fully retired and inaccessible after Jul 24th, 2026, 15:59 (UTC Time). (Currently routing to deepseek-v4-flash non-thinking/thinking).' deepseek-reasoner = THINKING mode of deepseek-v4-flash. Exact retirement timestamp: 2026-07-24 15:59 UTC. Replacement: use explicit name deepseek-v4-flash (thinking mode). Pricing shown matches deepseek-v4-flash since it routes there. This alias historically mapped to DeepSeek-R1 / V3-series thinking; as of the V4 Preview it routes to V4-Flash thinking.",
      "source_url": "https://api-docs.deepseek.com/news/news260424",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-a-plus-05-2026",
      "name": "Command A Plus",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-plus-05-2026",
      "notes": "Cohere's first Mixture-of-Experts model; combines vision input, agentic/reasoning, and world-class translation. Endpoint: Chat. Per-token pricing NOT publicly listed (premium Command A variant; gated behind sales@cohere.com per multiple 2026 trackers), so input/output prices set null rather than guessed. Context 128k / max output 64k confirmed from docs.cohere.com/docs/models.md. Knowledge cutoff not published by Cohere.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-a-03-2025",
      "name": "Command A",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 256000,
      "max_output_tokens": 8000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-03",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-03-2025",
      "notes": "Flagship general model (111B params), 23 languages, tool use/RAG/agents. Context 256k / max output 8k from docs.cohere.com/docs/models.md. Price $2.50 in / $10.00 out per 1M corroborated across metacto (updated May 2026), eesel.ai, pecollective, aipricing.guru. cohere.com/pricing main table is JS-rendered and could not be read directly by fetch. Knowledge cutoff not published. No published cached-input price.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-a-reasoning-08-2025",
      "name": "Command A Reasoning",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 256000,
      "max_output_tokens": 32000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-reasoning-08-2025",
      "notes": "Cohere's first reasoning model ('thinks' before generating). Context 256k / max output 32k from docs.cohere.com/docs/models.md. Per-token pricing not publicly listed (sales-gated per eesel.ai/pricepertoken 2026 notes) -> null. Knowledge cutoff not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-a-vision-07-2025",
      "name": "Command A Vision",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": 8000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-vision-07-2025",
      "notes": "First Cohere model capable of processing images (enterprise image analysis). Context 128k / max output 8k from docs.cohere.com/docs/models.md. Per-token pricing not publicly listed (sales-gated) -> null. Knowledge cutoff not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-a-translate-08-2025",
      "name": "Command A Translate",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8000,
      "max_output_tokens": 8000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-translate-08-2025",
      "notes": "State-of-the-art machine translation model covering 23 languages. Context 8k / max output 8k from docs.cohere.com/docs/models.md. Per-token pricing not publicly listed (sales-gated) -> null. Knowledge cutoff not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-r7b-12-2024",
      "name": "Command R7B",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 0.0375,
      "price_output_per_mtok": 0.15,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-12",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-r7b-12-2024",
      "notes": "Smallest/fastest model in R series; strong RAG & tool use. Context 128k / max output 4k from docs.cohere.com/docs/models.md. Price $0.0375 in / $0.15 out per 1M corroborated by metacto (May 2026) and eesel.ai. Knowledge cutoff not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-r-plus-08-2024",
      "name": "Command R+ (08-2024)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-r-plus-08-2024",
      "notes": "Still listed as available. Price $2.50 in / $10.00 out per 1M from cohere.com/pricing legacy table (rendered) and metacto/eesel. Context 128k / max output 4k from docs.cohere.com/docs/models.md. Knowledge cutoff not published.",
      "source_url": "https://cohere.com/pricing",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-r-08-2024",
      "name": "Command R (08-2024)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": 0.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-r-08-2024",
      "notes": "Still listed as available. Price $0.15 in / $0.60 out per 1M corroborated by metacto (May 2026) and eesel.ai. Context 128k / max output 4k from docs.cohere.com/docs/models.md. Knowledge cutoff not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-r-plus-04-2024",
      "name": "Command R+ (04-2024)",
      "provider": "Cohere",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-04",
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "command-r-plus-08-2024",
      "api_string": "command-r-plus-04-2024",
      "notes": "Deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md (alias 'command-r-plus'). Replacement: command-r-plus-08-2024 or command-a-03-2025. Legacy price $3.00 in / $15.00 out per 1M from cohere.com/pricing legacy table. Effectively retired.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command-r-03-2024",
      "name": "Command R (03-2024)",
      "provider": "Cohere",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-03",
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "command-r-08-2024",
      "api_string": "command-r-03-2024",
      "notes": "Deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md (alias 'command-r'). Replacement: command-r-08-2024 or command-a-03-2025. Legacy price $0.50 in / $1.50 out per 1M from cohere.com/pricing legacy table. Associated fine-tuned models were deprecated/shut down earlier on 2025-03-08.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "command",
      "name": "Command (legacy)",
      "provider": "Cohere",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 4096,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "command-r-08-2024",
      "api_string": "command",
      "notes": "Original Command generative model. Deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md. Replacement: command-r-08-2024. Legacy price $1.00 in / $2.00 out per 1M from cohere.com/pricing. Context window not specified on current docs; 4096 is the historical value (low confidence) - treat as approximate.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "command-light",
      "name": "Command Light (legacy)",
      "provider": "Cohere",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 4096,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 0.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "command-r-08-2024",
      "api_string": "command-light",
      "notes": "Smaller/faster legacy Command. Deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md. Replacement: command-r-08-2024. Legacy price $0.30 in / $0.60 out per 1M from cohere.com/pricing. Context window historical (~4096), low confidence.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "embed-v4-0",
      "name": "Embed 4 (embed-v4.0)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.12,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-v4.0",
      "notes": "Multimodal embedding model: text, images, PDFs. 128k context; output dimensions configurable 256-1536. Embedding model so no output-token price. Price $0.12 per 1M text input tokens; image embeddings ~$0.47 per 1M image tokens (per metacto/eesel/pecollective 2026). Embedding models have no 'output token' concept; image price recorded in notes only. Knowledge cutoff/release date not published in docs.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": 1536
    },
    {
      "id": "embed-english-v3-0",
      "name": "Embed English v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 512,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-english-v3.0",
      "notes": "English embedding model, 1024 dims, 512-token context. Endpoints: Embed, Embed Jobs. Text embed price commonly $0.10 per 1M tokens (v3 family, per aipricing.guru/eesel) - moderate confidence; image input ~$0.47/1M per Cohere v3 image-embed pricing. Knowledge cutoff/release date not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": 1024
    },
    {
      "id": "embed-english-light-v3-0",
      "name": "Embed English Light v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 512,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-english-light-v3.0",
      "notes": "Smaller/faster English embedding model, 384 dims, 512-token context. Text embed price ~$0.10 per 1M tokens (v3 family) - moderate confidence; not separately broken out on rendered pricing table fetched. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": 384
    },
    {
      "id": "embed-multilingual-v3-0",
      "name": "Embed Multilingual v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 512,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-multilingual-v3.0",
      "notes": "Multilingual embedding model, 1024 dims, 512-token context. Text embed price ~$0.10 per 1M tokens (v3 family) - moderate confidence. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": 1024
    },
    {
      "id": "embed-multilingual-light-v3-0",
      "name": "Embed Multilingual Light v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 512,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-multilingual-light-v3.0",
      "notes": "Smaller/faster multilingual embedding model, 384 dims, 512-token context. Text embed price ~$0.10 per 1M tokens (v3 family) - moderate confidence. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": 384
    },
    {
      "id": "rerank-v4-0-pro",
      "name": "Rerank 4 Pro (rerank-v4.0-pro)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 32000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-v4.0-pro",
      "notes": "Reranks English & non-English documents and semi-structured JSON. 32k context. Priced PER SEARCH (1 query + up to 100 docs), not per token: ~$0.0025 per search (per eesel.ai, pecollective, aipricing.guru, 2026). Per-token fields null because Rerank uses search-unit pricing. Cohere also offers dedicated 'Model Vault' instances: Medium $5.00/hr or $3,250/mo, Large $10.00/hr or $6,500/mo (cohere.com/pricing). Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "rerank-v4-0-fast",
      "name": "Rerank 4 Fast (rerank-v4.0-fast)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 32000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-v4.0-fast",
      "notes": "Low-latency 'light' version of Rerank 4. 32k context. Priced PER SEARCH (1 query + up to 100 docs): ~$0.002 per search (per eesel.ai, pecollective, aipricing.guru, 2026). Per-token fields null (search-unit pricing). Dedicated Model Vault Medium $5.00/hr or $3,250/mo (cohere.com/pricing). Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "rerank-v3-5",
      "name": "Rerank 3.5 (rerank-v3.5)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 4000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-v3.5",
      "notes": "4k context. Priced PER SEARCH (1 query + up to 100 docs): $0.001 per search = $2.00 per 1,000 searches (corroborated metacto May 2026, eesel.ai, pecollective). Per-token fields null (search-unit pricing). Dedicated Model Vault Medium $5.00/hr or $3,250/mo (cohere.com/pricing). Replaced rerank-english/multilingual-v2.0 (shut down 2025-04-30). Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "rerank-english-v3-0",
      "name": "Rerank English v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 4000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-english-v3.0",
      "notes": "English-specific reranking, 4k context. Priced per search ($0.001/search = $2.00/1000, same Rerank v3 tier per Cohere pricing). Per-token fields null (search-unit pricing). Still listed in docs.cohere.com/docs/models. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "rerank-multilingual-v3-0",
      "name": "Rerank Multilingual v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 4000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-multilingual-v3.0",
      "notes": "Non-English reranking, 4k context. Priced per search ($0.001/search = $2.00/1000, same Rerank v3 tier). Per-token fields null (search-unit pricing). Still listed in docs.cohere.com/docs/models. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "rerank-english-v2-0",
      "name": "Rerank English v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2024-12-02",
      "retires_on": "2025-04-30",
      "replacement": "rerank-v3.5",
      "api_string": "rerank-english-v2.0",
      "notes": "Deprecated 2024-12-02, shut down 2025-04-30 per docs.cohere.com/docs/deprecations.md. Replacement: rerank-v3.5. Retired - no longer callable.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "rerank-multilingual-v2-0",
      "name": "Rerank Multilingual v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2024-12-02",
      "retires_on": "2025-04-30",
      "replacement": "rerank-v3.5",
      "api_string": "rerank-multilingual-v2.0",
      "notes": "Deprecated 2024-12-02, shut down 2025-04-30 per docs.cohere.com/docs/deprecations.md. Replacement: rerank-v3.5. Retired - no longer callable.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "embed-english-v2-0",
      "name": "Embed English v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-04",
      "retires_on": "2026-04-04",
      "replacement": "embed-v4.0",
      "api_string": "embed-english-v2.0",
      "notes": "Deprecated AND shut down same day 2026-04-04 per docs.cohere.com/docs/deprecations.md. Replacement: embed-english-v3.0 or embed-v4.0. Retired.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "embed-english-light-v2-0",
      "name": "Embed English Light v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-04",
      "retires_on": "2026-04-04",
      "replacement": "embed-v4.0",
      "api_string": "embed-english-light-v2.0",
      "notes": "Deprecated AND shut down 2026-04-04 per docs.cohere.com/docs/deprecations.md. Replacement: embed-english-v3.0 or embed-v4.0. Retired.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "embed-multilingual-v2-0",
      "name": "Embed Multilingual v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-04",
      "retires_on": "2026-04-04",
      "replacement": "embed-v4.0",
      "api_string": "embed-multilingual-v2.0",
      "notes": "Deprecated AND shut down 2026-04-04 per docs.cohere.com/docs/deprecations.md. Replacement: embed-multilingual-v3.0 or embed-v4.0. Retired.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "summarize",
      "name": "Summarize (legacy endpoint)",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "Use the /chat endpoint",
      "api_string": "summarize",
      "notes": "Legacy Summarize endpoint deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md. Replacement: use the Chat endpoint with a summarization prompt.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "amazon-nova-micro",
      "name": "Amazon Nova Micro",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 5000,
      "price_input_per_mtok": 0.035,
      "price_output_per_mtok": 0.14,
      "price_cached_input_per_mtok": 0.00875,
      "knowledge_cutoff": "2024-10",
      "released": "2024-12-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "amazon.nova-2-lite-v1:0",
      "api_string": "amazon.nova-micro-v1:0",
      "notes": "Text-only model (fastest/cheapest Nova). Lifecycle: Active. Model card states 'Model EOL date: No sooner than 12/4/2025' (a floor, not a fixed retirement; no firm retirement date published as of pricing data dated Aug 2025 / model cards current 2026-06). Geo inference IDs us./eu.amazon.nova-micro-v1:0. Per-1K equivalents: input $0.000035, output $0.00014. Cached input (cache read) = 25% of input = $0.00000875/1K, corroborated for Nova Micro by AWS prompt-caching coverage; cache-WRITE may be priced above standard input. Pricing corroborated by AWS Bedrock pricing page + multiple aggregators; AW",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-micro.html",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "amazon-nova-lite",
      "name": "Amazon Nova Lite",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 300000,
      "max_output_tokens": 5000,
      "price_input_per_mtok": 0.06,
      "price_output_per_mtok": 0.24,
      "price_cached_input_per_mtok": 0.015,
      "knowledge_cutoff": "2024-10",
      "released": "2024-12-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "amazon.nova-2-lite-v1:0",
      "api_string": "amazon.nova-lite-v1:0",
      "notes": "Low-cost multimodal (text, image, video input; text output). Lifecycle: Active. Model card states 'Model EOL date: No sooner than 12/4/2025' (floor, not fixed). Geo inference IDs us./eu.amazon.nova-lite-v1:0. Per-1K: input $0.00006, output $0.00024. Input/output prices confirmed directly on AWS blog 'Demystifying Amazon Bedrock Pricing' (dated 2025-08-11) and corroborated by aggregators. Cached input = 25% of input = $0.000015/1K (derived from AWS uniform Nova cache-read ratio; not separately listed per-model on a fetchable AWS page, so treat as approximate). AWS recommends migrating to Nova 2",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-lite.html",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "amazon-nova-pro",
      "name": "Amazon Nova Pro",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 300000,
      "max_output_tokens": 5000,
      "price_input_per_mtok": 0.8,
      "price_output_per_mtok": 3.2,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": "2024-10",
      "released": "2024-12-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "amazon.nova-2-lite-v1:0",
      "api_string": "amazon.nova-pro-v1:0",
      "notes": "Balanced multimodal (text, image, video input; text output). Lifecycle: Active. Model card states 'Model EOL date: No sooner than 12/4/2025' (floor). Supports Standard, Priority and Flex service tiers. Geo inference IDs us./eu.amazon.nova-pro-v1:0. Per-1K: input $0.0008, output $0.0032 — input/output confirmed on AWS blog 'Demystifying Amazon Bedrock Pricing' (2025-08-11). Cached input (cache read) = 25% of input = $0.0002/1K (= $0.20/M); cache-WRITE is priced at a premium (reported ~$1.00/M vs $0.80/M input). Cache-read figure derived from AWS Nova cache ratio, not separately fetched from a p",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-pro.html",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "amazon-nova-premier",
      "name": "Amazon Nova Premier",
      "provider": "Amazon",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 25000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 12.5,
      "price_cached_input_per_mtok": 0.625,
      "knowledge_cutoff": "2024-10",
      "released": "2025-10-31",
      "deprecated_on": null,
      "retires_on": "2026-09-14",
      "replacement": "amazon.nova-2-lite-v1:0",
      "api_string": "amazon.nova-premier-v1:0",
      "notes": "Most capable Nova 1 model: complex reasoning, agentic workflows, model distillation. Reasoning supported. 1M-token context, 25K max output. Model card lifecycle = 'Legacy' with 'Model EOL date: September 14, 2026' (firm retirement). Status set to 'deprecated' (= Legacy) given the firm EOL. NOTE: model card shows 'Model launch date: Oct 31, 2025' which conflicts with the public GA of Nova Premier (Apr/May 2025); the Oct 31 2025 date may reflect a card/version update rather than original GA — flagged as uncertain. Only us. geo inference ID (us.amazon.nova-premier-v1:0). Per-1K: input $0.0025, ou",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-premier.html",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "amazon-nova-2-lite",
      "name": "Amazon Nova 2 Lite",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2025-10",
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "amazon.nova-2-lite-v1:0",
      "notes": "Current-gen Nova 2 cost-efficient multimodal reasoning model (text, image, video input; text output). GA, Lifecycle: Active. Announced at re:Invent 2025. 1M context, 64K max output, knowledge cutoff Oct 2025. Supports extended thinking (low/medium/high), built-in code interpreter + web grounding, remote MCP tools. Model IDs: amazon.nova-2-lite-v1:0; geo IDs us./eu./jp.amazon.nova-2-lite-v1:0; global ID global.amazon.nova-2-lite-v1:0 (global cross-region inference). Per-1K: input $0.00030, output $0.0025 (= $0.30/M in, $2.50/M out), corroborated by aggregators citing AWS Bedrock pricing; AWS pr",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-2-lite.html",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "amazon-nova-2-pro",
      "name": "Amazon Nova 2 Pro",
      "provider": "Amazon",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": null,
      "notes": "Announced at re:Invent 2025 (Dec 2, 2025) as 'most intelligent' Nova 2 model for complex multistep tasks. Status: PREVIEW / early access (limited to Amazon Nova Forge customers as of announcement). No public Bedrock model card yet (fetch of model-card-amazon-nova-2-pro.html returned no content), so model ID, max output, knowledge cutoff and pricing are not yet officially published -> null. Context window stated as 1M tokens (shared Nova 2 family spec). Expected stable model ID pattern 'amazon.nova-2-pro-v1:0' but NOT confirmed on an official page -> left null to avoid guessing.",
      "source_url": "https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-7-max",
      "name": "Qwen3.7-Max",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 7.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-05-21",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.7-max",
      "notes": "Flagship Max model, API-only/proprietary (no open weights). Official International (Singapore) list price $2.5 in / $7.5 out per 1M tokens, single tier 0<token<=1M, Non-Thinking and Thinking modes; alias currently = qwen3.7-max-2026-05-20 (the 2026-06-08 snapshot added visual-modal understanding -> text+image+video). Thinking enabled by default; supports explicit/context cache (caching priced as a discount, no separate cached-input column published) and Function Calling. Max output: official Alibaba blog/config shows 65536; a benchmark footnote on the same blog mentions max_tokens=80K (unconfi",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-7-max-preview",
      "name": "Qwen3.7-Max-Preview",
      "provider": "Alibaba",
      "status": "preview",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 7.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-05-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "qwen3.7-max",
      "api_string": "qwen3.7-max-preview",
      "notes": "Preview snapshot, alias = qwen3.7-max-2026-05-17. Text-only input, Thinking mode only (per official release notes 2026-05-25, International). Same International price as GA Max ($2.5/$7.5 per 1M). Superseded by the GA qwen3.7-max. Max output assumed same as GA (65536) - not separately confirmed.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-7-plus",
      "name": "Qwen3.7-Plus",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 1.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-06-01",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.7-plus",
      "notes": "Mid-tier vision-language Plus model, alias = qwen3.7-plus-2026-05-26. Official International TIERED pricing: tier1 0<token<=256K: $0.4 in / $1.6 out (same for Non-Thinking and Thinking output); tier2 256K<token<=1M: $1.2 in / $4.8 out. Multimodal (text/image/video input), full agent/coding/tool-use; supports context caching (discount). Context window 1,000,000 (tier extends to 1M). Max output 65536 inferred from Qwen3.6-Plus spec corroboration; not separately confirmed on official spec page (SPA-rendered). Released ~2026-06-01 (International) per official release notes. API-only/proprietary.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-6-flash",
      "name": "Qwen3.6-Flash",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.25,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-04-16",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.6-flash",
      "notes": "Most cost-effective current Flash model, alias = qwen3.6-flash-2026-04-16; native vision-language (multimodal text/image/video). Official International TIERED pricing: tier1 0<token<=256K: $0.25 in / $1.5 out; tier2 256K<token<=1M: $1 in / $4 out. Supports 50% batch-inference discount and context caching. Has an open-weight sibling listed: qwen3.6-35b-a3b (per official release notes, the qwen3.6-flash entry groups qwen3.6-flash / qwen3.6-flash-2026-04-16 / qwen3.6-35b-a3b). Context window 1M (tier to 1M). Max output not published on accessible official page (null). Released 2026-04-16 (Interna",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-max",
      "name": "Qwen3-Max",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": 32768,
      "price_input_per_mtok": 1.2,
      "price_output_per_mtok": 6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2025-06",
      "released": "2025-09-23",
      "deprecated_on": "2026-09-08",
      "retires_on": "2026-09-08",
      "replacement": "qwen3.7-max",
      "api_string": "qwen3-max",
      "notes": "Previous flagship Max, alias = qwen3-max-2026-01-23. Official International TIERED pricing: 0<token<=32K: $1.2 in / $6 out; 32K<token<=128K: $2.4 / $12; 128K<token<=256K: $3 / $15. Non-Thinking and Thinking modes. Context window 262,144 (256K) and max output 32,768 per OpenRouter (Alibaba spec table is SPA-rendered, not directly fetchable); knowledge cutoff Jun 2025 per OpenRouter. SCHEDULED DEPRECATION: the qwen3-max alias and qwen3-max-preview are listed for deprecation 2026-09-08 00:00:00 -> replacement qwen3.7-max; dated snapshots qwen3-max-2026-01-23 and qwen3-max-2025-09-23 are listed fo",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-max-preview",
      "name": "Qwen3-Max-Preview",
      "provider": "Alibaba",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": 32768,
      "price_input_per_mtok": 1.2,
      "price_output_per_mtok": 6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-05",
      "deprecated_on": "2026-09-08",
      "retires_on": "2026-09-08",
      "replacement": "qwen3.7-max",
      "api_string": "qwen3-max-preview",
      "notes": "Early preview of Qwen3-Max (trillion-parameter MoE). Listed in official deprecation table for retirement 2026-09-08 00:00:00 -> replacement qwen3.7-max. Same International tiered pricing as qwen3-max ($1.2/$6 at <=32K etc.). API-only/proprietary. Release date approximate (Sep 2025); context/output mirror qwen3-max (256K / 32K).",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-6-max-preview",
      "name": "Qwen3.6-Max-Preview",
      "provider": "Alibaba",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 1.3,
      "price_output_per_mtok": 7.8,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-09-08",
      "retires_on": "2026-09-08",
      "replacement": "qwen3.7-max",
      "api_string": "qwen3.6-max-preview",
      "notes": "Preview Max snapshot. Official International TIERED pricing: 0<token<=128K: $1.3 in / $7.8 out; 128K<token<=256K: $2 / $12. Listed in official deprecation table for retirement 2026-09-08 00:00:00 -> replacement qwen3.7-max. Context 262,144 / max output 65,536 per web corroboration (search result citing Qwen3.6-Max-Preview 262144 ctx, 65536 max output); not on accessible official spec page. API-only/proprietary.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen-max",
      "name": "Qwen-Max (Qwen2.5-Max)",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 32768,
      "max_output_tokens": 8192,
      "price_input_per_mtok": 1.6,
      "price_output_per_mtok": 6.4,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-01-27",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen-max",
      "notes": "Legacy stable Max alias = qwen-max-2025-01-25 (a.k.a. Qwen2.5-Max). Official International price: $1.6 in / $6.4 out per 1M, no tiered pricing, Non-Thinking mode only; supports 50% batch-inference discount. Context window ~32,768 (33K) and max output ~8,192 per third-party corroboration (CloudPrice/DataStudios); not on accessible official spec page. Still listed/purchasable on pricing page (Jun 22, 2026); not in current deprecation table. Released 2025-01-27 (International). API-only/proprietary.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-7-plus-2026-05-26",
      "name": "Qwen3.7-Plus (snapshot 2026-05-26)",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 1.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-06-01",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.7-plus-2026-05-26",
      "notes": "Dated snapshot currently pinned by the qwen3.7-plus alias. Same tiered International pricing as the alias: <=256K $0.4/$1.6, 256K-1M $1.2/$4.8. Included so trackers can pin a stable version string. Multimodal.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-5-plus",
      "name": "Qwen3.5-Plus",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 2.4,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-02-15",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.5-plus",
      "notes": "Prior-gen native vision-language Plus, alias = qwen3.5-plus-2026-02-15 (latest dated snapshot qwen3.5-plus-2026-04-20). Official International price: 0<token<=256K $0.4 in / $2.4 out per 1M (single tier listed). Still listed on pricing page Jun 22, 2026; not in deprecation table. Max output not on accessible official page (null). Released 2026-02-15 (International).",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-6-plus",
      "name": "Qwen3.6-Plus",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-04-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.6-plus",
      "notes": "Prior-gen Plus, alias = qwen3.6-plus-2026-04-02. Official International price: 0<token<=256K $0.5 in / $3 out per 1M (single tier listed). Context window 1,000,000 and max output 65,536 per web corroboration (multiple sources cite Qwen3.6-Plus 1M ctx / 65536 max). Still listed on pricing page Jun 22, 2026. Multimodal.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-5-flash",
      "name": "Qwen3.5-Flash",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-02-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.5-flash",
      "notes": "Prior-gen Flash, alias = qwen3.5-flash-2026-02-23. Official International price: 0<token<=1M $0.1 in / $0.4 out per 1M (single tier, context to 1M). Supports 50% batch-inference discount and context caching. Still listed on pricing page Jun 22, 2026. Max output not on accessible official page (null).",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen-plus",
      "name": "Qwen-Plus (Qwen3-series)",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 32768,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 1.2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen-plus",
      "notes": "Legacy stable Plus alias = qwen-plus-2025-12-01 (Qwen3 series). Official International TIERED pricing: 0<token<=256K $0.4 in; output $1.2 (Non-Thinking) / $4.0 (Thinking, chain-of-thought+answer) per 1M. Context length increased to 1,000,000 per official release note (qwen-plus-2025-07-28). Max output ~32K (typical for Qwen3 Plus; not on accessible official spec page). Older dated snapshots qwen-plus-2024-11-27/-11-25/-09-19/-08-06 were deprecated 2026-01-30 -> replacement qwen-plus-2025-12-01. Current alias still GA. Text-only.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen-flash",
      "name": "Qwen-Flash",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.05,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen-flash",
      "notes": "Legacy stable Flash alias = qwen-flash-2025-07-28; the recommended replacement for the discontinued Qwen-Turbo. Official International TIERED pricing: 0<token<=256K $0.05 in / $0.4 out; 256K<token<=1M $0.25 in / $2 out per 1M. Supports 50% batch-inference discount and context caching. Context window 1M. Max output not on accessible official page (null). Still GA on pricing page Jun 22, 2026. Text-only.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen-turbo",
      "name": "Qwen-Turbo",
      "provider": "Alibaba",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.05,
      "price_output_per_mtok": 0.2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-04-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "qwen-flash",
      "api_string": "qwen-turbo",
      "notes": "DEPRECATED (no end-of-service date published, still callable). Official pricing page (Jun 22, 2026) states verbatim: 'Qwen-Turbo will no longer be updated. We recommend switching to Qwen-Flash.' Alias = qwen-turbo-2025-04-28. Official International price: $0.05 in; output $0.2 (Non-Thinking) / $0.5 (Thinking) per 1M. Older snapshot qwen-turbo-2024-09-19 listed in deprecation table -> replacement qwen-flash-2025-07-28. Marked status=deprecated because it is no longer updated; deprecated_on/retires_on left null because no formal retirement date is published for the qwen-turbo alias. Text-only.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-5-omni-plus",
      "name": "Qwen3.5-Omni-Plus",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "audio",
        "video"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.5-omni-plus",
      "notes": "Current Omni (any-to-any) model in the lineup, listed on the official Models overview (Jun 22, 2026) for image/video understanding, speech-to-speech, and omni use cases; a realtime variant qwen3.5-omni-plus-realtime also exists. Pricing not captured here (omni/audio pricing is on separate per-modality tables of the pricing page billed per token AND per second/character for audio; not a simple per-1M-token text rate). Context/output/release not on the accessible (server-rendered) pages -> null. Multimodal (text/image/audio/video in, text+audio out).",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/models",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-rerank",
      "name": "Qwen3-Rerank",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3-rerank",
      "notes": "Current reranking model; listed as the replacement for the deprecated gte-rerank (gte-rerank deprecation 2026-05-30 per official deprecation table). Rerank models are not priced per input/output token in the standard text table; pricing null here. Included because it is the current GA rerank model in the lineup.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "qwen3-235b-a22b",
      "name": "Qwen3-235B-A22B (open weights)",
      "provider": "Alibaba",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-04-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "qwen3.7-plus",
      "api_string": "qwen3-235b-a22b",
      "notes": "Representative OPEN-WEIGHT flagship from the Qwen3 family (released Apr 2025, Apache-2.0 license, weights on Hugging Face/ModelScope). On Model Studio its HOSTED endpoint is in the deprecation table (Qwen3 open source edition group; deprecation date July 8, 2026 shared with that table) -> replacement qwen3.7-plus; the open weights themselves remain downloadable regardless of the hosted-API deprecation. Included to flag open-weight status; many sibling open-weight variants exist (qwen3-8b/14b/32b, qwen3-30b-a3b, qwen3-235b-a22b-instruct-2507/-thinking-2507, qwen3-next-80b-a3b, qwen3-vl-* , qwen",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "perplexity-sonar",
      "name": "Sonar",
      "provider": "Perplexity",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 1,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-01",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "sonar",
      "notes": "Lightweight, cost-effective real-time web search model with grounding/citations. 128K context. Token pricing $1/$1 per 1M in/out. Additional per-request search fee NOT included in token price: $5/$8/$12 per 1,000 requests for low/medium/high search context size (pricing page). Supports image uploads (added Apr 2025) and file attachments PDF/DOC/DOCX/TXT/RTF (added Sep 2025). Max output tokens not published by Perplexity. Knowledge cutoff not published (search-grounded model with live web access). Pricing page fetched 2026-06-28.",
      "source_url": "https://docs.perplexity.ai/docs/sonar/models/sonar",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "perplexity-sonar-pro",
      "name": "Sonar Pro",
      "provider": "Perplexity",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": null,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-01",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "sonar-pro",
      "notes": "Advanced search model for complex queries and follow-ups; ~2x more search results than Sonar; non-reasoning. 200K context. Token pricing $3/$15 per 1M in/out. Additional per-request search fee NOT in token price: $6/$10/$14 per 1,000 requests for low/medium/high search context size (pricing page). Supports image input and the Dec 2025 media classifier (auto image/video selection). Max output tokens not published. Knowledge cutoff not published (search-grounded). Pricing page fetched 2026-06-28.",
      "source_url": "https://docs.perplexity.ai/docs/sonar/models/sonar-pro",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "perplexity-sonar-reasoning-pro",
      "name": "Sonar Reasoning Pro",
      "provider": "Perplexity",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "sonar-reasoning-pro",
      "notes": "Precise reasoning model with Chain-of-Thought; emits a <think> reasoning section before the answer. 128K context. Token pricing $2/$8 per 1M in/out. Per-request search fee NOT in token price: $6/$10/$14 per 1,000 requests for low/medium/high search context (pricing page). Successor to the now-removed 'sonar-reasoning'. Image input with structured outputs not supported in thinking models. Max output tokens and knowledge cutoff not published. Exact release date not published in docs. Pricing page fetched 2026-06-28.",
      "source_url": "https://docs.perplexity.ai/docs/sonar/models/sonar-reasoning-pro",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "perplexity-sonar-deep-research",
      "name": "Sonar Deep Research",
      "provider": "Perplexity",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "sonar-deep-research",
      "notes": "Expert-level research model that runs exhaustive multi-source searches and generates comprehensive reports. 128K context. Token pricing: input $2/M, output $8/M, citation tokens $2/M, reasoning tokens $3/M, plus search queries $5 per 1,000 (pricing page). Featured May 2025 with reasoning-effort parameter and async API. Max output tokens and knowledge cutoff not published. Pricing page fetched 2026-06-28.",
      "source_url": "https://docs.perplexity.ai/docs/sonar/models/sonar-deep-research",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "perplexity-sonar-reasoning",
      "name": "Sonar Reasoning",
      "provider": "Perplexity",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-01",
      "deprecated_on": "2025-12-15",
      "retires_on": "2025-12-15",
      "replacement": "sonar-reasoning-pro",
      "api_string": "sonar-reasoning",
      "notes": "DEPRECATED and removed from the API on 2025-12-15 per the official changelog; calls should migrate to sonar-reasoning-pro (enhanced multi-step reasoning with web search). The current models page (docs.perplexity.ai/getting-started/models) no longer lists it. Historical context window 128K. Historical token pricing was approx $1 input / $5 output per 1M while active, but this is NOT confirmed on a current official page (pricing page no longer lists it), so left null rather than guessed. Confirmed via changelog and the sonar-reasoning-pro model card (which names it as the deprecated predecessor)",
      "source_url": "https://docs.perplexity.ai/changelog/changelog",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-k2-7-code",
      "name": "Kimi K2.7 Code",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.95,
      "price_output_per_mtok": 4,
      "price_cached_input_per_mtok": 0.19,
      "knowledge_cutoff": null,
      "released": "2026-06-12",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "kimi-k2.7-code",
      "notes": "Open-weight 1T-parameter MoE (~32B active), coding-focused. Cache-miss input $0.95/M, cache-hit input $0.19/M, output $4.00/M (official pricing page, fetched 2026-06-28). Listed as a current model on platform.kimi.ai/docs/models. Release date 2026-06-12 from multiple secondary sources (MarkTechPost, digitalapplied, noqta) - the official pricing/model page does not publish release date or knowledge cutoff. Max output tokens not published. Modality text/code only (not multimodal).",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-k27-code.md",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-k2-7-code-highspeed",
      "name": "Kimi K2.7 Code HighSpeed",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.9,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": 0.38,
      "knowledge_cutoff": null,
      "released": "2026-06-12",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "kimi-k2.7-code-highspeed",
      "notes": "High-throughput variant of K2.7 Code: ~180 tokens/s (up to ~260 tokens/s short context). Cache-miss input $1.90/M, cache-hit input $0.38/M, output $8.00/M - i.e. 2x the standard K2.7 Code price for higher speed (official pricing page, fetched 2026-06-28). Max output tokens and knowledge cutoff not published officially.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-k27-code.md",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-k2-6",
      "name": "Kimi K2.6",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.95,
      "price_output_per_mtok": 4,
      "price_cached_input_per_mtok": 0.16,
      "knowledge_cutoff": null,
      "released": "2026-04-20",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "kimi-k2.6",
      "notes": "Moonshot's flagship/most intelligent model. Native multimodal (text, image, video input), supports thinking and non-thinking modes. Cache-miss input $0.95/M, cache-hit input $0.16/M, output $4.00/M (official pricing page, fetched 2026-06-28). Designated replacement for all deprecated kimi-k2 series and retired kimi-latest/kimi-thinking-preview models. Release date 2026-04-20 from secondary sources (Yicai Global, kimi-k2.org, miraflow) - official page does not list release date or knowledge cutoff. Max output tokens not published.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-k26.md",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-k2-5",
      "name": "Kimi K2.5",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.6,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": 0.1,
      "knowledge_cutoff": null,
      "released": "2026-01-27",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "kimi-k2.5",
      "notes": "Native multimodal model (text, image, video input), 1T-param MoE (~32B active). Cache-miss input $0.60/M, cache-hit input $0.10/M, output $3.00/M (official pricing page, fetched 2026-06-28). Still listed as a current/available model on platform.kimi.ai/docs/models (not deprecated). Release date 2026-01-27 from multiple secondary sources (Baidu Baike, ComfyUI Wiki, AWS Bedrock card, kimi.com). Official page does not list knowledge cutoff or max output tokens.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-k25.md",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "moonshot-v1-8k",
      "name": "Moonshot v1 8K",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-8k",
      "notes": "Legacy text model, still listed as current. Input $0.20/M, output $2.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified for the v1 family on the official page. Release date and knowledge cutoff not published.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "moonshot-v1-32k",
      "name": "Moonshot v1 32K",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 32768,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-32k",
      "notes": "Legacy text model. Input $1.00/M, output $3.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified. Release date and knowledge cutoff not published.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "moonshot-v1-128k",
      "name": "Moonshot v1 128K",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-128k",
      "notes": "Legacy text model. Input $2.00/M, output $5.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified. Release date and knowledge cutoff not published.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "moonshot-v1-8k-vision-preview",
      "name": "Moonshot v1 8K Vision (Preview)",
      "provider": "Moonshot",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-8k-vision-preview",
      "notes": "Vision-capable preview variant (text + image). Same price as moonshot-v1-8k: input $0.20/M, output $2.00/M (official pricing page, fetched 2026-06-28). 'preview' in the API id. Cache-hit pricing not specified.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "moonshot-v1-32k-vision-preview",
      "name": "Moonshot v1 32K Vision (Preview)",
      "provider": "Moonshot",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 32768,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-32k-vision-preview",
      "notes": "Vision-capable preview variant (text + image). Same price as moonshot-v1-32k: input $1.00/M, output $3.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "moonshot-v1-128k-vision-preview",
      "name": "Moonshot v1 128K Vision (Preview)",
      "provider": "Moonshot",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-128k-vision-preview",
      "notes": "Vision-capable preview variant (text + image). Same price as moonshot-v1-128k: input $2.00/M, output $5.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-k2-0905-preview",
      "name": "Kimi K2 (0905 Preview)",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-0905-preview",
      "notes": "Deprecated/officially discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Pricing no longer published on current official pages. Historical legacy K2 list price was reported around $0.60/M input, $2.50/M output by secondary trackers, but not confirmable on current official pages, so prices left null. Context window 256K (262144) per K2 family docs.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-k2-0711-preview",
      "name": "Kimi K2 (0711 Preview)",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-11",
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-0711-preview",
      "notes": "The original Kimi K2 (0711) release, deprecated/discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Release implied by '0711' (2025-07-11) and corroborated by Moonshot's original K2 launch. Pricing no longer on official pages so left null. Original K2 context window was 128K (131072) per Moonshot's original K2 docs/GitHub.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-k2-turbo-preview",
      "name": "Kimi K2 Turbo (Preview)",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-turbo-preview",
      "notes": "High-speed turbo variant of K2, deprecated/discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Pricing removed from official pages, left null. Context 256K per K2 family docs (unverified for this exact variant on current pages).",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-k2-thinking",
      "name": "Kimi K2 Thinking",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-thinking",
      "notes": "Reasoning/thinking variant of K2, deprecated/discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Pricing removed from official pages, left null.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-k2-thinking-turbo",
      "name": "Kimi K2 Thinking Turbo",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-thinking-turbo",
      "notes": "High-speed thinking variant of K2, deprecated/discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Pricing removed from official pages, left null.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-latest",
      "name": "Kimi Latest",
      "provider": "Moonshot",
      "status": "retired",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-01-28",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-latest",
      "notes": "Retired 2026-01-28 per official model list; replacement now kimi-k2.6. This was the rolling 'latest' alias (auto-updated to the newest Kimi chat model, historically vision-capable). Pricing not on current official pages, left null. Context window historically 128K (131072) - not re-verifiable on current pages.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": null,
      "embedding_dimensions": null
    },
    {
      "id": "kimi-thinking-preview",
      "name": "Kimi Thinking (Preview)",
      "provider": "Moonshot",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2025-11-11",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-thinking-preview",
      "notes": "Earliest reasoning preview model, retired 2025-11-11 per official model list; replacement now kimi-k2.6. Pricing not on current official pages, left null. Context window historically 128K - not re-verifiable on current pages.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": false,
      "embedding_dimensions": null
    }
  ]
}