{
  "version": 2,
  "updated": "2026-08-19",
  "count": 258,
  "_meta": {
    "source": "AI Model Watch",
    "url": "https://aimodelwatch.dev",
    "api_docs": "https://aimodelwatch.dev/api",
    "openapi": "https://aimodelwatch.dev/api/openapi.json",
    "feed_updated": "2026-08-19",
    "license": "MIT",
    "citation": "Data by AI Model Watch — https://aimodelwatch.dev",
    "providers": [
      "Alibaba",
      "Amazon",
      "Anthropic",
      "Cohere",
      "DeepSeek",
      "Google",
      "Meta",
      "Mistral",
      "Moonshot",
      "OpenAI",
      "Perplexity",
      "xAI"
    ],
    "methodology": "Compiled from official provider documentation. Each model carries its own source_url.",
    "contact": "https://aimodelwatch.dev/about"
  },
  "docs": "https://aimodelwatch.dev/api",
  "source": "Compiled from official provider documentation. Each model carries its own source_url.",
  "license": "MIT — attribution appreciated: AI Model Watch (https://aimodelwatch.dev)",
  "models": [
    {
      "id": "claude-fable-5",
      "name": "Claude Fable 5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 10,
      "price_output_per_mtok": 50,
      "price_cached_input_per_mtok": 1,
      "knowledge_cutoff": null,
      "released": "2026-06-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-fable-5",
      "notes": "Anthropic's most capable widely released model. GA on Claude API, Claude Platform on AWS, Amazon Bedrock, Google Cloud, and Microsoft Foundry beginning June 9, 2026. Adaptive thinking always on; no extended thinking. 5m cache write $12.50/MTok, 1h cache write $20/MTok (cached_input field here = Cache Hits & Refreshes $1/MTok). Batch: $5 in / $25 out per MTok. 1M context at standard pricing. Uses the Opus 4.7-generation tokenizer (~30-35% more tokens for the same text vs pre-4.7 models). Dateless model ID is a pinned snapshot. Knowledge cutoff not published on overview page (null).",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-mythos-5",
      "name": "Claude Mythos 5",
      "provider": "Anthropic",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 10,
      "price_output_per_mtok": 50,
      "price_cached_input_per_mtok": 1,
      "knowledge_cutoff": null,
      "released": "2026-06-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-mythos-5",
      "notes": "Limited availability (not GA) via Project Glasswing to approved customers, beginning June 9, 2026. Successor to Claude Mythos Preview. Adaptive thinking always on; no extended thinking. Pricing identical to Fable 5: $10 in / $50 out; 5m cache $12.50, 1h cache $20, cache hits $1/MTok; Batch $5/$25. 1M context at standard pricing. Status set to 'preview' to reflect limited/invite-only availability. Knowledge cutoff not published (null).",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-mythos-preview",
      "name": "Claude Mythos Preview",
      "provider": "Anthropic",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-06-30",
      "replacement": "claude-mythos-5",
      "api_string": "claude-mythos-preview",
      "notes": "Invitation-only research preview model for defensive cybersecurity workflows under Project Glasswing (no self-serve sign-up). Will be RETIRED on June 30, 2026; recommended migration to Claude Mythos 5 (claude-mythos-5). Standalone per-model pricing not published on the pricing page (Long context note groups it with the 1M-context models at standard pricing, but no $ figures listed) -> prices null. Included in 1M-token context group. Max output / knowledge cutoff not published (null). As of 2026-06-28 the doc lists it as still functional pending the June 30, 2026 retirement; marked 'deprecated'",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 25,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2026-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-opus-4-8",
      "notes": "Most capable Opus-tier model (complex reasoning, long-horizon agentic coding). API ID == alias 'claude-opus-4-8' (dateless pinned snapshot). Adaptive thinking; no extended thinking. effort defaults to 'high'. 5m cache $6.25/MTok, 1h cache $10/MTok, cache hits $0.50/MTok. Batch: $2.50 in / $12.50 out. 1M context at standard pricing; 200k on Microsoft Foundry. Supports up to 300k output via Batch API beta header. Fast mode (preview): $10 in / $50 out per MTok. Reliable knowledge cutoff Jan 2026; training data cutoff Jan 2026. Tentative retirement not sooner than May 28, 2027. Explicit release da",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-opus-4-7",
      "name": "Claude Opus 4.7",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 25,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2026-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-opus-4-7",
      "notes": "Listed under 'Legacy models' on overview but Active in deprecations table. Dateless pinned snapshot ID == alias. Adaptive thinking; no extended thinking. Introduced the new tokenizer (~30-35% more tokens for same text). temperature/top_p/top_k deprecated (400 error on non-default values) on Opus 4.7 and later. 5m cache $6.25, 1h cache $10, cache hits $0.50/MTok. Batch $2.50/$12.50. 1M context standard pricing. Fast mode (preview): $30 in / $150 out per MTok. Reliable knowledge cutoff Jan 2026; training cutoff Jan 2026. Tentative retirement not sooner than April 16, 2027. Status 'ga' per Active",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-opus-4-6",
      "name": "Claude Opus 4.6",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 25,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2025-05",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-opus-4-6",
      "notes": "Listed under 'Legacy models' on overview but Active in deprecations table. Dateless pinned snapshot ID == alias. Extended thinking: Yes; adaptive thinking: Yes. 5m cache $6.25, 1h cache $10, cache hits $0.50/MTok. Batch $2.50/$12.50. 1M context standard pricing. Fast mode (preview): $30 in / $150 out per MTok. Reliable knowledge cutoff May 2025; training data cutoff Aug 2025. Tentative retirement not sooner than February 5, 2027. Status 'ga' per Active state. Release date not explicitly published (null).",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-opus-4-5",
      "name": "Claude Opus 4.5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 25,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2025-05",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-opus-4-5-20251101",
      "notes": "API ID claude-opus-4-5-20251101 (alias claude-opus-4-5). Listed under 'Legacy models' on overview but Active in deprecations table. Extended thinking: Yes; adaptive thinking: No. 200k context (not 1M). Max output 64k. 5m cache $6.25, 1h cache $10, cache hits $0.50/MTok. Batch $2.50/$12.50. Reliable knowledge cutoff May 2025; training data cutoff Aug 2025. Snapshot date 20251101 implies a Nov 2025 release. Tentative retirement not sooner than November 24, 2026.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-opus-4-1",
      "name": "Claude Opus 4.1",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": 32000,
      "price_input_per_mtok": 15,
      "price_output_per_mtok": 75,
      "price_cached_input_per_mtok": 1.5,
      "knowledge_cutoff": "2025-01",
      "released": null,
      "deprecated_on": "2026-06-05",
      "retires_on": "2026-08-05",
      "replacement": "claude-opus-4-8",
      "api_string": "claude-opus-4-1-20250805",
      "notes": "RETIRED August 5, 2026 — Anthropic's deprecations table states Current state = Retired (read 2026-08-06); the endpoint no longer serves requests. Deprecated June 5, 2026. Recommended replacement claude-opus-4-8. API ID claude-opus-4-1-20250805 (alias claude-opus-4-1). Extended thinking: Yes; adaptive thinking: No. 200k context, max output 32k. Pricing $15 in / $75 out; 5m cache $18.75, 1h cache $30, cache hits $1.50/MTok. Batch $7.50/$37.50. Reliable knowledge cutoff Jan 2025; training data cutoff Mar 2025. Snapshot date 20250805 implies an Aug 5, 2025 release.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-opus-4",
      "name": "Claude Opus 4",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": null,
      "price_input_per_mtok": 15,
      "price_output_per_mtok": 75,
      "price_cached_input_per_mtok": 1.5,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-14",
      "retires_on": "2026-06-15",
      "replacement": "claude-opus-4-8",
      "api_string": "claude-opus-4-20250514",
      "notes": "RETIRED June 15, 2026 on Anthropic-operated platforms (deprecated April 14, 2026); the pricing page states it is still served on Google Cloud. Recommended replacement claude-opus-4-8. API ID claude-opus-4-20250514. PRICE CONFIRMED FIRST-PARTY 2026-07-29 against platform.claude.com/docs/en/about-claude/pricing.md, whose model-pricing table retains retired models with their rates and labels this row '(retired, except on Google Cloud)': $15 in / $75 out, 5m cache write $18.75, 1h cache write $30, cache hits & refreshes $1.50/MTok — unchanged to the cent. Batch $7.50/$37.50. Snapshot date 20250514 implies a May 14, 2025 release. Separate open item, NOT about price: context_window 200000 reflects the Opus 4-class family norm and is not stated on any current official page; max_output_tokens left null.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": "2026-01",
      "released": "2026-06-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-sonnet-5",
      "notes": "Best combination of speed and intelligence; Anthropic's most agentic Sonnet, performance close to Opus 4.8 at lower cost. Dateless pinned snapshot ID == alias claude-sonnet-5. Extended thinking: No; adaptive thinking: Yes (effort defaults to 'high' on the Claude API and Claude Code). 1M context at standard pricing; max output 128k (300k via Batch API beta header output-300k-2026-03-24). Prices shown are STANDARD ($3 in / $15 out, 5m cache $3.75, 1h cache $6, cache hit $0.30, Batch $1.50/$7.50). INTRODUCTORY pricing of $2 in / $10 out per MTok is in effect through August 31, 2026 (5m cache $2.50, 1h cache $4, cache hit $0.20, Batch $1/$5), reverting to standard on September 1, 2026 — WATCH: if standard is not yet in effect, current live cost is the intro rate. Uses the newer tokenizer (~30% more tokens for the same text). Reliable knowledge cutoff Jan 2026; training data cutoff Jan 2026. Supersedes Claude Sonnet 4.6 (now a legacy model). Released 2026-06-30 (system card date).",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-sonnet-4-6",
      "name": "Claude Sonnet 4.6",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 1000000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": "2025-08",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-sonnet-4-6",
      "notes": "Best combination of speed and intelligence. Dateless pinned snapshot ID == alias claude-sonnet-4-6. Extended thinking: Yes; adaptive thinking: Yes. 1M context, max output 128k (300k via Batch API beta header). 5m cache $3.75, 1h cache $6, cache hits $0.30/MTok. Batch $1.50/$7.50. Reliable knowledge cutoff Aug 2025; training data cutoff Jan 2026. Tentative retirement not sooner than February 17, 2027. Release date not explicitly published (null). Now a LEGACY model (still generally available, not deprecated) — superseded by Claude Sonnet 5 as of 2026-06-30; consider migrating for improved performance.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-sonnet-4-5",
      "name": "Claude Sonnet 4.5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": "2025-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-sonnet-4-5-20250929",
      "notes": "API ID claude-sonnet-4-5-20250929 (alias claude-sonnet-4-5). Listed under 'Legacy models' on overview but Active in deprecations table. Extended thinking: Yes; adaptive thinking: No. 200k context, max output 64k. 5m cache $3.75, 1h cache $6, cache hits $0.30/MTok. Batch $1.50/$7.50. Reliable knowledge cutoff Jan 2025; training data cutoff Jul 2025. Snapshot 20250929 implies Sep 29, 2025 release. Tentative retirement not sooner than September 29, 2026.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-sonnet-4",
      "name": "Claude Sonnet 4",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-14",
      "retires_on": "2026-06-15",
      "replacement": "claude-sonnet-4-6",
      "api_string": "claude-sonnet-4-20250514",
      "notes": "RETIRED June 15, 2026 on Anthropic-operated platforms (deprecated April 14, 2026); the pricing page states it is still served on Bedrock and Google Cloud. Recommended replacement claude-sonnet-4-6. API ID claude-sonnet-4-20250514. PRICE CONFIRMED FIRST-PARTY 2026-07-29 against platform.claude.com/docs/en/about-claude/pricing.md, whose model-pricing table retains retired models with their rates and labels this row '(retired, except on Bedrock and Google Cloud)': $3 in / $15 out, 5m cache write $3.75, 1h cache write $6, cache hits & refreshes $0.30/MTok — unchanged to the cent. Batch $1.50/$7.50. Snapshot 20250514 implies May 14, 2025 release. Separate open item, NOT about price: context_window and max_output_tokens are absent from current official tables and are therefore left null.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "provider": "Anthropic",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": 200000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": 0.1,
      "knowledge_cutoff": "2025-02",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "claude-haiku-4-5-20251001",
      "notes": "Fastest model with near-frontier intelligence. API ID claude-haiku-4-5-20251001 (alias claude-haiku-4-5). Extended thinking: Yes; adaptive thinking: No. 200k context, max output 64k. 5m cache $1.25, 1h cache $2, cache hits $0.10/MTok. Batch $0.50/$2.50. Reliable knowledge cutoff Feb 2025; training data cutoff Jul 2025. Snapshot 20251001 implies Oct 1, 2025 release. Tentative retirement not sooner than October 15, 2026.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-haiku-3-5",
      "name": "Claude Haiku 3.5",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.8,
      "price_output_per_mtok": 4,
      "price_cached_input_per_mtok": 0.08,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-12-19",
      "retires_on": "2026-02-19",
      "replacement": "claude-haiku-4-5-20251001",
      "api_string": "claude-3-5-haiku-20241022",
      "notes": "RETIRED February 19, 2026 on Anthropic-operated platforms (deprecated December 19, 2025); pricing page notes still available on Bedrock and Google Cloud. Recommended replacement claude-haiku-4-5-20251001. API ID claude-3-5-haiku-20241022. Pricing $0.80 in / $4 out; 5m cache $1, 1h cache $1.60, cache hits $0.08/MTok; Batch $0.40/$2. Snapshot 20241022 implies Oct 22, 2024 release. Context/max-output not in current overview tables -> null.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-3-7-sonnet",
      "name": "Claude Sonnet 3.7",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-10-28",
      "retires_on": "2026-02-19",
      "replacement": "claude-sonnet-4-6",
      "api_string": "claude-3-7-sonnet-20250219",
      "notes": "RETIRED February 19, 2026 (deprecated October 28, 2025). Recommended replacement claude-sonnet-4-6. API ID claude-3-7-sonnet-20250219. No longer listed in the current pricing table -> per-model prices null (not corroborated on current pricing page). Snapshot 20250219 implies Feb 19, 2025 release.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "claude-3-haiku",
      "name": "Claude Haiku 3",
      "provider": "Anthropic",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-02-19",
      "retires_on": "2026-04-20",
      "replacement": "claude-haiku-4-5-20251001",
      "api_string": "claude-3-haiku-20240307",
      "notes": "RETIRED April 20, 2026 (deprecated February 19, 2026). Recommended replacement claude-haiku-4-5-20251001. API ID claude-3-haiku-20240307. Not listed in the current pricing table -> per-model prices null. Snapshot 20240307 implies Mar 7, 2024 release.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/model-deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-6-sol",
      "name": "GPT-5.6 Sol",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 30,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2026-02-16",
      "released": "2026-07-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.6-sol",
      "notes": "Current OpenAI flagship — the top 'Sol' tier of the GPT-5.6 family (Sol/Terra/Luna), GA on the API + Codex 2026-07-09 (preview from 2026-06-26). Text+image input, text output, configurable reasoning effort. Priced identically to GPT-5.5 ($5/$0.50/$30) but stronger on coding/knowledge-work/cyber/science per OpenAI. New caching model: explicit cache breakpoints, 30-min min cache life, cache writes billed 1.25x uncached input, reads keep the 90% discount. Standard tier ($5.00/$0.50/$30.00) re-confirmed against the pricing page and the model page 2026-08-08; cache write $6.25/M. Prompts >272K input tokens are billed at 2x input / 1.5x output for the whole request (stated on the model page). Batch ~50%.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-6-cyber",
      "name": "GPT-5.6 Cyber",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 12.5,
      "price_output_per_mtok": 75,
      "price_cached_input_per_mtok": 1.25,
      "knowledge_cutoff": "2026-02-16",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.6-cyber",
      "notes": "Purpose-trained cybersecurity model in the GPT-5.6 family, for approved defenders doing authorized vulnerability research, exploit validation and security testing. ACCESS IS GATED: OpenAI states it 'requires separate approval and provisioning' via the Daybreak program, so the published rate is not self-serve. Priced in its own 'Cyber models' table on the pricing page (12.50 / 1.25 cached / 75.00 per 1M, cache writes 15.625) - a table check-price-drift.mjs does not read, because that script anchors on the Standard-tier tables. 400K context, 128K max output, Feb 16 2026 cutoff, text+image in / text out, reasoning tokens supported. The pricing page's long-context columns are '-' for this model, i.e. no long-context tier is published, unlike Sol. 'daybreak-red-latest' is an alias currently pointing here (and 'daybreak-blue-latest' to gpt-5.6-sol); the aliases are stated to re-point as new Daybreak models ship, with pricing following the underlying model. No release date published (null).",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.6-cyber",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-6-terra",
      "name": "GPT-5.6 Terra",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 12,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": "2026-02-16",
      "released": "2026-07-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.6-terra",
      "notes": "Balanced 'Terra' tier of the GPT-5.6 family — GPT-5.5-class quality at lower cost, positioned for everyday work. GA on the API + Codex 2026-07-09. Text+image input, text output. PRICE CORRECTED 2026-08-08: we published $2.50/$0.25/$15.00, which appears in no cell of OpenAI's pricing page; both first-party surfaces state $2.00/$0.20/$12.00 — developers.openai.com/api/docs/pricing (Standard tier) and the model page. Cache write $2.50/M. Prompts >272K input tokens are billed at 2x input / 1.5x output for the whole request (stated on the model page). Batch ~50% ($1.00/$0.10/$6.00).",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-6-luna",
      "name": "GPT-5.6 Luna",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": 1.2,
      "price_cached_input_per_mtok": 0.02,
      "knowledge_cutoff": "2026-02-16",
      "released": "2026-07-09",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.6-luna",
      "notes": "Fast, cost-efficient 'Luna' tier of the GPT-5.6 family — built for high-volume tasks. GA on the API + Codex 2026-07-09. Text+image input, text output, full 1.05M context. PRICE CORRECTED 2026-08-08: we published $1.00/$0.10/$6.00 — 5x high on all three — against $0.20/$0.02/$1.20 stated on developers.openai.com/api/docs/pricing (Standard tier) and on the model page. Cache write $0.25/M. Prompts >272K input tokens are billed at 2x input / 1.5x output for the whole request (stated on the model page). Batch ~50% ($0.10/$0.01/$0.60).",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-5",
      "name": "GPT-5.5",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 30,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2025-12-01",
      "released": "2026-04-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.5",
      "notes": "Prior flagship, still GA — superseded as OpenAI's top tier by GPT-5.6 Sol (2026-07-09) at the same $5/$0.50/$30 price. Snapshot alias gpt-5.5-2026-04-23 (snapshot date taken as release date). Text input / text output, configurable reasoning effort. Pricing from developers.openai.com/api/docs/pricing (standard tier); batch tier is ~50% ($2.50/$0.25/$15.00). OpenAI's pricing page qualifies this rate as \"<272K context length\" (verified 2026-08-08); it publishes no rate above that band on this surface.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.5",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-5-pro",
      "name": "GPT-5.5 Pro",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 30,
      "price_output_per_mtok": 180,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2025-12-01",
      "released": "2026-04-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.5-pro",
      "notes": "Snapshot alias gpt-5.5-pro-2026-04-23. Responses API only (incl. Batch); long-running, background mode recommended. Reasoning effort medium/high/xhigh. No cached-input price published. Regional data-residency endpoints add ~10% surcharge. OpenAI's pricing page qualifies this rate as \"<272K context length\" (verified 2026-08-08); it publishes no rate above that band on this surface.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.5-pro",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-4",
      "name": "GPT-5.4",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": 0.25,
      "knowledge_cutoff": "2025-08-31",
      "released": "2026-03-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.4",
      "notes": "Snapshot alias gpt-5.4-2026-03-05 (snapshot date taken as release date; not explicitly stated on doc page). Reasoning model with configurable effort. OpenAI's pricing page qualifies this rate as \"<272K context length\" (verified 2026-08-08); it publishes no rate above that band on this surface.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.4",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-4-mini",
      "name": "GPT-5.4 mini",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 0.75,
      "price_output_per_mtok": 4.5,
      "price_cached_input_per_mtok": 0.075,
      "knowledge_cutoff": "2025-08-31",
      "released": "2026-03-17",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.4-mini",
      "notes": "Snapshot alias gpt-5.4-mini-2026-03-17 (snapshot date taken as release date; not explicitly stated). Positioned for coding, computer use, subagents. Note: context window 400K (smaller than full gpt-5.4's 1.05M).",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.4-mini",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-4-nano",
      "name": "GPT-5.4 nano",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": 1.25,
      "price_cached_input_per_mtok": 0.02,
      "knowledge_cutoff": "2025-08-31",
      "released": "2026-03-17",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.4-nano",
      "notes": "Snapshot alias gpt-5.4-nano-2026-03-17. Cheapest GPT-5.4-class model for high-volume classification/extraction/ranking/subagents. Reasoning effort none(default)/low/medium/high/xhigh.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.4-nano",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-4-pro",
      "name": "GPT-5.4 Pro",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1050000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 30,
      "price_output_per_mtok": 180,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2025-08-31",
      "released": "2026-03-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.4-pro",
      "notes": "Snapshot alias gpt-5.4-pro-2026-03-05. Responses API only. Reasoning effort medium/high/xhigh. Doc notes potential 2x/1.5x multipliers for sessions exceeding 272K input tokens. No cached-input price published.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.4-pro",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-3-codex",
      "name": "GPT-5.3-Codex",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 1.75,
      "price_output_per_mtok": 14,
      "price_cached_input_per_mtok": 0.175,
      "knowledge_cutoff": "2025-08-31",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-5.3-codex",
      "notes": "Agentic coding model optimized for Codex. Reasoning effort low/medium/high/xhigh. Release date not stated on doc page. Cached-input price ($0.175) from pricing page.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-5.3-codex",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-realtime-2",
      "name": "GPT-Realtime-2",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "audio",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": 32000,
      "price_input_per_mtok": 4,
      "price_output_per_mtok": 24,
      "price_cached_input_per_mtok": 0.4,
      "knowledge_cutoff": "2024-09-30",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-realtime-2",
      "notes": "Speech-to-speech realtime model. Input: text/audio/image; Output: text/audio. Prices listed are the TEXT-token rates ($4 in / $24 out / $0.40 cached). Audio tokens are billed separately at $32 input / $64 output per 1M; image input $5/1M. Max output 32K tokens.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-realtime-2",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-image-2",
      "name": "GPT Image 2",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 8,
      "price_output_per_mtok": 30,
      "price_cached_input_per_mtok": 2,
      "knowledge_cutoff": null,
      "released": "2026-04-21",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-image-2",
      "notes": "Current flagship image-gen model (v1/images/generations). Snapshot gpt-image-2-2026-04-21. Input text+image, output image. Prices are per-1M-token for image input ($8, cached $2) and image output ($30). Designated replacement for dall-e-2/dall-e-3 (retired May 12, 2026).",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-image-2",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-image-1-5",
      "name": "GPT Image 1.5",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 8,
      "price_output_per_mtok": 32,
      "price_cached_input_per_mtok": 2,
      "knowledge_cutoff": null,
      "released": "2025-12-16",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "gpt-image-2",
      "api_string": "gpt-image-1.5",
      "notes": "Previous-generation image model, still available. Snapshot gpt-image-1.5-2025-12-16 (the dated snapshot is marked Deprecated on the doc page, though the base alias remains listed). Image output token price $32/1M.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-image-1.5",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-image-1-mini",
      "name": "GPT Image 1 mini",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": 0.25,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-image-1-mini",
      "notes": "Low-cost image model. Pricing per 1M tokens: image input $2.50 (cached $0.25), output $8.00. Spec page not separately retrieved; data from pricing page.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-4o-transcribe",
      "name": "GPT-4o Transcribe",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "audio",
        "text"
      ],
      "context_window": 16000,
      "max_output_tokens": 2000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-06-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-4o-transcribe",
      "notes": "Speech-to-text. Input audio+text, output text. ~$0.006/min. Token prices shown are text-token equivalents; audio input billed differently. Still listed as current on models index.",
      "source_url": "https://developers.openai.com/api/docs/models/gpt-4o-transcribe",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-4o-mini-transcribe",
      "name": "GPT-4o mini Transcribe",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "audio",
        "text"
      ],
      "context_window": 16000,
      "max_output_tokens": 2000,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-06-01",
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gpt-4o-mini-transcribe",
      "notes": "Cheaper transcription variant (~$0.003/min). Context window/max output assumed same as gpt-4o-transcribe (16K/2K) but not separately confirmed on its own spec page; price from pricing page.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "text-embedding-3-large",
      "name": "text-embedding-3-large",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.13,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-01-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "text-embedding-3-large",
      "notes": "Most capable embedding model. Output up to 3072 dimensions (reducible via the dimensions param). Price is first-party: $0.13 per 1M input tokens, confirmed unchanged 2026-08-08 against developers.openai.com/api/docs/pricing (Standard tier, embeddings table). Embeddings carry no output-token billing. Max input tokens CORRECTED 8191 -> 8192 on 2026-08-18: the model spec page states no limit at all, and the live API reference (developers.openai.com/api/docs/api-reference/embeddings/create) states it as a constraint on the input parameter -- \"the max input tokens for the model (8192 tokens for all embedding models)\". 8191 was the figure OpenAI's 2024 new-embedding-models announcement carried; the reference carries the field as data and is current, so it wins on the precision ranking. Not an upstream change and therefore no changelog entry.",
      "source_url": "https://developers.openai.com/api/docs/models/text-embedding-3-large",
      "open_weight": false,
      "embedding_dimensions": 3072,
      "price_per_1k_searches": null
    },
    {
      "id": "text-embedding-3-small",
      "name": "text-embedding-3-small",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.02,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-01-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "text-embedding-3-small",
      "notes": "Cost-efficient embedding model, default 1536 dimensions. Price is first-party: $0.02 per 1M input tokens, confirmed unchanged 2026-08-08 against developers.openai.com/api/docs/pricing (Standard tier, embeddings table); no output-token billing. Max input tokens CORRECTED 8191 -> 8192 on 2026-08-18: the model spec page states no limit at all, and the live API reference (developers.openai.com/api/docs/api-reference/embeddings/create) states it as a constraint on the input parameter -- \"the max input tokens for the model (8192 tokens for all embedding models)\". 8191 was the figure OpenAI's 2024 new-embedding-models announcement carried; the reference carries the field as data and is current, so it wins on the precision ranking. Not an upstream change and therefore no changelog entry.",
      "source_url": "https://developers.openai.com/api/docs/models/text-embedding-3-small",
      "open_weight": false,
      "embedding_dimensions": 1536,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5",
      "name": "GPT-5",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": 0.125,
      "knowledge_cutoff": "2024-09-30",
      "released": "2025-08-07",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.5",
      "api_string": "gpt-5-2025-08-07",
      "notes": "Previous flagship. Deprecation announced 2026-06-11; snapshot gpt-5-2025-08-07 shuts down 2026-12-11, replaced by gpt-5.5. Release date = snapshot date. Price FILLED 2026-08-08 from developers.openai.com/api/docs/pricing (Standard tier), which still lists deprecated models: $1.25 in / $0.125 cached / $10 out. The prior note claimed the price was \"not on the current pricing summary\" — that was a claim about which page had been opened, not about the provider.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-mini",
      "name": "GPT-5 mini",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 0.25,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": 0.025,
      "knowledge_cutoff": null,
      "released": "2025-08-07",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.4-mini",
      "api_string": "gpt-5-mini-2025-08-07",
      "notes": "Snapshot gpt-5-mini-2025-08-07 shuts down 2026-12-11, replaced by gpt-5.4-mini. Context/max-output assumed same as gpt-5 family (400K/128K). Price FILLED 2026-08-08 from developers.openai.com/api/docs/pricing (Standard tier), which still lists deprecated models: $0.25 in / $0.025 cached / $2 out. The prior note claimed the price was \"not on the current pricing summary\" — that was a claim about which page had been opened, not about the provider.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-nano",
      "name": "GPT-5 nano",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 0.05,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": 0.005,
      "knowledge_cutoff": null,
      "released": "2025-08-07",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.4-nano",
      "api_string": "gpt-5-nano-2025-08-07",
      "notes": "Snapshot gpt-5-nano-2025-08-07 shuts down 2026-12-11, replaced by gpt-5.4-nano. Context/max-output assumed same as gpt-5 family. Price FILLED 2026-08-08 from developers.openai.com/api/docs/pricing (Standard tier), which still lists deprecated models: $0.05 in / $0.005 cached / $0.4 out. The prior note claimed the price was \"not on the current pricing summary\" — that was a claim about which page had been opened, not about the provider.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-pro",
      "name": "GPT-5 Pro",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 15,
      "price_output_per_mtok": 120,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-10-06",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.5-pro",
      "api_string": "gpt-5-pro-2025-10-06",
      "notes": "Snapshot gpt-5-pro-2025-10-06 shuts down 2026-12-11, replaced by gpt-5.5-pro. Price FILLED 2026-08-08 from developers.openai.com/api/docs/pricing (Standard tier), which still lists deprecated models: $15 in / no cached-input rate published / $120 out. The prior note claimed the price was \"not on the current pricing summary\" — that was a claim about which page had been opened, not about the provider.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "o3",
      "name": "o3",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": 100000,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": 0.5,
      "knowledge_cutoff": "2024-06-01",
      "released": "2025-04-16",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.5",
      "api_string": "o3-2025-04-16",
      "notes": "o-series reasoning model. Snapshot o3-2025-04-16 shuts down 2026-12-11, replaced by gpt-5.5. Pricing still on spec page: $2 in / $0.50 cached / $8 out per 1M.",
      "source_url": "https://developers.openai.com/api/docs/models/o3",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "o3-pro",
      "name": "o3-pro",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": 100000,
      "price_input_per_mtok": 20,
      "price_output_per_mtok": 80,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-06-01",
      "released": "2025-06-10",
      "deprecated_on": "2026-06-11",
      "retires_on": "2026-12-11",
      "replacement": "gpt-5.5-pro",
      "api_string": "o3-pro-2025-06-10",
      "notes": "Higher-compute o3 variant. Snapshot o3-pro-2025-06-10 shuts down 2026-12-11, replaced by gpt-5.5-pro. Context/max-output assumed same as o3 (200K/100K). Price FILLED 2026-08-08 from developers.openai.com/api/docs/pricing (Standard tier), which still lists deprecated models: $20 in / no cached-input rate published / $80 out. The prior note claimed the price was \"not on the current pricing summary\" — that was a claim about which page had been opened, not about the provider.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "o3-deep-research",
      "name": "o3-deep-research",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": 100000,
      "price_input_per_mtok": 5,
      "price_output_per_mtok": 20,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-22",
      "retires_on": "2026-07-23",
      "replacement": "gpt-5.5-pro",
      "api_string": "o3-deep-research",
      "notes": "Deep-research model. Deprecation announced 2026-04-22; shuts down 2026-07-23, replaced by gpt-5.5-pro. Pricing shown is BATCH tier ($5 in / $20 out per 1M). Context/max-output assumed o3-class (200K/100K).",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "o4-mini-deep-research",
      "name": "o4-mini-deep-research",
      "provider": "OpenAI",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": 100000,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "o4-mini-deep-research",
      "notes": "Still listed on pricing page. Prices shown are BATCH tier ($1 in / $4 out per 1M). Context/max-output assumed o4-mini-class (200K/100K), not separately confirmed. Status GA but uncertain; not in current deprecation list.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "computer-use-preview",
      "name": "computer-use-preview",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.5,
      "price_output_per_mtok": 6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-22",
      "retires_on": "2026-07-23",
      "replacement": "gpt-5.4-mini",
      "api_string": "computer-use-preview",
      "notes": "Computer-use agent model. Deprecation announced 2026-04-22; shuts down 2026-07-23, replaced by gpt-5.4-mini. Pricing shown is BATCH tier ($1.50 in / $6 out per 1M). Context/max-output not published.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gpt-5-codex",
      "name": "GPT-5-Codex",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 400000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-22",
      "retires_on": "2026-07-23",
      "replacement": "gpt-5.5",
      "api_string": "gpt-5-codex",
      "notes": "Earlier Codex model. Deprecation announced 2026-04-22; shuts down 2026-07-23, replaced by gpt-5.5. Also covers gpt-5.1-codex* per the deprecation table. Context/max-output assumed gpt-5-class; price not captured.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "sora-2",
      "name": "Sora 2",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "video"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-03-24",
      "retires_on": "2026-09-24",
      "replacement": null,
      "api_string": "sora-2",
      "notes": "Video generation (Videos API). Priced per second, not per token: sora-2 $0.10/s (720p, standard) / $0.05/s batch. Videos API + sora-2* announced for discontinuation 2026-03-24; shuts down 2026-09-24 with no replacement (discontinued). Token-price fields N/A (per-second billing).",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "sora-2-pro",
      "name": "Sora 2 Pro",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "video"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-03-24",
      "retires_on": "2026-09-24",
      "replacement": null,
      "api_string": "sora-2-pro",
      "notes": "Higher-quality video model. Per-second pricing: $0.30/s (720p) up to $0.70/s (1080p) standard; batch half. Discontinued with Videos API, shuts down 2026-09-24, no replacement. Token-price fields N/A.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "dall-e-3",
      "name": "DALL-E 3",
      "provider": "OpenAI",
      "status": "deprecated",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2023-11-06",
      "deprecated_on": "2025-11-14",
      "retires_on": "2026-05-12",
      "replacement": "gpt-image-2",
      "api_string": "dall-e-3",
      "notes": "Legacy image model. Priced per-image (not per token). Deprecation announced 2025-11-14; retired 2026-05-12, replaced by gpt-image-2. As of 2026-06-28 this is past its shutdown date (effectively retired).",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "dall-e-2",
      "name": "DALL-E 2",
      "provider": "OpenAI",
      "status": "retired",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-11-14",
      "retires_on": "2026-05-12",
      "replacement": "gpt-image-2",
      "api_string": "dall-e-2",
      "notes": "Legacy image model, per-image pricing. Retired 2026-05-12 (past shutdown date as of 2026-06-28), replaced by gpt-image-2.",
      "source_url": "https://developers.openai.com/api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-1-pro-preview",
      "name": "Gemini 3.1 Pro Preview",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 12,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": "2025-01",
      "released": "2026-02-19",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.1-pro-preview",
      "notes": "Output modality: text only. Tiered pricing by prompt size: input $2.00 (<=200k tokens) / $4.00 (>200k); output $12.00 (<=200k) / $18.00 (>200k); cached input $0.20 (<=200k) / $0.40 (>200k) per 1M tokens. The values shown here are the <=200k tier. gemini-3-pro-preview alias now points to this model (the original gemini-3-pro-preview was shut down 2026-03-09). Preview/not-stable; may change. Release date refined from month precision (2026-02) to the exact day 2026-08-05, read first-party from the Release date column of Google's deprecations table, which states February 19, 2026. Source pages also include ai.google.dev/gemini-api/docs/pricing and /deprecations.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-pro-preview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-7-flash",
      "name": "Gemini 3.7 Flash",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.75,
      "price_output_per_mtok": 3.75,
      "price_cached_input_per_mtok": 0.075,
      "knowledge_cutoff": null,
      "released": "2026-08-13",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.7-flash",
      "notes": "GA/stable ('Stable: gemini-3.7-flash' on the model page), described by Google as 'our most capable Flash model' on the pricing page and 'our most intelligent workhorse model yet for coding and agents' in the August 13, 2026 release note. THE PRICE FIELDS CARRY THE COLUMN IN FORCE THROUGH 2026-12-31, not the 2027 one — Google prices this model on a two-date schedule and calls it an introductory price in as many words ('available at an introductory price through December 31, 2026'): paid-tier Standard is $0.75 in / $3.75 out per 1M with context caching $0.075/1M through December 31, 2026, doubling to $1.50 / $7.50 / $0.15 starting January 1, 2027. A dated ROADMAP ticket moves these fields on the cutover. Other published tiers, none of which fit the per-MTok fields, on the same two dates: Batch and Flex $0.375 in / $1.875 out (caching $0.0375) then $0.75 / $3.75 (caching $0.075); Priority $1.35 / $6.75 (caching $0.135) then $2.70 / $13.50 (caching $0.27). The per-1M-tokens-per-hour cache STORAGE fee this schema has no field for follows the same schedule: $0.50 through 2026-12-31, $1.00 from 2027-01-01. Grounding with Google Search or Maps: 5,000 requests/month free shared across all Gemini 3.x models, then $14 per 1,000. Spec table gives 1,048,576 input / 65,536 output tokens, inputs text/image/video/audio/PDF and outputs text; thinking is supported at low, medium and high (Google notes 'minimal' returns an error). Released 2026-08-13 per the API release notes; the deprecations table carries it more coarsely as 'August 2026' and announces no shutdown date. Google publishes no knowledge cutoff on any of the three surfaces, so that stays null. Priced identically to gemini-3.6-flash on all three axes during the introductory window.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.7-flash",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-6-flash",
      "name": "Gemini 3.6 Flash",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.75,
      "price_output_per_mtok": 3.75,
      "price_cached_input_per_mtok": 0.075,
      "knowledge_cutoff": null,
      "released": "2026-07-21",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.6-flash",
      "notes": "GA/stable ('Stable' on ai.google.dev/gemini-api/docs/models, described there as Google's latest Flash). Added 2026-07-31 after Google's deprecations table began naming it as the migration target for gemini-2.5-flash and the gemini-2.0-flash line. Prices are the paid-tier STANDARD rates from the official pricing page. CORRECTED 2026-08-14: the fields here carried Google's January-2027 column ($1.50/$7.50/$0.15) rather than the rate actually in force. Google prices this model on a two-date schedule: $0.75 in / $3.75 out per 1M with context caching $0.075/1M 'through December 31, 2026', doubling to $1.50 / $7.50 / $0.15 'starting January 1, 2027'. The per-1M-tokens-per-hour cache STORAGE fee this schema has no field for follows the same schedule — $0.50 through 2026-12-31, $1.00 from 2027-01-01 — and the note here previously gave the 2027 figure for that too. Other published tiers, none of which fit the per-MTok fields either, all on the same two dates: Batch and Flex $0.375/$1.875 (caching $0.0375) then $0.75/$3.75 (caching $0.075); Priority $1.35/$6.75 (caching $0.135) then $2.70/$13.50 (caching $0.27). Grounding with Google Search or Maps: 5,000 requests/month free shared across all Gemini 3.x models, then $14 per 1,000. Spec table gives 1,048,576 input / 65,536 output tokens and inputs text, image, video, audio, PDF. CORRECTED 2026-08-02: the note here previously asserted Google publishes no release date. That was true of the two surfaces it was read from (model page, pricing page) and false about the world — Google's DEPRECATIONS table carries it as data, 'gemini-3.6-flash | July 21, 2026 | No shutdown date announced', and the API changelog entry of 2026-07-21 says the same ('Gemini 3.6 Flash and Gemini 3.5 Flash-Lite generally available'). released is now 2026-07-21. The knowledge cutoff genuinely is unpublished on all three surfaces and stays null; the model page's 'Latest update July 2026' is a docs-edit date, not a launch. While the promotional window runs it is cheaper than Gemini 3.5 Flash on BOTH axes, not just output — 3.5 Flash is a flat $1.50 in / $9.00 out with no 2026/2027 split on its own block — so the two Flash rows are not a straight ladder, and the gap closes on 2027-01-01.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-5-flash",
      "name": "Gemini 3.5 Flash",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 1.5,
      "price_output_per_mtok": 9,
      "price_cached_input_per_mtok": 0.15,
      "knowledge_cutoff": "2025-01",
      "released": "2026-05-19",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.5-flash",
      "notes": "GA/stable. Output token limit reported as 'up to 65,000' on the what's-new page; model spec convention is 65,536 - treated as 65536. The what's-new page describes additional output capabilities (images, audio, structured outputs) but the spec table for the core text model lists text output. Was listed as the recommended replacement for gemini-2.5-flash and the gemini-2.0-flash line; as of the deprecations table read 2026-07-31 that target is gemini-3.6-flash, which is cheaper on output ($7.50 vs $9.00) at the same input price. Context-caching also has a per-hour storage fee not captured here. What's-new page last updated 2026-06-24 UTC.",
      "source_url": "https://ai.google.dev/gemini-api/docs/whats-new-gemini-3.5",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-flash-preview",
      "name": "Gemini 3 Flash Preview",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": 0.05,
      "knowledge_cutoff": "2025-01",
      "released": "2025-12-17",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "gemini-3.6-flash",
      "api_string": "gemini-3-flash-preview",
      "notes": "Output: text only. Pricing differs for audio input: input $0.50 (text/image/video) / $1.00 (audio); cached input $0.05 (text/image/video) / $0.10 (audio) per 1M tokens. Values shown are the text/image/video tier. Release date refined from month precision (2025-12) to the exact day 2026-08-05, read first-party from the Release date column of the same deprecations table, which states December 17, 2025. Lifecycle read first-party from Google's deprecations table 2026-07-31: listed with recommended replacement gemini-3.6-flash and 'No shutdown date announced', so replacement is set and retires_on stays null. (Supersedes an earlier note claiming it was 'not on the deprecation schedule as of 2026-06-28' — that was read from the model page, which omits lifecycle.)",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-5-flash-lite",
      "name": "Gemini 3.5 Flash-Lite",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.03,
      "knowledge_cutoff": null,
      "released": "2026-07-21",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.5-flash-lite",
      "notes": "GA/stable — 'Stable: gemini-3.5-flash-lite' on its model page, described by Google as 'our most cost-efficient GA model, optimized for high-volume agentic tasks, translation, and simple data processing'. Added 2026-08-02 because Google's deprecations table names it the recommended replacement for gemini-3.1-flash-lite (shutdown 2027-05-07). Prices are the paid-tier STANDARD rates: $0.30 in / $2.50 out per 1M, context caching $0.03/1M plus a $1.00 per 1M-tokens-per-hour storage fee this schema has no field for. Unlike Gemini 3.1 Flash-Lite and 2.5 Flash-Lite, there is NO higher audio-input tier — Google publishes one input rate covering text/image/video/audio. Batch tier $0.15 / $1.25. Grounding with Google Search or Maps: 5,000 requests/month free shared across all Gemini 3.x models, then $14 per 1,000. Spec table gives 1,048,576 input / 65,536 output tokens, inputs text/image/video/audio/PDF, output text only; caching, code execution, function calling, file search and Google Maps grounding supported, computer use in preview, no Live API and no image or audio generation. Release date July 21, 2026 read from Google's deprecations table and corroborated by the API changelog entry of the same date ('Gemini 3.6 Flash and Gemini 3.5 Flash-Lite generally available'); no shutdown date announced. Google publishes no knowledge cutoff for it on the model page, so that field is null rather than inherited from a sibling — the page's 'Latest update July 2026' is a docs-edit date, not a cutoff.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.5-flash-lite",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-1-flash-lite",
      "name": "Gemini 3.1 Flash-Lite",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.25,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": 0.025,
      "knowledge_cutoff": "2025-01",
      "released": "2026-05-07",
      "deprecated_on": null,
      "retires_on": "2027-05-07",
      "replacement": "gemini-3.5-flash-lite",
      "api_string": "gemini-3.1-flash-lite",
      "notes": "Listed as Stable on the models overview. Output: text only. Audio input priced higher: input $0.25 (text/image/video) / $0.50 (audio); cached input $0.025 (text/image/video) / $0.05 (audio) per 1M tokens. Deprecations page lists release 2026-05-07, shutdown 2027-05-07 and, as of 2026-08-02, a recommended replacement of gemini-3.5-flash-lite (added to the catalog the same day; the earlier note here saying no replacement was yet named was true of the table when read on 2026-07-31 and is superseded). It is the recommended replacement for gemini-2.5-flash-lite and the gemini-2.0-flash-lite line.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-pro",
      "name": "Gemini 2.5 Pro",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": 0.125,
      "knowledge_cutoff": "2025-01",
      "released": "2025-06-17",
      "deprecated_on": null,
      "retires_on": "2026-10-16",
      "replacement": "gemini-3.1-pro-preview",
      "api_string": "gemini-2.5-pro",
      "notes": "Output: text only. Tiered pricing: input $1.25 (<=200k tokens) / $2.50 (>200k); output $10.00 (<=200k) / $15.00 (>200k); cached input $0.125 (<=200k) / $0.25 (>200k) per 1M tokens (values shown are <=200k tier). Deprecations page: released 2025-06-17, shutdown 2026-10-16, replacement gemini-3.1-pro-preview. Still callable as of 2026-06-28 but on the deprecation schedule. Latest update June 2025.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-pro",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-flash",
      "name": "Gemini 2.5 Flash",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.03,
      "knowledge_cutoff": "2025-01",
      "released": "2025-06-17",
      "deprecated_on": null,
      "retires_on": "2026-10-16",
      "replacement": "gemini-3.6-flash",
      "api_string": "gemini-2.5-flash",
      "notes": "Output: text only. Audio input priced higher: input $0.30 (text/image/video) / $1.00 (audio); cached input $0.03 (text/image/video) / $0.10 (audio) per 1M tokens (values shown are text/image/video tier). Deprecations page: released 2025-06-17, shutdown 2026-10-16, replacement gemini-3.5-flash. Still callable as of 2026-06-28. Latest update June 2025. Replacement re-verified first-party 2026-07-31: Google's deprecations table (Gemini 2.5 Flash models) now names gemini-3.6-flash as the recommended replacement, shutdown October 16, 2026 (release June 17, 2025). Corrected here from gemini-3.5-flash, which this row carried until today — the migration target CHANGED upstream.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-flash",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-flash-lite",
      "name": "Gemini 2.5 Flash-Lite",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 1048576,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": 0.01,
      "knowledge_cutoff": "2025-01",
      "released": "2025-07-22",
      "deprecated_on": null,
      "retires_on": "2026-10-16",
      "replacement": "gemini-3.1-flash-lite",
      "api_string": "gemini-2.5-flash-lite",
      "notes": "Output: text only. Audio input priced higher: input $0.10 (text/image/video) / $0.30 (audio); cached input $0.01 (text/image/video) / $0.03 (audio) per 1M tokens (values shown are text/image/video tier). Deprecations page: released 2025-07-22, shutdown 2026-10-16, replacement gemini-3.1-flash-lite. Still callable as of 2026-06-28. Latest update July 2025.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-flash-lite",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-0-flash",
      "name": "Gemini 2.0 Flash",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 1048576,
      "max_output_tokens": 8192,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": 0.025,
      "knowledge_cutoff": "2024-08",
      "released": "2025-02-05",
      "deprecated_on": null,
      "retires_on": "2026-06-01",
      "replacement": "gemini-3.6-flash",
      "api_string": "gemini-2.0-flash",
      "notes": "RETIRED/shut down 2026-06-01 (model page shows 'Gemini 2.0 Flash is deprecated and has been shut down June 1, 2026'). As of today 2026-06-28 it is no longer callable. Output: text only (8,192 max output tokens). Audio input priced higher: input $0.10 (text/image/video) / $0.70 (audio); cached input $0.025 (text/image/video) / $0.175 (audio) per 1M tokens. Knowledge cutoff August 2024. gemini-2.0-flash-001 alias retired same day. Replacement gemini-3.5-flash. Pricing retained on page for reference. Replacement re-verified first-party 2026-07-31: Google's deprecations table (Gemini 2.0 models) names gemini-3.6-flash for both gemini-2.0-flash and gemini-2.0-flash-001, shutdown June 1, 2026. Corrected here from gemini-3.5-flash.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.0-flash",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-0-flash-lite",
      "name": "Gemini 2.0 Flash-Lite",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 1048576,
      "max_output_tokens": 8192,
      "price_input_per_mtok": 0.075,
      "price_output_per_mtok": 0.3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-08",
      "released": "2025-02-25",
      "deprecated_on": null,
      "retires_on": "2026-06-01",
      "replacement": "gemini-3.1-flash-lite",
      "api_string": "gemini-2.0-flash-lite",
      "notes": "RETIRED/shut down 2026-06-01 per Google's deprecations table (released 2025-02-25, replacement gemini-3.1-flash-lite; gemini-2.0-flash-lite-001 shares both dates). Re-verified first-party 2026-07-31: Google keeps shut-down models in BOTH the live pricing page and a dedicated model page, so every field here is now read, not inferred. Pricing page, paid tier: $0.075 in / $0.30 out per 1M tokens (batch tier $0.0375 / $0.15 — half, but no schema field for it yet), \"Context caching price: Not available\" on both tiers. Model page confirms inputs audio/images/video/text, 1,048,576 input tokens, 8,192 output tokens, knowledge cutoff August 2024. Note the two surfaces differ on caching: the model page lists Caching as Supported while the pricing page publishes no cached-input rate, so price_cached_input_per_mtok stays null — the capability existed, the price was never published (an earlier note here wrongly said caching was not offered).",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-embedding-2",
      "name": "Gemini Embedding 2",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-04-22",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-embedding-2",
      "notes": "Multimodal embedding model (stable/GA). Output is a text embedding vector, not tokens, so there is no output-token price. Input token limit 8,192; output dimension size flexible 128-3072 (recommended 768 / 1536 / 3072). The $0.20/1M figure is the TEXT input rate; the pricing page bills other input types separately - image $0.45, audio $6.50, video $12.00 per 1M (batch is half of each). Release date April 22, 2026 read first-party 2026-08-05 from the Release date column of Google's deprecations table, which carries release dates for CURRENT models too, not only dying ones; Google states 'No shutdown date announced' for it, so retires_on stays null. (Supersedes an earlier note claiming the docs list 'latest update April 2026' rather than a release date - that was true of the model page, which omits the release date; the deprecations table carries it as data.)",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": 3072,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-embedding-001",
      "name": "Gemini Embedding 001",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 2048,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-14",
      "deprecated_on": null,
      "retires_on": "2028-05-14",
      "replacement": "gemini-embedding-2",
      "api_string": "gemini-embedding-001",
      "notes": "Text-only embedding model (stable/GA). Input token limit 2,048; output dimension size flexible 128-3072 (recommended 768 / 1536 / 3072). Embeddings have no output-token billing; $0.15/1M input, $0.075/1M batch. Lifecycle read first-party from Google's deprecations table 2026-07-31: shutdown date May 14, 2028, recommended replacement gemini-embedding-2. Google states no deprecation status for it, so status stays ga (never infer a lifecycle change the provider has not declared). Release date July 14, 2025 read from the SAME table's Release date column 2026-08-05. (Supersedes two earlier claims: that Google 'state no deprecation or retirement date' — read from the embeddings docs, which omit lifecycle — and that the docs list 'last updated June 2025' rather than a release date, so released is null; both were true of the page opened and false about the deprecations table, which carries both fields as data.)",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": 3072,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-computer-use-preview",
      "name": "Gemini 2.5 Computer Use",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-2.5-computer-use-preview-10-2025",
      "notes": "Specialist UI-automation model that emits browser/computer actions from a screenshot - not a general chat model. Preview. Input image+text, output text; 128,000 input / 64,000 output tokens. Priced on the same two-tier scale as gemini-2.5-pro: $1.25/M in and $10.00/M out at <=200K prompt tokens, rising to $2.50/M and $15.00/M above it - the base tier is stored here, matching the gemini-2-5-pro row's convention. Google's model page notes that Gemini 3 Pro and Flash have built-in computer use without a separate model, but it declares NO deprecation or retirement date for this model, so lifecycle fields stay null. Cached-input rate not published for this model. Spec table latest update October 2025 (page revised 2026-04-28), which is not a release date, so released is null. Second first-party surface checked and CLOSED 2026-08-05: the deprecations table carries a Release date column for current models too, but this model appears nowhere in its 33 rows (only as a nav link), so Google publishes no release date for it on either surface and released stays null.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-2.5-computer-use-preview-10-2025",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-robotics-er-2-preview",
      "name": "Gemini Robotics-ER 2",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 131072,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": "2026-07-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-robotics-er-2-preview",
      "notes": "Embodied-reasoning (Gemini Robotics Embodied Reasoning 2) vision-language endpoint for robotics — advanced spatial reasoning, agentic code execution, multi-step tool orchestration, video moment finding, progress classification and multi-robot coordination. A specialist, not a general chat model. Public preview. Released 2026-07-30 per Google's API changelog ('Gemini Robotics ER 2 in public preview'), which shipped this and gemini-robotics-er-2-streaming-preview together. Paid-tier standard rates: $2.00/M in (text/image/video/audio), $10.00/M out including thinking tokens, context caching $0.20/1M plus a $1.00 per 1M-tokens-per-hour storage fee with no schema field; batch $1.00/$5.00. Spec table: 131,072 input / 65,536 output tokens, inputs text/images/video/audio, output text. It is the stated replacement for gemini-robotics-er-1.6-preview, which Google shuts down 2026-08-31. Google publishes no knowledge cutoff for it (the ER 1.6 spec table lists January 2025; the ER 2 table carries no cutoff row at all), so the field is null rather than inherited from the sibling. No shutdown date announced, and Google states no lifecycle status beyond the 'Preview' version label, so status is preview and retires_on is null.",
      "source_url": "https://ai.google.dev/gemini-api/docs/robotics-overview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-robotics-er-2-streaming-preview",
      "name": "Gemini Robotics-ER 2 Streaming",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 131072,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-07-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-robotics-er-2-streaming-preview",
      "notes": "The streaming sibling of gemini-robotics-er-2-preview, shipped in the same 2026-07-30 public-preview launch: optimized for real-time text streaming over the Live API, for low-latency robot agents processing continuous bidirectional audio and video input, with function calling. Same headline rates as the non-streaming endpoint ($2.00/M in, $10.00/M out, paid tier) and the same 131,072 / 65,536 token limits, inputs text/images/video/audio, output text. Its pricing block publishes NO context-caching rate (the non-streaming endpoint publishes $0.20/1M) and no batch tier, so price_cached_input_per_mtok is null — that is an absence on Google's own pricing page, not an unread value. No knowledge cutoff, no release-lifecycle status and no shutdown date are published for it.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-robotics-er-1-6-preview",
      "name": "Gemini Robotics-ER 1.6",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 131072,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2025-01",
      "released": "2026-04-14",
      "deprecated_on": null,
      "retires_on": "2026-08-31",
      "replacement": "gemini-robotics-er-2-preview",
      "api_string": "gemini-robotics-er-1.6-preview",
      "notes": "Embodied-reasoning vision-language model for robotics (spatial understanding, pointing, trajectory planning) - a specialist, not a general chat model. Preview. Inputs text/image/video/audio, output text; 131,072 input / 65,536 output tokens. $1.00/M in and $5.00/M out standard, $0.50/$2.50 batch. Supports function calling, structured output, thinking and caching, but no separate cached-input rate is published. Release date April 14, 2026 read first-party 2026-08-05 from the Release date column of Google's deprecations table (supersedes an earlier note reading the model page's 'spec table latest update December 2025' as not-a-release-date and leaving released null - the table states the real one as data). Shutdown date August 31, 2026 read first-party from Google's deprecations table 2026-07-31; Google states no lifecycle status for it, so status stays preview. Its recommended replacement there is gemini-robotics-er-2-preview, added to the catalog 2026-08-02 and now pointed at from this row; Google also offers gemini-robotics-er-2-streaming-preview as an upgrade path for Live API users.",
      "source_url": "https://ai.google.dev/gemini-api/docs/models/gemini-robotics-er-1.6-preview",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-5",
      "name": "Grok 4.5",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 500000,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 6,
      "price_cached_input_per_mtok": 0.3,
      "knowledge_cutoff": null,
      "released": "2026-07-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.5",
      "notes": "New flagship launched 2026-07-08 (public access 2026-07-09); xAI's smartest model for chat, coding, agentic and knowledge work per docs.x.ai. Aliases: grok-4.5-latest, grok-build-latest. Modality 'text, image -> text' (vision) per official model page. Cached input $0.30/M per the official model page (2026-07-22); it read $0.50/M on 2026-07-12, so this may be a cached-tier reduction, but the change is not independently timestamped by xAI and is recorded here as a value correction, not a changelog price-change event. TIERED pricing: figures shown are the base tier for prompts <=200K tokens; prompts >200K are billed 2x ($4/M input, $0.60/M cached, $12/M output). 500K context window (smaller than Grok 4.3's 1M). Max output tokens and knowledge cutoff not published on the docs page.",
      "source_url": "https://docs.x.ai/developers/models/grok-4.5",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-3",
      "name": "Grok 4.3",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": "2026-04-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.3",
      "notes": "Current flagship/recommended model for chat and coding per docs.x.ai/developers/models. Aliases: grok-4.3-latest, grok-latest. Modality 'text, image -> text' per official model page. Reasoning model (supports function calling, structured outputs, reasoning). Cached input $0.20/M confirmed on official model page. Max output tokens not published on the docs page (third-party sources describe 'no output token limit' but this is not officially stated). Knowledge cutoff not explicitly stated on the 4.x model page; the main models page only states the Nov 2024 cutoff for 'Grok 3 and Grok 4'. Release",
      "source_url": "https://docs.x.ai/developers/models/grok-4.3",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-20-0309-reasoning",
      "name": "Grok 4.20 (0309) Reasoning",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.20-0309-reasoning",
      "notes": "Reasoning-optimized variant. Aliases include grok-4.20-reasoning-latest, grok-4.20, grok-4.20-reasoning. Modality 'text, image -> text' per official model page. Pricing and cached-input ($0.20/M) confirmed on official model page. Max output tokens, knowledge cutoff and release date not published on docs. Note: docs warn no logprobs support for grok-4.20+.",
      "source_url": "https://docs.x.ai/developers/models/grok-4.20-0309-reasoning",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-20-0309-non-reasoning",
      "name": "Grok 4.20 (0309) Non-Reasoning",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.20-0309-non-reasoning",
      "notes": "Non-reasoning (latency-sensitive) variant; reasoning disabled, function calling and structured outputs supported. Aliases include grok-4.20-non-reasoning-latest. Modality 'text, image -> text'. Pricing and cached input ($0.20/M) confirmed on official model page. Max output tokens, knowledge cutoff and release date not published on docs.",
      "source_url": "https://docs.x.ai/developers/models/grok-4.20-0309-non-reasoning",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-20-multi-agent-0309",
      "name": "Grok 4.20 Multi-Agent (0309)",
      "provider": "xAI",
      "status": "beta",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-4.20-multi-agent-0309",
      "notes": "Marked Beta on the official model page; designed for multi-agent orchestration. Aliases include grok-4.20-multi-agent, grok-4.20-multi-agent-latest. Modality 'text, image -> text'. Lower rate limits than the standard 4.20 SKUs (9 req/s, 2.5M tokens/min vs 37 req/s, 10M tokens/min). PRICE: $1.25 in / $2.50 out / $0.20 cached per 1M, re-read and confirmed first-party at docs.x.ai/docs/models on 2026-08-03 (the <200K-token tier; xAI doubles every rate to $2.50/$5.00/$0.40 above 200K input tokens, which this schema has no column for). CONTEXT: docs.x.ai states 1M and that is what we carry; a third-party claim of 2M appears on no xAI surface we have read, so it is not used.",
      "source_url": "https://docs.x.ai/developers/models/grok-4.20-multi-agent-0309",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-build-0-1",
      "name": "Grok Build 0.1",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-build-0.1",
      "notes": "Coding-focused model. Aliases: grok-code-fast-1, grok-code-fast, grok-code-fast-1-0825 (the retired grok-code-fast-1 slug now redirects here). Modality 'text, image -> text'. PRICE: $1.00 in / $2.00 out / $0.20 cached per 1M, re-read and confirmed first-party at docs.x.ai/docs/models on 2026-08-03 (the <200K-token tier; above 200K input tokens xAI charges $2.00/$4.00/$0.40, which this schema has no column for). CONTEXT: 256K, stated on the same table. RELEASE DATE: xAI publishes none on its models page or the per-model page, so released stays null; a third-party 2026-05-29 launch date exists but appears on no xAI surface.",
      "source_url": "https://docs.x.ai/developers/models/grok-build-0.1",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-imagine-image-quality",
      "name": "Grok Imagine Image (Quality)",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-imagine-image-quality",
      "notes": "Image-generation model priced per image ($0.05/image), not per token, so token price fields are null. This is the redirect/replacement target for the retired grok-imagine-image-pro (retired 2026-05-15). Standard tier grok-imagine-image is $0.02/image.",
      "source_url": "https://docs.x.ai/developers/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-imagine-image",
      "name": "Grok Imagine Image",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-imagine-image",
      "notes": "Standard image-generation model priced per image ($0.02/image), not per token. Token price fields null by design.",
      "source_url": "https://docs.x.ai/developers/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-imagine-video-1-5",
      "name": "Grok Imagine Video 1.5",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-imagine-video-1.5",
      "notes": "Video-generation model priced per second of output ($0.080/sec), not per token. Token price fields null by design.",
      "source_url": "https://docs.x.ai/developers/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-imagine-video",
      "name": "Grok Imagine Video",
      "provider": "xAI",
      "status": "ga",
      "modality": [
        "text-in",
        "image-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "grok-imagine-video",
      "notes": "Video-generation model priced per second ($0.050/sec), not per token. Token price fields null by design.",
      "source_url": "https://docs.x.ai/developers/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-0709",
      "name": "Grok 4 (0709)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-11",
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-0709",
      "notes": "Retired from the xAI API on 2026-05-15 12:00 PM PT per official migration page. Requests to this slug now auto-redirect to grok-4.3 with 'low' reasoning effort; the slug still resolves so existing code does not break. Knowledge cutoff Nov 2024 per docs note covering Grok 4. Per-token pricing no longer published on docs (was historically $3/$15 per M per third-party trackers); left null as not officially current.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-3",
      "name": "Grok 3",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "text-out"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-11",
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-3",
      "notes": "Retired from the xAI API on 2026-05-15 12:00 PM PT per official migration page; redirects to grok-4.3 with 'none' reasoning effort, slug still resolves. ~131K context window per third-party sources (not re-confirmed on current docs). Knowledge cutoff Nov 2024 per docs. Per-token pricing no longer published on docs; left null.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-fast-reasoning",
      "name": "Grok 4 Fast (Reasoning)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-fast-reasoning",
      "notes": "Retired from xAI API on 2026-05-15 12:00 PM PT; redirects to grok-4.3 with 'low' reasoning effort, slug still resolves. Specs/pricing no longer on docs; left null. (Oracle OCI mirror lists deprecated 2026-05-15, retires 2026-08-15 on their platform, but the xAI-native retirement date is 2026-05-15.)",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-fast-non-reasoning",
      "name": "Grok 4 Fast (Non-Reasoning)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-fast-non-reasoning",
      "notes": "Retired from xAI API on 2026-05-15 12:00 PM PT; redirects to grok-4.3 with 'none' reasoning effort, slug still resolves. Specs/pricing no longer on docs; left null.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-1-fast-reasoning",
      "name": "Grok 4.1 Fast (Reasoning)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-1-fast-reasoning",
      "notes": "Retired from xAI API on 2026-05-15 12:00 PM PT; redirects to grok-4.3 with 'low' reasoning effort, slug still resolves. Specs/pricing no longer on docs; left null. (Was ~$0.20/$0.50 per M per third-party trackers before retirement; not officially current.)",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-4-1-fast-non-reasoning",
      "name": "Grok 4.1 Fast (Non-Reasoning)",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-4.3",
      "api_string": "grok-4-1-fast-non-reasoning",
      "notes": "Retired from xAI API on 2026-05-15 12:00 PM PT; redirects to grok-4.3 with 'none' reasoning effort, slug still resolves. Specs/pricing no longer on docs; left null.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-code-fast-1",
      "name": "Grok Code Fast 1",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-in",
        "text-out"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-build-0.1",
      "api_string": "grok-code-fast-1",
      "notes": "Retired as a standalone slug on 2026-05-15 12:00 PM PT; redirects to grok-build-0.1. Note grok-code-fast-1 also persists as an ALIAS of grok-build-0.1 on the current model page, so the name still resolves. 256k context window inferred from grok-build-0.1 (same model lineage).",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "grok-imagine-image-pro",
      "name": "Grok Imagine Image Pro",
      "provider": "xAI",
      "status": "retired",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-05-15",
      "replacement": "grok-imagine-image-quality",
      "api_string": "grok-imagine-image-pro",
      "notes": "Image-generation model retired on 2026-05-15 12:00 PM PT; redirects to grok-imagine-image-quality. Priced per image, not per token, so token fields null.",
      "source_url": "https://docs.x.ai/developers/migration/may-15-retirement",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-4-maverick",
      "name": "Llama 4 Maverick (17B-128E Instruct)",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": 0.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-08",
      "released": "2025-04-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
      "notes": "Open-weight, natively multimodal MoE: 17B active / 400B total params, 128 experts. License: Llama 4 Community License Agreement (commercial use permitted for orgs with <700M MAU). Meta is the model owner; no first-party Meta API pricing for self-host. Official Llama API model ID is 'Llama-4-Maverick-17B-128E-Instruct-FP8' and the Llama API serves it at a 128k context window (developer.meta.com/Llama API docs), whereas the open weights support up to 1M tokens. Hosted price shown is OpenRouter slug 'meta-llama/llama-4-maverick' = $0.15 in / $0.60 out per 1M (OpenRouter page accessed 2026-06-28).",
      "source_url": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-4-scout",
      "name": "Llama 4 Scout (17B-16E Instruct)",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 10000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2024-08",
      "released": "2025-04-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-4-Scout-17B-16E-Instruct-FP8",
      "notes": "Open-weight, natively multimodal MoE: 17B active / 109B total params, 16 experts; fits on a single H100. License: Llama 4 Community License Agreement. Open weights support up to 10M-token context; the official Llama API serves it at 128k (model ID 'Llama-4-Scout-17B-16E-Instruct-FP8'). Hosted price is OpenRouter slug 'meta-llama/llama-4-scout' = $0.10 in / $0.30 out per 1M (page accessed 2026-06-28); Together AI reported ~$0.08 in / $0.30 out per 1M. No cached-input discount published. Same 12 supported languages as Maverick.",
      "source_url": "https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-4-behemoth",
      "name": "Llama 4 Behemoth (preview)",
      "provider": "Meta",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": null,
      "notes": "Announced as a preview/teacher model (~288B active params, 16 experts, ~2T total) used to distill Scout and Maverick. As of 2026-06-28 it is NOT released as open weights and has no public API string, pricing, context window, or confirmed release date. Included only because it is a currently-announced member of the Llama 4 herd. All numeric fields null (unconfirmed).",
      "source_url": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-3-3-70b-instruct",
      "name": "Llama 3.3 70B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.32,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-12-06",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.3-70B-Instruct",
      "notes": "Open-weight, text-only, 70B dense (GQA), 128k context. License: Llama 3.3 Community License Agreement. Also a first-party Llama API model ID 'Llama-3.3-70B-Instruct'. Hosted price = OpenRouter slug 'meta-llama/llama-3.3-70b-instruct' $0.10 in / $0.32 out per 1M (accessed 2026-06-28); Together AI lists ~$0.88-$1.04 per 1M flat (Together pricing page, accessed 2026-06-28). Free tier also exists on OpenRouter. Langs: English, German, French, Italian, Portuguese, Hindi, Spanish, Thai.",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-3-3-8b-instruct",
      "name": "Llama 3.3 8B Instruct (Llama API)",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "Llama-3.3-8B-Instruct",
      "notes": "Listed on the official Meta Llama API models page as a lightweight, ultra-fast text-only variant with 128k context (model ID 'Llama-3.3-8B-Instruct'). NOTE/UNCERTAINTY: there is no corresponding standalone 'Llama-3.3-8B' open-weight checkpoint on Meta's Hugging Face org (the 8B open weight in this generation is Llama-3.1-8B-Instruct); this ID appears specific to the hosted Llama API. Pricing, exact release date, and knowledge cutoff not published on the API models page (null). Not separately priced on OpenRouter/Together under this exact name. SOURCED ABSENCE, re-resolved 2026-08-12: the hosted API this row came from has been rebranded from the Llama API to the Meta Model API (llama.developer.meta.com now 302s to ai.developer.meta.com), and its models page no longer lists ANY Llama model - the 'Available models' table carries only muse-spark-1.1, muse-spark-1.2 and muse-spark-1.2-contributor. No first-party Meta surface names 'Llama-3.3-8B-Instruct' any more (llama.com/models is a 404). Status left at 'ga' because Meta has DECLARED no deprecation or retirement; delisting is not a declaration and this project records what a provider states, never what it implies.",
      "source_url": "https://ai.developer.meta.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-3-1-405b-instruct",
      "name": "Llama 3.1 405B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-07-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.1-405B-Instruct",
      "notes": "Open-weight, text-only, 405B dense, 128k context. License: Llama 3.1 Community License. Released 2024-07-23 alongside 8B/70B. Hosted price NOT reliably extractable in this pass: OpenRouter slug is 'meta-llama/llama-3.1-405b-instruct' (131k ctx) but the per-token figures did not render; Together AI reportedly does NOT serve 405B on its serverless API (their 405B model page states it is not available serverless), and third-party reports put serverless 405B around $3.00-$3.50 per 1M elsewhere (unverified). Prices left null rather than guessed. Source URL is the Llama 3.1 collection card (lists 8B",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-3-1-70b-instruct",
      "name": "Llama 3.1 70B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-07-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "llama-3-3-70b-instruct",
      "api_string": "meta-llama/Llama-3.1-70B-Instruct",
      "notes": "Open-weight, text-only, 70B dense, 128k context. License: Llama 3.1 Community License. Largely superseded by Llama 3.3 70B Instruct (same size, improved quality, Dec 2024) which Meta positions as the recommended 70B; not formally 'deprecated/retired' (open weights remain downloadable), so 'replacement' is advisory, not an enforced retirement. retires_on field repurposed here to point to the recommended successor id (no official retirement date exists - treat as null date). Per-token hosted price not separately captured this pass (null).",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-3-1-8b-instruct",
      "name": "Llama 3.1 8B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.02,
      "price_output_per_mtok": 0.03,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-07-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.1-8B-Instruct",
      "notes": "Open-weight, text-only, 8B dense, 128k context. License: Llama 3.1 Community License. Hosted price = OpenRouter slug 'meta-llama/llama-3.1-8b-instruct' $0.02 in / $0.03 out per 1M (page accessed 2026-06-28; 131k ctx). Together AI reports ~$0.18 per 1M flat. Free tier also available. Langs: English, German, French, Italian, Portuguese, Hindi, Spanish, Thai.",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-3-2-90b-vision-instruct",
      "name": "Llama 3.2 90B Vision Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.2-90B-Vision-Instruct",
      "notes": "Open-weight multimodal (text+image), ~88.8B params, 128k context. License: Llama 3.2 Community License. Source is the 3.2 11B Vision card which documents the 90B sibling (same architecture/128k ctx/Dec-2023 cutoff). Image+text tasks are English-primary; text-only adds German, French, Italian, Portuguese, Hindi, Spanish, Thai. Per-token hosted price not separately captured this pass (null). Dedicated card: huggingface.co/meta-llama/Llama-3.2-90B-Vision-Instruct.",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-3-2-11b-vision-instruct",
      "name": "Llama 3.2 11B Vision Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.2-11B-Vision-Instruct",
      "notes": "Open-weight multimodal (text+image), 10.6B params, 128k context. License: Llama 3.2 Community License. Image+text tasks English-primary; text-only adds German, French, Italian, Portuguese, Hindi, Spanish, Thai. Per-token hosted price not separately captured this pass (null).",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-3-2-3b-instruct",
      "name": "Llama 3.2 3B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.2-3B-Instruct",
      "notes": "Open-weight, text-only lightweight/on-device model, 128k context. License: Llama 3.2 Community License. Referenced on the 3.2 collection/11B card and has a dedicated card at huggingface.co/meta-llama/Llama-3.2-3B-Instruct. Free tier exists on OpenRouter (slug 'meta-llama/llama-3.2-3b-instruct'); paid per-token figures not captured this pass (null).",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "llama-3-2-1b-instruct",
      "name": "Llama 3.2 1B Instruct",
      "provider": "Meta",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2023-12",
      "released": "2024-09-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "meta-llama/Llama-3.2-1B-Instruct",
      "notes": "Open-weight, text-only smallest on-device model, 128k context. License: Llama 3.2 Community License. Dedicated card at huggingface.co/meta-llama/Llama-3.2-1B-Instruct. Per-token hosted price not captured this pass (null).",
      "source_url": "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mistral-large-3",
      "name": "Mistral Large 3",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "mistral-large-2512",
      "notes": "Current flagship (v25.12). 'latest' alias mistral-large-latest -> mistral-large-2512. Open-weight, sparse MoE 41B active / 675B total. 256k context per official model page. Pricing from mistral.ai/pricing (page fetched 2026-06-28): $0.5/M in, $1.5/M out. Max output and knowledge cutoff not published on official pages.",
      "source_url": "https://docs.mistral.ai/models/mistral-large-3-25-12",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mistral-medium-3-5",
      "name": "Mistral Medium 3.5",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.5,
      "price_output_per_mtok": 7.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-04-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "mistral-medium-latest",
      "notes": "Current frontier multimodal model (v26.04), agentic/coding focus. Dated alias resolves to mistral-medium-26.04 (full dated string not explicitly published; mistral-medium-2508 is the now-deprecated 3.1). Pricing $1.5/M in, $7.5/M out (mistral.ai/pricing, 2026-06-28). Listed because it is the named replacement for several deprecated models (Large 2.1, Pixtral Large). Context window not published on official overview. Release date added 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://mistral.ai/pricing/",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mistral-small-4",
      "name": "Mistral Small 4",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": 0.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-03-16",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "mistral-small-2603",
      "notes": "Current Small (v26.03). 'latest' alias mistral-small-latest -> mistral-small-2603. Hybrid instruct/reasoning/coding, 119B params (6.5B active), 256k context. Open-weight (Open v26.03). Pricing $0.15/M in, $0.6/M out (mistral.ai/pricing, 2026-06-28).",
      "source_url": "https://docs.mistral.ai/models/mistral-small-4-0-26-03",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mistral-small-3-2",
      "name": "Mistral Small 3.2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-06-20",
      "deprecated_on": "2026-04-30",
      "retires_on": "2026-07-31",
      "replacement": "mistral-small-2603",
      "api_string": "mistral-small-2506",
      "notes": "v25.06. Deprecated 2026-04-30, retirement 2026-07-31, replaced by Mistral Small 4. Pricing no longer listed on current pricing page. Release date added 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "codestral",
      "name": "Codestral (v25.08)",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "code"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 0.9,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "codestral-2508",
      "notes": "Current Codestral (Premier v25.08). 'latest' alias codestral-latest -> codestral-2508. Code completion / FIM, 128k context. Pricing $0.3/M in, $0.9/M out (mistral.ai/pricing, 2026-06-28). Release date 'July 30, 2025' per model page.",
      "source_url": "https://docs.mistral.ai/models/codestral-25-08",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "codestral-2501",
      "name": "Codestral (v25.01)",
      "provider": "Mistral",
      "status": "retired",
      "modality": [
        "text",
        "code"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-01-13",
      "deprecated_on": "2025-11-06",
      "retires_on": "2025-11-30",
      "replacement": "codestral-2508",
      "api_string": "codestral-2501",
      "notes": "Deprecated 2025-11-06, retired 2025-11-30, replaced by Codestral v25.08. An earlier codestral-2405 alias also exists in the legacy list. Release date added 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "codestral-embed",
      "name": "Codestral Embed",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "code"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "codestral-embed-2505",
      "notes": "Premier code-embedding model (v25.05). Embeddings only (input-priced): $0.15/M input tokens (mistral.ai/pricing, 2026-06-28). Max input 8,192 tokens (contextLength '8k' in the open-source docs schema codestral-embed-25-05.ts). Output dimension configurable via output_dimension: default 1536, max 3072 (docs.mistral.ai/capabilities/embeddings/code_embeddings). releaseDate 2025-05-28 per the docs schema.",
      "source_url": "https://docs.mistral.ai/models/codestral-embed-25-05",
      "open_weight": false,
      "embedding_dimensions": 1536,
      "price_per_1k_searches": null
    },
    {
      "id": "pixtral-large",
      "name": "Pixtral Large",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-11-18",
      "deprecated_on": "2026-02-27",
      "retires_on": "2026-05-31",
      "replacement": "mistral-medium-latest",
      "api_string": "pixtral-large-2411",
      "notes": "First frontier-class multimodal model (v24.11), 128k context. Deprecated 2026-02-27, retirement 2026-05-31, replaced by Mistral Medium 3.5. As of 2026-06-28 it is past its retirement date and no longer on the pricing page; classified deprecated/retiring per official legacy table. 'latest' alias pixtral-large-latest historically mapped here.",
      "source_url": "https://docs.mistral.ai/models/pixtral-large-24-11",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "pixtral-12b",
      "name": "Pixtral 12B",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-09-11",
      "deprecated_on": "2025-12-02",
      "retires_on": "2025-12-31",
      "replacement": "ministral-14b-2512",
      "api_string": "pixtral-12b-2409",
      "notes": "Open-weight 12B vision model (v24.09). Deprecated 2025-12-02, retired 2025-12-31, replaced by Ministral 3 14B. Past retirement as of 2026-06-28. Release date refined from '2024-09' 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "ministral-3-14b",
      "name": "Ministral 3 14B",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": 0.2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "ministral-14b-2512",
      "notes": "Ministral 3 family (v25.12), open-weight Apache 2.0, text+vision, 40+ languages. Pricing $0.2/M in & out (mistral.ai/pricing, 2026-06-28, listed as 'Ministral 14B'). Context window not published on official overview (Large 3 / Small 4 are 256k; Ministral 3 not explicitly stated). Dated alias expected ministral-3-14b-2512. CORRECTED 2026-08-07: api_string was 'ministral-3-14b-latest', which Mistral does not publish — the docs model page and the docs schema repo both state 'ministral-14b-2512' with the alias 'ministral-14b-latest' (read https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models/ministral-3-14b-25-12",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "ministral-3-8b",
      "name": "Ministral 3 8B",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": 0.15,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "ministral-8b-2512",
      "notes": "Ministral 3 family (v25.12), open-weight Apache 2.0, text+vision, edge deployment. Official model page states 256k context. Pricing $0.15/M in & out (mistral.ai/pricing 'Ministral 8B', 2026-06-28). Dated alias on page shown as ministral-8b-2512 / ministral-3-8b-2512. CORRECTED 2026-08-07: api_string was 'ministral-3-8b-latest', which Mistral does not publish — the docs model page and the docs schema repo both state 'ministral-8b-2512' with the alias 'ministral-8b-latest' (read https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models/ministral-3-8b-25-12",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "ministral-3-3b",
      "name": "Ministral 3 3B",
      "provider": "Mistral",
      "status": "ga",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.1,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "ministral-3b-2512",
      "notes": "Smallest Ministral 3 (v25.12), open-weight Apache 2.0, text+vision. Pricing $0.10/M in & out per the official API pricing page (mistral.ai/pricing/api), verified 2026-07-09. CORRECTION: previously stored as $0.04/M from a weak general-pricing 'cheapest tier' search corroboration — the authoritative per-model API table lists $0.10/M in & out. Context window not published on official overview. Dated alias expected ministral-3-3b-2512. CORRECTED 2026-08-07: api_string was 'ministral-3-3b-latest', which Mistral does not publish — the docs model page and the docs schema repo both state 'ministral-3b-2512' with the alias 'ministral-3b-latest' (read https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models/ministral-3-3b-25-12",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "ministral-8b-2410",
      "name": "Ministral 8B (2410)",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-10-09",
      "deprecated_on": "2025-12-02",
      "retires_on": "2025-12-31",
      "replacement": "ministral-8b-2512",
      "api_string": "ministral-8b-2410",
      "notes": "Original Ministral 8B (v24.10). Deprecated 2025-12-02, retired 2025-12-31, replaced by Ministral 3 8B. Past retirement as of 2026-06-28. Release date refined from '2024-10' 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "ministral-3b-2410",
      "name": "Ministral 3B (2410)",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-10-09",
      "deprecated_on": "2025-12-02",
      "retires_on": "2025-12-31",
      "replacement": "ministral-3b-2512",
      "api_string": "ministral-3b-2410",
      "notes": "Original Ministral 3B (v24.10). Deprecated 2025-12-02, retired 2025-12-31, replaced by Ministral 3 3B. Past retirement as of 2026-06-28. Release date refined from '2024-10' 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mistral-large-2-1",
      "name": "Mistral Large 2.1",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-11-18",
      "deprecated_on": "2026-02-27",
      "retires_on": "2026-05-31",
      "replacement": "mistral-medium-latest",
      "api_string": "mistral-large-2411",
      "notes": "v24.11. Deprecated 2026-02-27, retirement 2026-05-31, replaced by Mistral Medium 3.5. Past retirement as of 2026-06-28. Release date refined from '2024-11' 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mistral-large-2-0",
      "name": "Mistral Large 2.0",
      "provider": "Mistral",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-07-24",
      "deprecated_on": "2024-11-30",
      "retires_on": "2025-03-30",
      "replacement": "mistral-large-2512",
      "api_string": "mistral-large-2407",
      "notes": "v24.07. Deprecated 2024-11-30, retired 2025-03-30, replacement Mistral Large 3. Release date refined from '2024-07' 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mistral-medium-3-1",
      "name": "Mistral Medium 3.1",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-08-12",
      "deprecated_on": "2026-05-22",
      "retires_on": "2026-08-31",
      "replacement": "mistral-medium-latest",
      "api_string": "mistral-medium-2508",
      "notes": "v25.08. Deprecated 2026-05-22, retirement 2026-08-31, replaced by Mistral Medium 3.5. Still in deprecation window as of 2026-06-28. Release date refined from '2025-08' 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mistral-nemo",
      "name": "Mistral NeMo",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-07-18",
      "deprecated_on": "2026-05-22",
      "retires_on": "2026-07-31",
      "replacement": "ministral-8b-2512",
      "api_string": "open-mistral-nemo-2407",
      "notes": "Open-weight 12B (v24.07, with NVIDIA). Deprecated 2026-05-22, retirement 2026-07-31, replaced by Ministral 3 8B. Still in deprecation window as of 2026-06-28; pricing no longer on current page. Release date refined from '2024-07' 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mixtral-8x22b",
      "name": "Mixtral 8x22B",
      "provider": "Mistral",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 64000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-04-17",
      "deprecated_on": "2024-11-30",
      "retires_on": "2025-03-30",
      "replacement": "mistral-small-2603",
      "api_string": "open-mixtral-8x22b",
      "notes": "Open-weight MoE. Deprecated 2024-11-30, retired 2025-03-30, replacement Mistral Small 4. Context 64k commonly cited but not confirmed on the legacy table; treat as approximate. Release date refined from '2024-04' 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "mixtral-8x7b",
      "name": "Mixtral 8x7B",
      "provider": "Mistral",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 32000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2023-12-11",
      "deprecated_on": "2024-11-30",
      "retires_on": "2025-03-30",
      "replacement": "mistral-small-2603",
      "api_string": "open-mixtral-8x7b",
      "notes": "Open-weight MoE. Deprecated 2024-11-30, retired 2025-03-30, replacement Mistral Small 4. Context 32k commonly cited but not confirmed on the legacy table; treat as approximate. Release date refined from '2023-12' 2026-08-07 from Mistral's typed docs schema (`releaseDate`, https://github.com/mistralai/platform-docs-public/tree/main/src/schema/models/models).",
      "source_url": "https://docs.mistral.ai/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "magistral-medium-1-2",
      "name": "Magistral Medium 1.2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-18",
      "deprecated_on": "2026-05-22",
      "retires_on": "2026-07-31",
      "replacement": "mistral-medium-3-5",
      "api_string": "magistral-medium-2509",
      "notes": "Mistral's frontier multimodal reasoning model (v25.09); alias magistral-medium-latest -> magistral-medium-2509. Premier tier: API-only, no published weights. Text + image in, reasoning + text out. DEPRECATED 2026-05-22, RETIRES 2026-07-31 -> Mistral Medium 3.5, as Mistral folds its specialist reasoning line back into its general-purpose models. Lifecycle, price and context from Mistral's official docs model schema; $2/M in, $5/M out corroborated by the per-model table at mistral.ai/pricing/api, which still lists it as a current offering (the docs schema is the more precise surface for lifecycle). Max output and knowledge cutoff not published -> null.",
      "source_url": "https://docs.mistral.ai/models/magistral-medium-1-2-25-09",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "magistral-small-1-2",
      "name": "Magistral Small 1.2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-18",
      "deprecated_on": "2026-04-30",
      "retires_on": "2026-07-31",
      "replacement": "mistral-small-4",
      "api_string": "magistral-small-2509",
      "notes": "Mistral's small multimodal reasoning model (v25.09); alias magistral-small-latest -> magistral-small-2509. Open weights: Apache 2.0, 24B dense (huggingface.co/mistralai/Magistral-Small-2509). Text + image in, reasoning + text out. DEPRECATED 2026-04-30, RETIRES 2026-07-31 -> Mistral Small 4. Lifecycle, price and context from Mistral's official docs model schema; $0.5/M in, $1.50/M out corroborated by mistral.ai/pricing/api. Max output and knowledge cutoff not published -> null.",
      "source_url": "https://docs.mistral.ai/models/magistral-small-1-2-25-09",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "devstral-2",
      "name": "Devstral 2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-09",
      "deprecated_on": "2026-05-22",
      "retires_on": "2026-07-31",
      "replacement": "mistral-medium-3-5",
      "api_string": "devstral-2512",
      "notes": "Mistral's frontier agentic-coding model for software-engineering tasks (v25.12); aliases devstral-latest and devstral-medium-latest -> devstral-2512. Open weights: Modified MIT, 123B (huggingface.co/mistralai/Devstral-2-123B-Instruct-2512). Text in, text out. DEPRECATED 2026-05-22, RETIRES 2026-07-31 -> Mistral Medium 3.5, as Mistral folds its specialist coding line back into its general-purpose models. Lifecycle, price and context from Mistral's official docs model schema; $0.40/M in, $2/M out corroborated by mistral.ai/pricing/api, which still lists it as a current offering. Max output and knowledge cutoff not published -> null.",
      "source_url": "https://docs.mistral.ai/models/devstral-2-25-12",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "devstral-small-2",
      "name": "Devstral Small 2",
      "provider": "Mistral",
      "status": "deprecated",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-09",
      "deprecated_on": "2026-02-27",
      "retires_on": "2026-03-31",
      "replacement": "mistral-medium-3-5",
      "api_string": "labs-devstral-small-2512",
      "notes": "Small agentic-coding model (v25.12), Labs tier; alias devstral-small-latest -> labs-devstral-small-2512. Open weights: Apache 2.0, 24B. Text + image in, text out. DEPRECATED 2026-02-27 with a stated retirement of 2026-03-31 -> Mistral Medium 3.5. That date has passed, but Mistral still lists the model as deprecated rather than retired and still prices it publicly, so we record the status Mistral states rather than inferring a retirement it has not declared. Price and context from the official docs model schema, corroborated by mistral.ai/pricing/api. Max output and knowledge cutoff not published -> null.",
      "source_url": "https://docs.mistral.ai/models/devstral-small-2-25-12",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "labs-leanstral-1-5",
      "name": "Leanstral 1.5",
      "provider": "Mistral",
      "status": "preview",
      "modality": [
        "text",
        "vision"
      ],
      "context_window": 256000,
      "max_output_tokens": 128000,
      "price_input_per_mtok": 0,
      "price_output_per_mtok": 0,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-06-30",
      "deprecated_on": null,
      "retires_on": "2026-09-30",
      "replacement": null,
      "api_string": "labs-leanstral-1-5",
      "notes": "Specialist model for Lean 4 formal proof engineering, automated theorem proving and autoformalization -> not a general chat model. Labs tier, sparse MoE 119B total / 6.5B active. FREE during the Labs preview: pricing.free=true, $0/M in and out per the official docs model schema. Open weights: Apache 2.0 (huggingface.co/mistralai/Leanstral-1.5-119B-A6B). Text + image in, text out; 256K context, 128K max output. Mistral marks it status Active in its experimental Labs tier with a stated retirement of 2026-09-30 -> we normalize Labs/experimental to preview (it is not a GA production model), which also keeps the catalog's first $0 model out of the GA 'cheapest' rankings where a free specialist would rank misleadingly. Knowledge cutoff not published -> null.",
      "source_url": "https://mistral.ai/news/leanstral-1-5/",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "deepseek-v4-flash",
      "name": "DeepSeek-V4-Flash",
      "provider": "DeepSeek",
      "status": "preview",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "price_input_per_mtok": 0.44,
      "price_output_per_mtok": 1.32,
      "price_cached_input_per_mtok": 0.014,
      "knowledge_cutoff": null,
      "released": "2026-04-24",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "deepseek-v4-flash",
      "notes": "Smaller/cheaper V4 model (~284B total / ~13B active params per authoritative third-party reports). Context length 1M, max output 384K tokens. Input price $0.44/M cache-miss, $0.014/M cache-hit; output $1.32/M (USD). Supports dual modes (Thinking / Non-Thinking), JSON output, tool calls, chat prefix completion; FIM completion is non-thinking-mode only. Concurrency limit 2500. The legacy aliases deepseek-chat and deepseek-reasoner currently route to this model (non-thinking / thinking respectively). Part of the 'DeepSeek V4 Preview' generation (released 2026-04-24), hence status=preview. Knowledge cutoff NOT officially published by DeepSeek -> left null. Prices re-verified to the cent against api-docs.deepseek.com/quick_start/pricing/ on 2026-08-09. VENDOR-DECLARED FORWARD PRICE EVENT, RESOLVED 2026-08-14 — the same footnote now carries a date and a table. Until 2026-08-13 it read only 'We plan to raise the overall pricing for DeepSeek API services in the near future, with a significant increase expected... subject to official notice', with no date and no figure. DeepSeek now states: 'DeepSeek API pricing will be updated to peak / off-peak billing, with off-peak rates at half the peak rates. Peak hours are 01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak). The new prices take effect at 16:00 UTC on August 16, 2026'. Announced rates for this model per 1M tokens: OFF-PEAK $0.007 cache-hit / $0.22 cache-miss / $0.66 out; PEAK $0.014 cache-hit / $0.44 cache-miss / $1.32 out. EXECUTED 2026-08-16: the priced fields above were re-pointed at the PEAK column on the cutover day, because DeepSeek presents peak as the rate and off-peak as half of it, not the other way round. The OFF-PEAK figure is exactly half on every axis and is what you actually pay for 17 of every 24 hours, since peak is only 01:00-04:00 and 06:00-10:00 UTC. Re-read first-hand from api-docs.deepseek.com/quick_start/pricing/ on 2026-08-16, when the page still stated the cutover as forthcoming: the fields were moved ~8 hours early, in the run of the cutover day, because this catalog is refreshed once daily at ~08:05 UTC and the alternative was to publish the superseded rates for ~16 hours after they stopped being true. THAT WINDOW IS NOW CLOSED: re-read first-hand 2026-08-17, the page presents the peak/off-peak table as the rates in force and the forward-dated announcement banner is gone; every figure above still matches it to the cent, so the fields are simply current from here on. Responses API supported (footnote (1)). The pricing page is served at the TRAILING-SLASH url (2026-08-11): the extensionless form returns a different document (\"Your First API Call\") with HTTP 200.",
      "source_url": "https://api-docs.deepseek.com/quick_start/pricing/",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "deepseek-v4-pro",
      "name": "DeepSeek-V4-Pro",
      "provider": "DeepSeek",
      "status": "preview",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "price_input_per_mtok": 1.32,
      "price_output_per_mtok": 3.96,
      "price_cached_input_per_mtok": 0.044,
      "knowledge_cutoff": null,
      "released": "2026-04-24",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "deepseek-v4-pro",
      "notes": "Larger/most-capable V4 model (~1.6T total / ~49B active params per HuggingFace model card + authoritative third-party reports; MIT License; mixed FP4/FP8). Context length 1M, max output 384K tokens. Input price $1.32/M cache-miss, $0.044/M cache-hit; output $3.96/M (USD). Supports three reasoning-effort modes (non-think / think high / think max), JSON output, tool calls; FIM completion non-thinking-mode only. Concurrency limit 500. Part of the 'DeepSeek V4 Preview' generation (released 2026-04-24), hence status=preview. Knowledge cutoff NOT officially published by DeepSeek -> left null. Prices re-verified to the cent against api-docs.deepseek.com/quick_start/pricing/ on 2026-08-09. VENDOR-DECLARED FORWARD PRICE EVENT, RESOLVED 2026-08-14 — the same footnote now carries a date and a table. Until 2026-08-13 it read only 'We plan to raise the overall pricing for DeepSeek API services in the near future, with a significant increase expected... subject to official notice', with no date and no figure. DeepSeek now states: 'DeepSeek API pricing will be updated to peak / off-peak billing, with off-peak rates at half the peak rates. Peak hours are 01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak). The new prices take effect at 16:00 UTC on August 16, 2026'. Announced rates for this model per 1M tokens: OFF-PEAK $0.022 cache-hit / $0.66 cache-miss / $1.98 out; PEAK $0.044 cache-hit / $1.32 cache-miss / $3.96 out. EXECUTED 2026-08-16: the priced fields above were re-pointed at the PEAK column on the cutover day, because DeepSeek presents peak as the rate and off-peak as half of it, not the other way round. The OFF-PEAK figure is exactly half on every axis and is what you actually pay for 17 of every 24 hours, since peak is only 01:00-04:00 and 06:00-10:00 UTC. Re-read first-hand from api-docs.deepseek.com/quick_start/pricing/ on 2026-08-16, when the page still stated the cutover as forthcoming: the fields were moved ~8 hours early, in the run of the cutover day, because this catalog is refreshed once daily at ~08:05 UTC and the alternative was to publish the superseded rates for ~16 hours after they stopped being true. THAT WINDOW IS NOW CLOSED: re-read first-hand 2026-08-17, the page presents the peak/off-peak table as the rates in force and the forward-dated announcement banner is gone; every figure above still matches it to the cent, so the fields are simply current from here on. Responses API SUPPORTED as of 2026-08-17: footnote (1) had committed to 'early August 2026' and the feature table still showed it unsupported on 2026-08-09; the table now marks Responses API and Anthropic-format API as supported for BOTH V4 models (re-read first-hand 2026-08-17). Model version string on that table: DeepSeek-V4-Pro-0813. The pricing page is served at the TRAILING-SLASH url (2026-08-11): the extensionless form returns a different document (\"Your First API Call\") with HTTP 200.",
      "source_url": "https://api-docs.deepseek.com/quick_start/pricing/",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "deepseek-chat",
      "name": "deepseek-chat (legacy alias)",
      "provider": "DeepSeek",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "price_input_per_mtok": 0.14,
      "price_output_per_mtok": 0.28,
      "price_cached_input_per_mtok": 0.0028,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-07-24",
      "replacement": "deepseek-v4-flash",
      "api_string": "deepseek-chat",
      "notes": "Legacy model-name alias, NOT a separate model. Official docs: 'deepseek-chat & deepseek-reasoner will be fully retired and inaccessible after Jul 24th, 2026, 15:59 (UTC Time). (Currently routing to deepseek-v4-flash non-thinking/thinking).' deepseek-chat = NON-thinking mode of deepseek-v4-flash. Exact retirement timestamp: 2026-07-24 15:59 UTC. Replacement: use explicit name deepseek-v4-flash (non-thinking). Pricing shown is the rate that was in force when this alias was retired, and it deliberately no longer tracks deepseek-v4-flash: DeepSeek moved the live API to peak/off-peak billing at 16:00 UTC on 2026-08-16, roughly three weeks AFTER this alias became inaccessible, so the new schedule never applied to it. Until 2026-08-16 this note read 'Pricing shown matches deepseek-v4-flash since it routes there', which stopped being true the moment the target row was re-pointed. This alias historically mapped to DeepSeek-V3-series non-thinking; as of the V4 Preview it routes to V4-Flash.",
      "source_url": "https://api-docs.deepseek.com/news/news260424",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "deepseek-reasoner",
      "name": "deepseek-reasoner (legacy alias)",
      "provider": "DeepSeek",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 384000,
      "price_input_per_mtok": 0.14,
      "price_output_per_mtok": 0.28,
      "price_cached_input_per_mtok": 0.0028,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-07-24",
      "replacement": "deepseek-v4-flash",
      "api_string": "deepseek-reasoner",
      "notes": "Legacy model-name alias, NOT a separate model. Official docs: 'deepseek-chat & deepseek-reasoner will be fully retired and inaccessible after Jul 24th, 2026, 15:59 (UTC Time). (Currently routing to deepseek-v4-flash non-thinking/thinking).' deepseek-reasoner = THINKING mode of deepseek-v4-flash. Exact retirement timestamp: 2026-07-24 15:59 UTC. Replacement: use explicit name deepseek-v4-flash (thinking mode). Pricing shown is the rate that was in force when this alias was retired, and it deliberately no longer tracks deepseek-v4-flash: DeepSeek moved the live API to peak/off-peak billing at 16:00 UTC on 2026-08-16, roughly three weeks AFTER this alias became inaccessible, so the new schedule never applied to it. Until 2026-08-16 this note read 'Pricing shown matches deepseek-v4-flash since it routes there', which stopped being true the moment the target row was re-pointed. This alias historically mapped to DeepSeek-R1 / V3-series thinking; as of the V4 Preview it routes to V4-Flash thinking.",
      "source_url": "https://api-docs.deepseek.com/news/news260424",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-a-plus-05-2026",
      "name": "Command A Plus",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-plus-05-2026",
      "notes": "Cohere's first Mixture-of-Experts model; combines vision input, agentic/reasoning, and world-class translation. Endpoint: Chat. Per-token pricing NOT publicly listed (premium Command A variant; gated behind sales@cohere.com per multiple 2026 trackers), so input/output prices set null rather than guessed. Context 128k / max output 64k confirmed from docs.cohere.com/docs/models.md. Knowledge cutoff not published by Cohere.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-a-03-2025",
      "name": "Command A",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 256000,
      "max_output_tokens": 8000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-03",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-03-2025",
      "notes": "Flagship general model (111B params), 23 languages, tool use/RAG/agents. Context 256k / max output 8k from docs.cohere.com/docs/models.md. No public per-token price: as of 2026-07-26 Cohere publishes no rate for Command A on cohere.com/pricing (no card in the API pricing table, no row in the Model Vault instance table), its docs delegate all figures to that page, and Cohere is absent from the AWS Bedrock Price List API — so this is null rather than the $2.50 in / $10.00 out that aggregators report. Weights are openly released; commercial access appears to be Model Vault / sales-gated.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-a-reasoning-08-2025",
      "name": "Command A Reasoning",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 256000,
      "max_output_tokens": 32000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-reasoning-08-2025",
      "notes": "Cohere's first reasoning model ('thinks' before generating). Context 256k / max output 32k from docs.cohere.com/docs/models.md. Per-token pricing not publicly listed (sales-gated per eesel.ai/pricepertoken 2026 notes) -> null. Knowledge cutoff not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-a-vision-07-2025",
      "name": "Command A Vision",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": 8000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-vision-07-2025",
      "notes": "First Cohere model capable of processing images (enterprise image analysis). Context 128k / max output 8k from docs.cohere.com/docs/models.md. Per-token pricing not publicly listed (sales-gated) -> null. Knowledge cutoff not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-a-translate-08-2025",
      "name": "Command A Translate",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8000,
      "max_output_tokens": 8000,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-a-translate-08-2025",
      "notes": "State-of-the-art machine translation model covering 23 languages. Context 8k / max output 8k from docs.cohere.com/docs/models.md. Per-token pricing not publicly listed (sales-gated) -> null. Knowledge cutoff not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-r7b-12-2024",
      "name": "Command R7B",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 0.0375,
      "price_output_per_mtok": 0.15,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-12",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-r7b-12-2024",
      "notes": "Smallest/fastest model in R series; strong RAG & tool use. Price $0.0375 in / $0.15 out per 1M tokens read first-party from the \"Command R7B\" card on cohere.com/pricing (Sanity payload: inputPrice 0.0375 / outputPrice 0.15, per \"1M tokens\"), 2026-07-26 — previously aggregator-corroborated, now confirmed at source with the same values. Context 128k / max output 4k from docs.cohere.com/docs/models.md. Knowledge cutoff not published.",
      "source_url": "https://cohere.com/pricing",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-r-plus-08-2024",
      "name": "Command R+ (08-2024)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-r-plus-08-2024",
      "notes": "Still listed as available. Price $2.50 in / $10.00 out per 1M tokens confirmed first-party 2026-07-26 from the cohere.com/pricing FAQ (\"Command R+ 08-2024 pricing is $2.50/1M tokens for input and $10.00/1M tokens for output\"). No card in the current API pricing table — the FAQ is the published surface. Context 128k / max output 4k from docs.cohere.com/docs/models.md. Knowledge cutoff not published.",
      "source_url": "https://cohere.com/pricing",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-r-08-2024",
      "name": "Command R (08-2024)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 0.15,
      "price_output_per_mtok": 0.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-08",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "command-r-08-2024",
      "notes": "Still listed as available. Price $0.15 in / $0.60 out per 1M tokens read first-party from the \"Command R\" card on cohere.com/pricing (Sanity payload: inputPrice 0.15 / outputPrice 0.6, per \"1M tokens\"), 2026-07-26 — previously aggregator-corroborated, now confirmed at source with the same values. Context 128k / max output 4k from docs.cohere.com/docs/models.md. Knowledge cutoff not published.",
      "source_url": "https://cohere.com/pricing",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-r-plus-04-2024",
      "name": "Command R+ (04-2024)",
      "provider": "Cohere",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-04",
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "command-r-plus-08-2024",
      "api_string": "command-r-plus-04-2024",
      "notes": "Deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md (alias 'command-r-plus'). Replacement: command-r-plus-08-2024 or command-a-03-2025. Legacy price $3.00 in / $15.00 out per 1M confirmed first-party 2026-07-26 from the cohere.com/pricing FAQ (\"Command R+ 04-2024 pricing is $3.00/1M tokens for input and $15.00/1M tokens for output\"). Effectively retired.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-r-03-2024",
      "name": "Command R (03-2024)",
      "provider": "Cohere",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-03",
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "command-r-08-2024",
      "api_string": "command-r-03-2024",
      "notes": "Deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md (alias 'command-r'). Replacement: command-r-08-2024 or command-a-03-2025. Legacy price $0.50 in / $1.50 out per 1M confirmed first-party 2026-07-26 from the cohere.com/pricing FAQ (\"Command R 03-2024 pricing is $0.50/1M tokens for input and $1.50/1M tokens for output\"). Associated fine-tuned models were deprecated/shut down earlier on 2025-03-08.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command",
      "name": "Command (legacy)",
      "provider": "Cohere",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 4096,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "command-r-08-2024",
      "api_string": "command",
      "notes": "Original Command generative model. Deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md. Replacement: command-r-08-2024. Legacy price $1.00 in / $2.00 out per 1M confirmed first-party 2026-07-26 from the cohere.com/pricing FAQ (\"Command pricing is $1.00/1M tokens for input and $2.00/1M tokens for output\"). Context window not specified on current docs; 4096 is the historical value (low confidence) - treat as approximate.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "command-light",
      "name": "Command Light (legacy)",
      "provider": "Cohere",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 4096,
      "max_output_tokens": 4000,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 0.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "command-r-08-2024",
      "api_string": "command-light",
      "notes": "Smaller/faster legacy Command. Deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md. Replacement: command-r-08-2024. Legacy price $0.30 in / $0.60 out per 1M confirmed first-party 2026-07-26 from the cohere.com/pricing FAQ (\"Command-light pricing is $0.30/1M tokens for input and $0.60/1M tokens for output\"). Context window historical (~4096), low confidence.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "embed-v4-0",
      "name": "Embed 4 (embed-v4.0)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.12,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-v4.0",
      "notes": "Multimodal embedding model: text, images, PDFs. 128k context; output dimensions configurable 256-1536. Embedding model so no output-token price. Price $0.12 per 1M text input tokens AND $0.47 per 1M image tokens both read first-party from the \"Embed 4\" card on cohere.com/pricing (Sanity payload: inputPrice 0.12 label \"Cost\" / outputPrice 0.47 label \"Image cost\", per \"1M tokens\"), 2026-07-26 — the image rate was previously aggregator-sourced. Cohere also publishes Model Vault dedicated-instance rates for this model: Small $4.00/hr ($2,500/mo), Medium $5.00/hr ($3,250/mo). Knowledge cutoff/release date not published.",
      "source_url": "https://cohere.com/pricing",
      "open_weight": false,
      "embedding_dimensions": 1536,
      "price_per_1k_searches": null
    },
    {
      "id": "embed-english-v3-0",
      "name": "Embed English v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 512,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-english-v3.0",
      "notes": "English embedding model, 1024 dims, 512-token context. Endpoints: Embed, Embed Jobs. No public price: as of 2026-07-26 cohere.com/pricing lists only Embed 4 among embedding models, so the v3 family has no rate on any first-party Cohere surface and Cohere is absent from the AWS Bedrock Price List API — null rather than the ~$0.10 per 1M aggregators report (a figure our own notes previously flagged as \"moderate confidence\"). Knowledge cutoff/release date not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": 1024,
      "price_per_1k_searches": null
    },
    {
      "id": "embed-english-light-v3-0",
      "name": "Embed English Light v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 512,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-english-light-v3.0",
      "notes": "Smaller/faster English embedding model, 384 dims, 512-token context. No public price: as of 2026-07-26 cohere.com/pricing lists only Embed 4 among embedding models, so the v3 family has no rate on any first-party Cohere surface and Cohere is absent from the AWS Bedrock Price List API — null rather than the ~$0.10 per 1M aggregators assume for the family. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": 384,
      "price_per_1k_searches": null
    },
    {
      "id": "embed-multilingual-v3-0",
      "name": "Embed Multilingual v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 512,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-multilingual-v3.0",
      "notes": "Multilingual embedding model, 1024 dims, 512-token context. No public price: as of 2026-07-26 cohere.com/pricing lists only Embed 4 among embedding models, so the v3 family has no rate on any first-party Cohere surface and Cohere is absent from the AWS Bedrock Price List API — null rather than the ~$0.10 per 1M aggregators assume for the family. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": 1024,
      "price_per_1k_searches": null
    },
    {
      "id": "embed-multilingual-light-v3-0",
      "name": "Embed Multilingual Light v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 512,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "embed-multilingual-light-v3.0",
      "notes": "Smaller/faster multilingual embedding model, 384 dims, 512-token context. No public price: as of 2026-07-26 cohere.com/pricing lists only Embed 4 among embedding models, so the v3 family has no rate on any first-party Cohere surface and Cohere is absent from the AWS Bedrock Price List API — null rather than the ~$0.10 per 1M aggregators assume for the family. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": 384,
      "price_per_1k_searches": null
    },
    {
      "id": "rerank-v4-0-pro",
      "name": "Rerank 4 Pro (rerank-v4.0-pro)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 32000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-v4.0-pro",
      "notes": "Rerank 4 Pro — AI search foundation model for relevance in search/RAG. 32,768 context window, multilingual across 100+ languages, no data pre-processing required, state-of-the-art performance with low latency. Price $2.50 per 1,000 search units read first-party from the \"Rerank 4 Pro\" card on cohere.com/pricing (Sanity payload: inputPrice 2.5, label \"Cost\", overridePer \"1K searches\"), 2026-07-26. A search unit is one query with up to 100 documents. Cohere also publishes Model Vault dedicated-instance rates: Medium $5.00/hr ($3,250/mo), Large $10.00/hr ($6,500/mo). Knowledge cutoff/release date not published.",
      "source_url": "https://cohere.com/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": 2.5
    },
    {
      "id": "rerank-v4-0-fast",
      "name": "Rerank 4 Fast (rerank-v4.0-fast)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 32000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-v4.0-fast",
      "notes": "Rerank 4 Fast — AI search foundation model for relevance in search/RAG. 32,768 context window, multilingual across 100+ languages, no data pre-processing required, high performance / lowest latency. Price $2.00 per 1,000 search units read first-party from the \"Rerank 4 Fast\" card on cohere.com/pricing (Sanity payload: inputPrice 2, label \"Cost\", overridePer \"1K searches\"), 2026-07-26. A search unit is one query with up to 100 documents. Cohere also publishes a Model Vault dedicated-instance rate: Medium $5.00/hr ($3,250/mo). Knowledge cutoff/release date not published.",
      "source_url": "https://cohere.com/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": 2
    },
    {
      "id": "rerank-v3-5",
      "name": "Rerank 3.5 (rerank-v3.5)",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 4000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-v3.5",
      "notes": "4k context. Billed PER SEARCH UNIT (one query + up to 100 documents); per-token fields null. Cohere publishes no public per-search rate (checked 2026-07-25 across cohere.com/pricing and the docs pricing page) — price_per_1k_searches null; the earlier figure came from third-party aggregators and was removed. First-party Model Vault rate: Medium $5.00/hr or $3,250/mo. Also offered through Amazon Bedrock as cohere.rerank-v3-5:0 (the only reranker available in us-east-1). Replaced rerank-english/multilingual-v2.0 (shut down 2025-04-30). Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "rerank-english-v3-0",
      "name": "Rerank English v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 4000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-english-v3.0",
      "notes": "English-specific reranking, 4k context. Billed PER SEARCH UNIT (one query + up to 100 documents); per-token fields null. Cohere publishes no public per-search rate (checked 2026-07-25) — price_per_1k_searches null; the earlier figure came from third-party aggregators and was removed. Still listed in docs.cohere.com/docs/models. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "rerank-multilingual-v3-0",
      "name": "Rerank Multilingual v3.0",
      "provider": "Cohere",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 4000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "rerank-multilingual-v3.0",
      "notes": "Non-English reranking, 4k context. Billed PER SEARCH UNIT (one query + up to 100 documents); per-token fields null. Cohere publishes no public per-search rate (checked 2026-07-25) — price_per_1k_searches null; the earlier figure came from third-party aggregators and was removed. Still listed in docs.cohere.com/docs/models. Knowledge cutoff/release not published.",
      "source_url": "https://docs.cohere.com/docs/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "rerank-english-v2-0",
      "name": "Rerank English v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2024-12-02",
      "retires_on": "2025-04-30",
      "replacement": "rerank-v3.5",
      "api_string": "rerank-english-v2.0",
      "notes": "Deprecated 2024-12-02, shut down 2025-04-30 per docs.cohere.com/docs/deprecations.md. Replacement: rerank-v3.5. Retired - no longer callable.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "rerank-multilingual-v2-0",
      "name": "Rerank Multilingual v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2024-12-02",
      "retires_on": "2025-04-30",
      "replacement": "rerank-v3.5",
      "api_string": "rerank-multilingual-v2.0",
      "notes": "Deprecated 2024-12-02, shut down 2025-04-30 per docs.cohere.com/docs/deprecations.md. Replacement: rerank-v3.5. Retired - no longer callable.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "embed-english-v2-0",
      "name": "Embed English v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-04",
      "retires_on": "2026-04-04",
      "replacement": "embed-v4.0",
      "api_string": "embed-english-v2.0",
      "notes": "Deprecated AND shut down same day 2026-04-04 per docs.cohere.com/docs/deprecations.md. Replacement: embed-english-v3.0 or embed-v4.0. Retired.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "embed-english-light-v2-0",
      "name": "Embed English Light v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-04",
      "retires_on": "2026-04-04",
      "replacement": "embed-v4.0",
      "api_string": "embed-english-light-v2.0",
      "notes": "Deprecated AND shut down 2026-04-04 per docs.cohere.com/docs/deprecations.md. Replacement: embed-english-v3.0 or embed-v4.0. Retired.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "embed-multilingual-v2-0",
      "name": "Embed Multilingual v2.0",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-04-04",
      "retires_on": "2026-04-04",
      "replacement": "embed-v4.0",
      "api_string": "embed-multilingual-v2.0",
      "notes": "Deprecated AND shut down 2026-04-04 per docs.cohere.com/docs/deprecations.md. Replacement: embed-multilingual-v3.0 or embed-v4.0. Retired.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "summarize",
      "name": "Summarize (legacy endpoint)",
      "provider": "Cohere",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2025-09-15",
      "retires_on": "2025-09-15",
      "replacement": "Use the /chat endpoint",
      "api_string": "summarize",
      "notes": "Legacy Summarize endpoint deprecated AND shut down 2025-09-15 per docs.cohere.com/docs/deprecations.md. Replacement: use the Chat endpoint with a summarization prompt.",
      "source_url": "https://docs.cohere.com/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-nova-micro",
      "name": "Amazon Nova Micro",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": 5000,
      "price_input_per_mtok": 0.035,
      "price_output_per_mtok": 0.14,
      "price_cached_input_per_mtok": 0.00875,
      "knowledge_cutoff": "2024-10",
      "released": "2024-12-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "amazon.nova-2-lite-v1:0",
      "api_string": "amazon.nova-micro-v1:0",
      "notes": "Text-only model (fastest/cheapest Nova). Lifecycle: Active. Model card states 'Model EOL date: No sooner than 12/4/2025' (a floor, not a fixed retirement). Geo inference IDs us./eu.amazon.nova-micro-v1:0. Prices verified 2026-07-27 against the AWS Price List API (public, keyless): pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json, version 20260723233555 (2026-07-23) — us-east-1 on-demand SKUs USE1-NovaMicro-input-tokens $0.000035/1K, -output-tokens $0.00014/1K, -cache-read-input-token-count $0.00000875/1K. Cache WRITE is published at $0.00 (SKU -cache-write-input-token-count), correcting an earlier aggregator claim that it carries a premium. Batch tier is exactly 50% of on-demand.",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-micro.html",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-nova-lite",
      "name": "Amazon Nova Lite",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 300000,
      "max_output_tokens": 5000,
      "price_input_per_mtok": 0.06,
      "price_output_per_mtok": 0.24,
      "price_cached_input_per_mtok": 0.015,
      "knowledge_cutoff": "2024-10",
      "released": "2024-12-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "amazon.nova-2-lite-v1:0",
      "api_string": "amazon.nova-lite-v1:0",
      "notes": "Low-cost multimodal (text, image, video input; text output). Lifecycle: Active. Model card states 'Model EOL date: No sooner than 12/4/2025' (floor, not fixed). Geo inference IDs us./eu.amazon.nova-lite-v1:0. AWS recommends migrating to Nova 2 Lite. Prices verified 2026-07-27 against the AWS Price List API (public, keyless): pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json, version 20260723233555 (2026-07-23) — us-east-1 on-demand SKUs USE1-NovaLite-input-tokens $0.00006/1K, -output-tokens $0.00024/1K, -cache-read-input-token-count $0.000015/1K (published per-model, not derived from a ratio). Cache write $0.00. Batch tier exactly 50% of on-demand.",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-lite.html",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-nova-pro",
      "name": "Amazon Nova Pro",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 300000,
      "max_output_tokens": 5000,
      "price_input_per_mtok": 0.8,
      "price_output_per_mtok": 3.2,
      "price_cached_input_per_mtok": 0.2,
      "knowledge_cutoff": "2024-10",
      "released": "2024-12-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "amazon.nova-2-lite-v1:0",
      "api_string": "amazon.nova-pro-v1:0",
      "notes": "Balanced multimodal (text, image, video input; text output). Lifecycle: Active. Model card states 'Model EOL date: No sooner than 12/4/2025' (floor). Geo inference IDs us./eu.amazon.nova-pro-v1:0. Prices verified 2026-07-27 against the AWS Price List API (public, keyless): pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json, version 20260723233555 (2026-07-23) — us-east-1 on-demand SKUs USE1-NovaPro-input-tokens $0.0008/1K, -output-tokens $0.0032/1K, -cache-read-input-token-count $0.0002/1K. Cache WRITE is published at $0.00, correcting an earlier aggregator claim of a premium (they reported 1.00 per M against 0.80 per M input). Other service tiers (same source): Flex $0.40/$1.60 per M, Priority $1.40/$5.60 per M, Batch $0.40/$1.60 per M.",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-pro.html",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-nova-premier",
      "name": "Amazon Nova Premier",
      "provider": "Amazon",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 25000,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 12.5,
      "price_cached_input_per_mtok": 0.625,
      "knowledge_cutoff": "2024-10",
      "released": "2025-10-31",
      "deprecated_on": null,
      "retires_on": "2026-09-14",
      "replacement": "amazon.nova-2-lite-v1:0",
      "api_string": "amazon.nova-premier-v1:0",
      "notes": "Most capable Nova 1 model: complex reasoning, agentic workflows, model distillation. Reasoning supported. 1M-token context, 25K max output. Model card lifecycle = 'Legacy' with 'Model EOL date: September 14, 2026' (firm retirement). Status set to 'deprecated' (= Legacy) given the firm EOL. NOTE: model card shows 'Model launch date: Oct 31, 2025' which conflicts with the public GA of Nova Premier (Apr/May 2025); the Oct 31 2025 date may reflect a card/version update rather than original GA — flagged as uncertain. Only us. geo inference ID (us.amazon.nova-premier-v1:0). Per-1K: input $0.0025, ou",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-premier.html",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-nova-2-lite",
      "name": "Amazon Nova 2 Lite",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 64000,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": 0.075,
      "knowledge_cutoff": "2025-10",
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "amazon.nova-2-lite-v1:0",
      "notes": "Current-gen Nova 2 cost-efficient multimodal reasoning model (text, image, video input; text output). GA, Lifecycle: Active. Announced at re:Invent 2025. 1M context, 64K max output, knowledge cutoff Oct 2025. Extended thinking (low/medium/high), built-in code interpreter + web grounding, remote MCP tools. Model IDs: amazon.nova-2-lite-v1:0; geo us./eu./jp.; global.amazon.nova-2-lite-v1:0. Prices verified 2026-07-27 against the AWS Price List API (public, keyless): pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json, version 20260723233555 (2026-07-23). AWS publishes Nova 2 pricing ONLY under \"Global Cross-region Inference\", Standard tier: SKUs USE1-Nova2.0Lite-input-tokens-cross-region-global $0.0003/1K, -output-tokens-cross-region-global $0.0025/1K, -cache-read-...-cross-region-global $0.000075/1K (= 25% of input). The in-region SKU is 10% higher ($0.33/$2.75 per M) but is not the published headline rate. Cache write $0.00; Batch = 50%; Flex = 50%; Priority = 175%.",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-nova-2-lite.html",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-nova-2-pro",
      "name": "Amazon Nova 2 Pro",
      "provider": "Amazon",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.25,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": null,
      "notes": "Announced at re:Invent 2025 (Dec 2, 2025) as the 'most intelligent' Nova 2 model for complex multistep tasks. Still labelled 'Amazon Nova 2 Pro (Preview)' on the AWS Bedrock pricing page (Nova Forge / early access). Its model-card URL soft-404s (redirect stub to the Bedrock user-guide index, checked 2026-07-27), so model ID, max output and knowledge cutoff stay null rather than guessed. Context stated as 1M tokens (shared Nova 2 family spec). Prices ARE now published and were read 2026-07-27 from the AWS Price List API (public, keyless): pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json, version 20260723233555 (2026-07-23) — Global Cross-region Inference, Standard tier: USE1-Nova2.0Pro-text-input-tokens-cross-region-global $0.00125/1K, -text-output-tokens-cross-region-global $0.01/1K. Image, video and audio input bill at the same rate as text input. In-region SKUs are 10% higher ($1.375/$11 per M). No cache-read SKU is published (AWS shows N/A), so cached input stays null.",
      "source_url": "https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-nova-2-omni",
      "name": "Amazon Nova 2 Omni",
      "provider": "Amazon",
      "status": "preview",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 2.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": null,
      "notes": "The omni-modal Nova 2 variant, and the only Nova 2 row that GENERATES images. AWS lists it as 'Amazon Nova 2 Omni (Preview)' on the Bedrock pricing page, whose table headings state its input modalities as Text, Image, Video, Audio and give it both a text-output and an image-output token column. No specs are published anywhere first-party: the Bedrock model card (model-card-amazon-nova-2-omni.html) and the Nova user-guide omni pages all soft-404 (HTTP 200, 1,033-1,048 byte meta-refresh stub to the guide index, checked 2026-07-28), and the Dec 2025 Nova 2 launch announcement does not mention Omni at all. Context window, max output, model ID, knowledge cutoff and release date therefore stay null rather than guessed. Prices read first-party on 2026-07-28 from the AWS Price List API (public, keyless): pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json, version 20260723233555 (2026-07-23), 49 Omni SKUs. Headline = Global Cross-region Inference, Standard tier (the only tier AWS shows Nova 2 under): USE1-Nova2.0Omni-text-input-tokens-cross-region-global $0.0003/1K = $0.30/M; -text-output-tokens-cross-region-global $0.0025/1K = $2.50/M. Image and video input bill at the same $0.30/M as text; AUDIO input is priced separately at $1.00/M and IMAGE OUTPUT at $40.00/M. Unlike Nova 2 Lite/Pro, the us-east-1 in-region SKUs are not a flat markup: text input is identical ($0.30/M), text output is $2.80/M, audio input $1.10/M, image output $44.00/M. No cache-read SKU is published (0 cache SKUs across all 49 Omni usagetypes), so cached input stays null. Web grounding $0.03 per request. Batch and Flex tiers are roughly half of Standard and Priority is higher, per their own SKUs.",
      "source_url": "https://aws.amazon.com/bedrock/pricing/",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-titan-embed-text-v2",
      "name": "Amazon Titan Text Embeddings V2",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.02,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-04-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "amazon.titan-embed-text-v2:0",
      "notes": "Text embedding model (2nd-gen Titan). Output is an embedding vector, not tokens, so there is no output-token price. Input limit 8,192 tokens (~50,000 chars); output dimensions flexible 256 / 512 / 1,024 (default 1,024). Price is the us-east-1 on-demand rate from the AWS Price List API ($0.00002/1K input; batch $0.01/M). Model card lifecycle: Active; 'Model EOL date: No sooner than 4/30/2024' is a floor, not a fixed retirement, so lifecycle fields stay null.",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-titan-text-embeddings-v2.html",
      "open_weight": false,
      "embedding_dimensions": 1024,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-nova-2-multimodal-embeddings",
      "name": "Amazon Nova Multimodal Embeddings",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.135,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-10-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "amazon.nova-2-multimodal-embeddings-v1:0",
      "notes": "Unified multimodal embedding model (text, documents, images, video, audio in; vector out — no output-token price). Context up to 8K tokens; video/audio segments up to 30s with built-in chunking; sync + async (StartAsyncInvoke) APIs. Output dimensions 3,072 / 1,024 / 384 / 256 (docs examples default 3,072). The $0.135/M figure is the TEXT input rate (us-east-1 on-demand, AWS Price List API: $0.000135/1K); other inputs billed separately — standard image $0.00006/image, document image $0.0006/image, audio $0.00014/s, video $0.0007/s (batch = 50% of each). Model card lifecycle: Active, EOL N/A.",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-amazon-amazon-nova-multimodal-embeddings.html",
      "open_weight": false,
      "embedding_dimensions": 3072,
      "price_per_1k_searches": null
    },
    {
      "id": "amazon-rerank-v1",
      "name": "Amazon Rerank 1.0",
      "provider": "Amazon",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "amazon.rerank-v1:0",
      "notes": "Amazon's first-party reranker model in Bedrock: scores and reorders document chunks by relevance to a query, callable via the Rerank API operation or inside Bedrock Knowledge Bases. Billed PER SEARCH UNIT, not per token — AWS defines a search unit as one query containing up to 100 document chunks (docs: 'Reranking is priced per query. A query is a single call to the reranker model that can contain up to 100 document chunks.'), so all per-token price fields are null. Rate is the us-west-2 on-demand price from the AWS Price List API (SKU Z7M6S4MRBXNXJRB4: $0.001 per search unit = $1.00 per 1,000). Text only. Single-region model: ap-northeast-1, ca-central-1, eu-central-1, us-west-2 — explicitly NOT available in us-east-1, where Cohere Rerank 3.5 is the only reranker. No token context window is published (the limit is expressed in document chunks per query), and AWS declares no deprecation or EOL date, so those fields stay null. No release date published in the docs, so `released` is null rather than invented.",
      "source_url": "https://docs.aws.amazon.com/bedrock/latest/userguide/rerank-supported.html",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": 1
    },
    {
      "id": "qwen3-7-max",
      "name": "Qwen3.7-Max",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 7.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-05-21",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.7-max",
      "notes": "Flagship Max model, API-only/proprietary (no open weights). Official International (Singapore) list price $2.5 in / $7.5 out per 1M tokens, single tier 0<token<=1M, Non-Thinking and Thinking modes; alias currently = qwen3.7-max-2026-05-20 (the 2026-06-08 snapshot added visual-modal understanding -> text+image+video). Thinking enabled by default; supports explicit/context cache (caching priced as a discount, no separate cached-input column published) and Function Calling. Max output: official Alibaba blog/config shows 65536; a benchmark footnote on the same blog mentions max_tokens=80K (unconfi",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-7-max-preview",
      "name": "Qwen3.7-Max-Preview",
      "provider": "Alibaba",
      "status": "preview",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 2.5,
      "price_output_per_mtok": 7.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-05-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "qwen3.7-max",
      "api_string": "qwen3.7-max-preview",
      "notes": "Preview snapshot, alias = qwen3.7-max-2026-05-17. Text-only input, Thinking mode only (per official release notes 2026-05-25, International). Same International price as GA Max ($2.5/$7.5 per 1M). Superseded by the GA qwen3.7-max. Max output assumed same as GA (65536) - not separately confirmed.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-7-plus",
      "name": "Qwen3.7-Plus",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 1.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-06-01",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.7-plus",
      "notes": "Mid-tier vision-language Plus model, alias = qwen3.7-plus-2026-05-26. Official International TIERED pricing: tier1 0<token<=256K: $0.4 in / $1.6 out (same for Non-Thinking and Thinking output); tier2 256K<token<=1M: $1.2 in / $4.8 out. Multimodal (text/image/video input), full agent/coding/tool-use; supports context caching (discount). Context window 1,000,000 (tier extends to 1M). Max output 65536 inferred from Qwen3.6-Plus spec corroboration; not separately confirmed on official spec page (SPA-rendered). Released ~2026-06-01 (International) per official release notes. API-only/proprietary.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-6-flash",
      "name": "Qwen3.6-Flash",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.25,
      "price_output_per_mtok": 1.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-04-16",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.6-flash",
      "notes": "Most cost-effective current Flash model, alias = qwen3.6-flash-2026-04-16; native vision-language (multimodal text/image/video). Official International TIERED pricing: tier1 0<token<=256K: $0.25 in / $1.5 out; tier2 256K<token<=1M: $1 in / $4 out. Supports 50% batch-inference discount and context caching. Has an open-weight sibling listed: qwen3.6-35b-a3b (per official release notes, the qwen3.6-flash entry groups qwen3.6-flash / qwen3.6-flash-2026-04-16 / qwen3.6-35b-a3b). Context window 1M (tier to 1M). Max output not published on accessible official page (null). Released 2026-04-16 (Interna",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-max",
      "name": "Qwen3-Max",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": 32768,
      "price_input_per_mtok": 1.2,
      "price_output_per_mtok": 6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": "2025-06",
      "released": "2025-09-23",
      "deprecated_on": "2026-09-08",
      "retires_on": "2026-09-08",
      "replacement": "qwen3.7-max",
      "api_string": "qwen3-max",
      "notes": "Previous flagship Max, alias = qwen3-max-2026-01-23. Official International TIERED pricing: 0<token<=32K: $1.2 in / $6 out; 32K<token<=128K: $2.4 / $12; 128K<token<=256K: $3 / $15. Non-Thinking and Thinking modes. Context window 262,144 (256K) and max output 32,768 per OpenRouter (Alibaba spec table is SPA-rendered, not directly fetchable); knowledge cutoff Jun 2025 per OpenRouter. SCHEDULED DEPRECATION: the qwen3-max alias and qwen3-max-preview are listed for deprecation 2026-09-08 00:00:00 -> replacement qwen3.7-max; dated snapshots qwen3-max-2026-01-23 and qwen3-max-2025-09-23 are listed fo",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-max-preview",
      "name": "Qwen3-Max-Preview",
      "provider": "Alibaba",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": 32768,
      "price_input_per_mtok": 1.2,
      "price_output_per_mtok": 6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-05",
      "deprecated_on": "2026-09-08",
      "retires_on": "2026-09-08",
      "replacement": "qwen3.7-max",
      "api_string": "qwen3-max-preview",
      "notes": "Early preview of Qwen3-Max (trillion-parameter MoE). Listed in official deprecation table for retirement 2026-09-08 00:00:00 -> replacement qwen3.7-max. Same International tiered pricing as qwen3-max ($1.2/$6 at <=32K etc.). API-only/proprietary. Release date approximate (Sep 2025); context/output mirror qwen3-max (256K / 32K).",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-6-max-preview",
      "name": "Qwen3.6-Max-Preview",
      "provider": "Alibaba",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 1.3,
      "price_output_per_mtok": 7.8,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-09-08",
      "retires_on": "2026-09-08",
      "replacement": "qwen3.7-max",
      "api_string": "qwen3.6-max-preview",
      "notes": "Preview Max snapshot. Official International TIERED pricing: 0<token<=128K: $1.3 in / $7.8 out; 128K<token<=256K: $2 / $12. Listed in official deprecation table for retirement 2026-09-08 00:00:00 -> replacement qwen3.7-max. Context 262,144 / max output 65,536 per web corroboration (search result citing Qwen3.6-Max-Preview 262144 ctx, 65536 max output); not on accessible official spec page. API-only/proprietary.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen-max",
      "name": "Qwen-Max (Qwen2.5-Max)",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 32768,
      "max_output_tokens": 8192,
      "price_input_per_mtok": 1.6,
      "price_output_per_mtok": 6.4,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-01-27",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen-max",
      "notes": "Legacy stable Max alias = qwen-max-2025-01-25 (a.k.a. Qwen2.5-Max). Official International price: $1.6 in / $6.4 out per 1M, no tiered pricing, Non-Thinking mode only; supports 50% batch-inference discount. Context window ~32,768 (33K) and max output ~8,192 per third-party corroboration (CloudPrice/DataStudios); not on accessible official spec page. Still listed/purchasable on pricing page (Jun 22, 2026); not in current deprecation table. Released 2025-01-27 (International). API-only/proprietary.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-7-plus-2026-05-26",
      "name": "Qwen3.7-Plus (snapshot 2026-05-26)",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 1.6,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-06-01",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.7-plus-2026-05-26",
      "notes": "Dated snapshot currently pinned by the qwen3.7-plus alias. Same tiered International pricing as the alias: <=256K $0.4/$1.6, 256K-1M $1.2/$4.8. Included so trackers can pin a stable version string. Multimodal.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-5-plus",
      "name": "Qwen3.5-Plus",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 2.4,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-02-15",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.5-plus",
      "notes": "Prior-gen native vision-language Plus, alias = qwen3.5-plus-2026-02-15 (latest dated snapshot qwen3.5-plus-2026-04-20). Official International price: 0<token<=256K $0.4 in / $2.4 out per 1M (single tier listed). Still listed on pricing page Jun 22, 2026; not in deprecation table. Max output not on accessible official page (null). Released 2026-02-15 (International).",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-6-plus",
      "name": "Qwen3.6-Plus",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 1000000,
      "max_output_tokens": 65536,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-04-02",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.6-plus",
      "notes": "Prior-gen Plus, alias = qwen3.6-plus-2026-04-02. Official International price: 0<token<=256K $0.5 in / $3 out per 1M (single tier listed). Context window 1,000,000 and max output 65,536 per web corroboration (multiple sources cite Qwen3.6-Plus 1M ctx / 65536 max). Still listed on pricing page Jun 22, 2026. Multimodal.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-5-flash",
      "name": "Qwen3.5-Flash",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-02-23",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.5-flash",
      "notes": "Prior-gen Flash, alias = qwen3.5-flash-2026-02-23. Official International price: 0<token<=1M $0.1 in / $0.4 out per 1M (single tier, context to 1M). Supports 50% batch-inference discount and context caching. Still listed on pricing page Jun 22, 2026. Max output not on accessible official page (null).",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen-plus",
      "name": "Qwen-Plus (Qwen3-series)",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": 32768,
      "price_input_per_mtok": 0.4,
      "price_output_per_mtok": 1.2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-30",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen-plus",
      "notes": "Legacy stable Plus alias = qwen-plus-2025-12-01 (Qwen3 series). Official International TIERED pricing: 0<token<=256K $0.4 in; output $1.2 (Non-Thinking) / $4.0 (Thinking, chain-of-thought+answer) per 1M. Context length increased to 1,000,000 per official release note (qwen-plus-2025-07-28). Max output ~32K (typical for Qwen3 Plus; not on accessible official spec page). Older dated snapshots qwen-plus-2024-11-27/-11-25/-09-19/-08-06 were deprecated 2026-01-30 -> replacement qwen-plus-2025-12-01. Current alias still GA. Text-only.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen-flash",
      "name": "Qwen-Flash",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.05,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen-flash",
      "notes": "Legacy stable Flash alias = qwen-flash-2025-07-28; the recommended replacement for the discontinued Qwen-Turbo. Official International TIERED pricing: 0<token<=256K $0.05 in / $0.4 out; 256K<token<=1M $0.25 in / $2 out per 1M. Supports 50% batch-inference discount and context caching. Context window 1M. Max output not on accessible official page (null). Still GA on pricing page Jun 22, 2026. Text-only.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen-turbo",
      "name": "Qwen-Turbo",
      "provider": "Alibaba",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 1000000,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.05,
      "price_output_per_mtok": 0.2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-04-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "qwen-flash",
      "api_string": "qwen-turbo",
      "notes": "DEPRECATED (no end-of-service date published, still callable). Official pricing page (Jun 22, 2026) states verbatim: 'Qwen-Turbo will no longer be updated. We recommend switching to Qwen-Flash.' Alias = qwen-turbo-2025-04-28. Official International price: $0.05 in; output $0.2 (Non-Thinking) / $0.5 (Thinking) per 1M. Older snapshot qwen-turbo-2024-09-19 listed in deprecation table -> replacement qwen-flash-2025-07-28. Marked status=deprecated because it is no longer updated; deprecated_on/retires_on left null because no formal retirement date is published for the qwen-turbo alias. Text-only.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-5-omni-plus",
      "name": "Qwen3.5-Omni-Plus",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "audio",
        "video"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3.5-omni-plus",
      "notes": "Current Omni (any-to-any) model in the lineup, listed on the official Models overview (Jun 22, 2026) for image/video understanding, speech-to-speech, and omni use cases; a realtime variant qwen3.5-omni-plus-realtime also exists. Pricing not captured here (omni/audio pricing is on separate per-modality tables of the pricing page billed per token AND per second/character for audio; not a simple per-1M-token text rate). Context/output/release not on the accessible (server-rendered) pages -> null. Multimodal (text/image/audio/video in, text+audio out).",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/models",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-rerank",
      "name": "Qwen3-Rerank",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "qwen3-rerank",
      "notes": "Current reranking model; listed as the replacement for the deprecated gte-rerank (gte-rerank deprecation 2026-05-30 per official deprecation table). Rerank models are not priced per input/output token in the standard text table; pricing null here. Included because it is the current GA rerank model in the lineup.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen-text-embedding-v4",
      "name": "Qwen Text-Embedding v4",
      "provider": "Alibaba",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.07,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "text-embedding-v4",
      "notes": "Alibaba Model Studio (DashScope) commercial text/code embedding model, based on the Qwen3-Embedding family. Output is an embedding vector, not tokens, so there is no output-token price. Max input 8,192 tokens per text; batch up to 10 texts per call. Flexible (MRL) output dimensions: 2048 / 1536 / 1024 (default) / 768 / 512 / 256 / 128 / 64. Supports 100+ languages incl. programming languages, plus task instructions and sparse vectors. The $0.07/1M figure is the Singapore/Hong Kong international-region rate; Beijing region is $0.072/1M. The open-weight Qwen3-Embedding checkpoints (0.6B/4B/8B, Apache-2.0 on Hugging Face) are separate downloadable models, not this hosted API; open_weight here reflects the commercial API model. Docs fetched 2026-07-24.",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/embedding",
      "open_weight": false,
      "embedding_dimensions": 1024,
      "price_per_1k_searches": null
    },
    {
      "id": "qwen3-235b-a22b",
      "name": "Qwen3-235B-A22B (open weights)",
      "provider": "Alibaba",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-04-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "qwen3.7-plus",
      "api_string": "qwen3-235b-a22b",
      "notes": "Representative OPEN-WEIGHT flagship from the Qwen3 family (released Apr 2025, Apache-2.0 license, weights on Hugging Face/ModelScope). On Model Studio its HOSTED endpoint is in the deprecation table (Qwen3 open source edition group; deprecation date July 8, 2026 shared with that table) -> replacement qwen3.7-plus; the open weights themselves remain downloadable regardless of the hosted-API deprecation. Included to flag open-weight status; many sibling open-weight variants exist (qwen3-8b/14b/32b, qwen3-30b-a3b, qwen3-235b-a22b-instruct-2507/-thinking-2507, qwen3-next-80b-a3b, qwen3-vl-* , qwen",
      "source_url": "https://www.alibabacloud.com/help/en/model-studio/model-depreciation",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "perplexity-sonar",
      "name": "Sonar",
      "provider": "Perplexity",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 1,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-01",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "sonar",
      "notes": "Lightweight, cost-effective real-time web search model with grounding/citations. 128K context. Token pricing $1/$1 per 1M in/out. Additional per-request search fee NOT included in token price: $5/$8/$12 per 1,000 requests for low/medium/high search context size (pricing page). Supports image uploads (added Apr 2025) and file attachments PDF/DOC/DOCX/TXT/RTF (added Sep 2025). Max output tokens not published by Perplexity. Knowledge cutoff not published (search-grounded model with live web access). Pricing page fetched 2026-06-28.",
      "source_url": "https://docs.perplexity.ai/docs/sonar/models/sonar",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "perplexity-sonar-pro",
      "name": "Sonar Pro",
      "provider": "Perplexity",
      "status": "ga",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 200000,
      "max_output_tokens": null,
      "price_input_per_mtok": 3,
      "price_output_per_mtok": 15,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-01",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "sonar-pro",
      "notes": "Advanced search model for complex queries and follow-ups; ~2x more search results than Sonar; non-reasoning. 200K context. Token pricing $3/$15 per 1M in/out. Additional per-request search fee NOT in token price: $6/$10/$14 per 1,000 requests for low/medium/high search context size (pricing page). Supports image input and the Dec 2025 media classifier (auto image/video selection). Max output tokens not published. Knowledge cutoff not published (search-grounded). Pricing page fetched 2026-06-28.",
      "source_url": "https://docs.perplexity.ai/docs/sonar/models/sonar-pro",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "perplexity-sonar-reasoning-pro",
      "name": "Sonar Reasoning Pro",
      "provider": "Perplexity",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "sonar-reasoning-pro",
      "notes": "Precise reasoning model with Chain-of-Thought; emits a <think> reasoning section before the answer. 128K context. Token pricing $2/$8 per 1M in/out. Per-request search fee NOT in token price: $6/$10/$14 per 1,000 requests for low/medium/high search context (pricing page). Successor to the now-removed 'sonar-reasoning'. Image input with structured outputs not supported in thinking models. Max output tokens and knowledge cutoff not published. Exact release date not published in docs. Pricing page fetched 2026-06-28.",
      "source_url": "https://docs.perplexity.ai/docs/sonar/models/sonar-reasoning-pro",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "perplexity-sonar-deep-research",
      "name": "Sonar Deep Research",
      "provider": "Perplexity",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "sonar-deep-research",
      "notes": "Expert-level research model that runs exhaustive multi-source searches and generates comprehensive reports. 128K context. Token pricing: input $2/M, output $8/M, citation tokens $2/M, reasoning tokens $3/M, plus search queries $5 per 1,000 (pricing page). Featured May 2025 with reasoning-effort parameter and async API. Max output tokens and knowledge cutoff not published. Pricing page fetched 2026-06-28.",
      "source_url": "https://docs.perplexity.ai/docs/sonar/models/sonar-deep-research",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "perplexity-sonar-reasoning",
      "name": "Sonar Reasoning",
      "provider": "Perplexity",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 128000,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-01",
      "deprecated_on": "2025-12-15",
      "retires_on": "2025-12-15",
      "replacement": "sonar-reasoning-pro",
      "api_string": "sonar-reasoning",
      "notes": "DEPRECATED and removed from the API on 2025-12-15 per the official changelog; calls should migrate to sonar-reasoning-pro (enhanced multi-step reasoning with web search). The current models page (docs.perplexity.ai/getting-started/models) no longer lists it. Historical context window 128K. Historical token pricing was approx $1 input / $5 output per 1M while active, but this is NOT confirmed on a current official page (pricing page no longer lists it), so left null rather than guessed. Confirmed via changelog and the sonar-reasoning-pro model card (which names it as the deprecated predecessor)",
      "source_url": "https://docs.perplexity.ai/docs/resources/changelog",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-k2-7-code",
      "name": "Kimi K2.7 Code",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.95,
      "price_output_per_mtok": 4,
      "price_cached_input_per_mtok": 0.19,
      "knowledge_cutoff": null,
      "released": "2026-06-12",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "kimi-k2.7-code",
      "notes": "Open-weight 1T-parameter MoE (~32B active), coding-focused. Cache-miss input $0.95/M, cache-hit input $0.19/M, output $4.00/M (official pricing page, fetched 2026-06-28). Listed as a current model on platform.kimi.ai/docs/models. Release date 2026-06-12 from multiple secondary sources (MarkTechPost, digitalapplied, noqta) - the official pricing/model page does not publish release date or knowledge cutoff. Max output tokens not published. Modality text/code only (not multimodal).",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-k27-code.md",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-k2-7-code-highspeed",
      "name": "Kimi K2.7 Code HighSpeed",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 1.9,
      "price_output_per_mtok": 8,
      "price_cached_input_per_mtok": 0.38,
      "knowledge_cutoff": null,
      "released": "2026-06-12",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "kimi-k2.7-code-highspeed",
      "notes": "High-throughput variant of K2.7 Code: ~180 tokens/s (up to ~260 tokens/s short context). Cache-miss input $1.90/M, cache-hit input $0.38/M, output $8.00/M - i.e. 2x the standard K2.7 Code price for higher speed (official pricing page, fetched 2026-06-28). Max output tokens and knowledge cutoff not published officially.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-k27-code.md",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-k2-6",
      "name": "Kimi K2.6",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.95,
      "price_output_per_mtok": 4,
      "price_cached_input_per_mtok": 0.16,
      "knowledge_cutoff": null,
      "released": "2026-04-20",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "kimi-k2.6",
      "notes": "Moonshot's flagship/most intelligent model. Native multimodal (text, image, video input), supports thinking and non-thinking modes. Cache-miss input $0.95/M, cache-hit input $0.16/M, output $4.00/M (official pricing page, fetched 2026-06-28). Designated replacement for all deprecated kimi-k2 series and retired kimi-latest/kimi-thinking-preview models. Release date 2026-04-20 from secondary sources (Yicai Global, kimi-k2.org, miraflow) - official page does not list release date or knowledge cutoff. Max output tokens not published.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-k26.md",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-k2-5",
      "name": "Kimi K2.5",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.6,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": 0.1,
      "knowledge_cutoff": null,
      "released": "2026-01-27",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "kimi-k2.5",
      "notes": "Native multimodal model (text, image, video input), 1T-param MoE (~32B active). Cache-miss input $0.60/M, cache-hit input $0.10/M, output $3.00/M (official pricing page, fetched 2026-06-28). Still listed as a current/available model on platform.kimi.ai/docs/models (not deprecated). Release date 2026-01-27 from multiple secondary sources (Baidu Baike, ComfyUI Wiki, AWS Bedrock card, kimi.com). Official page does not list knowledge cutoff or max output tokens.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-k25.md",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "moonshot-v1-8k",
      "name": "Moonshot v1 8K",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-8k",
      "notes": "Legacy text model, still listed as current. Input $0.20/M, output $2.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified for the v1 family on the official page. Release date and knowledge cutoff not published.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "moonshot-v1-32k",
      "name": "Moonshot v1 32K",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 32768,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-32k",
      "notes": "Legacy text model. Input $1.00/M, output $3.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified. Release date and knowledge cutoff not published.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "moonshot-v1-128k",
      "name": "Moonshot v1 128K",
      "provider": "Moonshot",
      "status": "ga",
      "modality": [
        "text"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-128k",
      "notes": "Legacy text model. Input $2.00/M, output $5.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified. Release date and knowledge cutoff not published.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "moonshot-v1-8k-vision-preview",
      "name": "Moonshot v1 8K Vision (Preview)",
      "provider": "Moonshot",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 8192,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.2,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-8k-vision-preview",
      "notes": "Vision-capable preview variant (text + image). Same price as moonshot-v1-8k: input $0.20/M, output $2.00/M (official pricing page, fetched 2026-06-28). 'preview' in the API id. Cache-hit pricing not specified.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "moonshot-v1-32k-vision-preview",
      "name": "Moonshot v1 32K Vision (Preview)",
      "provider": "Moonshot",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 32768,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 3,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-32k-vision-preview",
      "notes": "Vision-capable preview variant (text + image). Same price as moonshot-v1-32k: input $1.00/M, output $3.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "moonshot-v1-128k-vision-preview",
      "name": "Moonshot v1 128K Vision (Preview)",
      "provider": "Moonshot",
      "status": "preview",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "moonshot-v1-128k-vision-preview",
      "notes": "Vision-capable preview variant (text + image). Same price as moonshot-v1-128k: input $2.00/M, output $5.00/M (official pricing page, fetched 2026-06-28). Cache-hit pricing not specified.",
      "source_url": "https://platform.kimi.ai/docs/pricing/chat-v1.md",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-k2-0905-preview",
      "name": "Kimi K2 (0905 Preview)",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-0905-preview",
      "notes": "Deprecated/officially discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Pricing no longer published on current official pages. Historical legacy K2 list price was reported around $0.60/M input, $2.50/M output by secondary trackers, but not confirmable on current official pages, so prices left null. Context window 256K (262144) per K2 family docs.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-k2-0711-preview",
      "name": "Kimi K2 (0711 Preview)",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-11",
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-0711-preview",
      "notes": "The original Kimi K2 (0711) release, deprecated/discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Release implied by '0711' (2025-07-11) and corroborated by Moonshot's original K2 launch. Pricing no longer on official pages so left null. Original K2 context window was 128K (131072) per Moonshot's original K2 docs/GitHub.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-k2-turbo-preview",
      "name": "Kimi K2 Turbo (Preview)",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-turbo-preview",
      "notes": "High-speed turbo variant of K2, deprecated/discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Pricing removed from official pages, left null. Context 256K per K2 family docs (unverified for this exact variant on current pages).",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-k2-thinking",
      "name": "Kimi K2 Thinking",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-thinking",
      "notes": "Reasoning/thinking variant of K2, deprecated/discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Pricing removed from official pages, left null.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-k2-thinking-turbo",
      "name": "Kimi K2 Thinking Turbo",
      "provider": "Moonshot",
      "status": "deprecated",
      "modality": [
        "text"
      ],
      "context_window": 262144,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": "2026-05-25",
      "retires_on": "2026-05-25",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-k2-thinking-turbo",
      "notes": "High-speed thinking variant of K2, deprecated/discontinued 2026-05-25 per official model list; replacement kimi-k2.6. Pricing removed from official pages, left null.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": true,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-latest",
      "name": "Kimi Latest",
      "provider": "Moonshot",
      "status": "retired",
      "modality": [
        "text",
        "image"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2026-01-28",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-latest",
      "notes": "Retired 2026-01-28 per official model list; replacement now kimi-k2.6. This was the rolling 'latest' alias (auto-updated to the newest Kimi chat model, historically vision-capable). Pricing not on current official pages, left null. Context window historically 128K (131072) - not re-verifiable on current pages.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": null,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "kimi-thinking-preview",
      "name": "Kimi Thinking (Preview)",
      "provider": "Moonshot",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": 131072,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2025-11-11",
      "replacement": "kimi-k2.6",
      "api_string": "kimi-thinking-preview",
      "notes": "Earliest reasoning preview model, retired 2025-11-11 per official model list; replacement now kimi-k2.6. Pricing not on current official pages, left null. Context window historically 128K - not re-verifiable on current pages.",
      "source_url": "https://platform.kimi.ai/docs/models.md",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-1-flash-image",
      "name": "Gemini 3.1 Flash Image (Nano Banana 2)",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 60,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-05-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.1-flash-image",
      "notes": "Image generation/editing. Two output rates: text and thinking tokens $3.00/1M, image tokens $60.00/1M — price_output_per_mtok holds the image rate (the model's product). Google states the per-image equivalents: $0.045 per 0.5K (512px, 747 tokens), $0.067 per 1K (1024px, 1120 tokens), $0.101 per 2K, $0.151 per 4K (2520 tokens). Batch tier $0.25 in / $1.50 text out / $30.00 image out. Grounding with Google Web and Image Search: 5,000 free requests per month shared across all Gemini 3.x models, then $14 per 1,000 requests. Named by Google as the recommended replacement for all three Imagen 4 models. No context window or max output published for this model.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-1-flash-lite-image",
      "name": "Gemini 3.1 Flash Lite Image (Nano Banana 2 Lite)",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "video",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.25,
      "price_output_per_mtok": 30,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.1-flash-lite-image",
      "notes": "Low-latency image generation/editing. Input $0.25/1M covers text, image and video. Two output rates: text and thinking $1.50/1M, image tokens $30.00/1M — price_output_per_mtok holds the image rate. Google states $0.0336 per 1K image (1024x1024px, 1120 tokens). Batch tier $0.125 in / $0.75 text out / $15.00 image out. released is null as a sourced absence: this model is priced on the pricing page but does not appear anywhere in Google's deprecations table, which is the only Google surface that publishes release dates. Status read as GA because Google suffixes every preview endpoint with -preview and this one carries none.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-pro-image",
      "name": "Gemini 3 Pro Image (Nano Banana Pro)",
      "provider": "Google",
      "status": "ga",
      "modality": [
        "text",
        "image",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 2,
      "price_output_per_mtok": 120,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-05-28",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3-pro-image",
      "notes": "Highest-quality Gemini image model; Google prices its text input and output the same as Gemini 3.1 Pro. Input $2.00/1M text/image, which Google states is $0.0011 per image (560 tokens). Two output rates: text and thinking $12.00/1M, image tokens $120.00/1M — price_output_per_mtok holds the image rate ($0.134 per 1K/2K image at 1120 tokens, $0.24 per 4K image at 2000 tokens). Batch and Flex $1.00 text in / $0.0006 image in / $6.00 text out; Priority $3.60 in / $21.60 text out / $216.00 image out. Grounding with Google Search: 5,000 free requests per month shared across Gemini 3.x, then $14 per 1,000.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-flash-image",
      "name": "Gemini 2.5 Flash Image (Nano Banana)",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.3,
      "price_output_per_mtok": 30,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-10-02",
      "deprecated_on": null,
      "retires_on": "2026-10-02",
      "replacement": "gemini-3.1-flash-image-preview",
      "api_string": "gemini-2.5-flash-image",
      "notes": "The original Nano Banana. Input $0.30/1M text/image; image output $30.00/1M, which Google states as $0.039 per image (images up to 1024x1024px consume 1290 tokens). Batch $0.15 in / $0.0195 per image; Priority $0.54 in / $0.0702 per image. Google's Imagen 4 shutdown banner on the pricing page tells Imagen users to migrate to THIS model, but the deprecations table shows Gemini 2.5 Flash Image is itself deprecated with a 2026-10-02 shutdown, and points Imagen at gemini-3.1-flash-image instead. Its own stated replacement is gemini-3.1-flash-image-preview, which is itself already past its shutdown date — follow that chain to gemini-3.1-flash-image. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-1-flash-image-preview",
      "name": "Gemini 3.1 Flash Image Preview",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-02-26",
      "deprecated_on": null,
      "retires_on": "2026-06-25",
      "replacement": "gemini-3.1-flash-image",
      "api_string": "gemini-3.1-flash-image-preview",
      "notes": "Preview endpoint for Nano Banana 2, superseded by the stable gemini-3.1-flash-image. Google no longer prices it on the pricing page, so token prices are null rather than carried forward. Its stated shutdown date has passed but Google has not marked the row as shut down. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-pro-image-preview",
      "name": "Gemini 3 Pro Image Preview",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text",
        "image",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-11-20",
      "deprecated_on": null,
      "retires_on": "2026-06-25",
      "replacement": "gemini-3-pro-image",
      "api_string": "gemini-3-pro-image-preview",
      "notes": "Preview endpoint for Nano Banana Pro, superseded by the stable gemini-3-pro-image. No longer priced on the pricing page, so token prices are null. Stated shutdown date has passed but Google has not marked the row as shut down. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-flash-image-preview",
      "name": "Gemini 2.5 Flash Image Preview",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05-07",
      "deprecated_on": null,
      "retires_on": "2026-01-15",
      "replacement": "gemini-2.5-flash-image",
      "api_string": "gemini-2.5-flash-image-preview",
      "notes": "Preview endpoint for the original Nano Banana. Marked shut down by Google. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-0-flash-preview-image-generation",
      "name": "Gemini 2.0 Flash Preview Image Generation",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05-07",
      "deprecated_on": null,
      "retires_on": "2025-11-14",
      "replacement": "gemini-2.5-flash-image",
      "api_string": "gemini-2.0-flash-preview-image-generation",
      "notes": "Google's first native Gemini image-generation endpoint. Marked shut down. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "imagen-4-0-generate-001",
      "name": "Imagen 4",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-06-24",
      "deprecated_on": null,
      "retires_on": "2026-08-17",
      "replacement": "gemini-3.1-flash-image",
      "api_string": "imagen-4.0-generate-001",
      "notes": "Text-to-image model billed per image ($0.04 per image, paid tier only — no free tier), so the per-token price fields are N/A. Google's pricing-page banner: \"Imagen 4 models (imagen-4.0-generate-001, imagen-4.0-ultra-generate-001, imagen-4.0-fast-generate-001) are deprecated and will be shut down on August 17, 2026; migrate to Gemini 2.5 Flash Image to avoid service disruption.\" The deprecations table names gemini-3.1-flash-image as the recommended replacement instead, and Gemini 2.5 Flash Image is itself deprecated (shutdown 2026-10-02) — replacement follows the table, which carries the field as data. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "imagen-4-0-ultra-generate-001",
      "name": "Imagen 4 Ultra",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-06-24",
      "deprecated_on": null,
      "retires_on": "2026-08-17",
      "replacement": "gemini-3.1-flash-image",
      "api_string": "imagen-4.0-ultra-generate-001",
      "notes": "Highest-quality Imagen 4 tier, billed per image ($0.06 per image, paid tier only), so token price fields are N/A. Covered by the same 2026-08-17 shutdown banner as the rest of the Imagen 4 line; the deprecations table names gemini-3.1-flash-image as the replacement while the pricing banner says Gemini 2.5 Flash Image. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "imagen-4-0-fast-generate-001",
      "name": "Imagen 4 Fast",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-06-24",
      "deprecated_on": null,
      "retires_on": "2026-08-17",
      "replacement": "gemini-3.1-flash-image",
      "api_string": "imagen-4.0-fast-generate-001",
      "notes": "Cheapest Imagen 4 tier, billed per image ($0.02 per image, paid tier only), so token price fields are N/A. Covered by the same 2026-08-17 shutdown banner; the deprecations table names gemini-3.1-flash-image as the replacement while the pricing banner says Gemini 2.5 Flash Image. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "imagen-3-0-generate-002",
      "name": "Imagen 3",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-02-06",
      "deprecated_on": null,
      "retires_on": "2025-11-10",
      "replacement": "imagen-4.0-generate-001",
      "api_string": "imagen-3.0-generate-002",
      "notes": "Previous-generation text-to-image model, marked shut down by Google. No longer priced on the pricing page. Its stated replacement, Imagen 4, is itself now deprecated with a 2026-08-17 shutdown. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "imagen-4-0-generate-preview-06-06",
      "name": "Imagen 4 Preview (06-06)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-06-24",
      "deprecated_on": null,
      "retires_on": "2026-02-17",
      "replacement": "imagen-4.0-generate-001",
      "api_string": "imagen-4.0-generate-preview-06-06",
      "notes": "Preview endpoint for Imagen 4, marked shut down by Google. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "imagen-4-0-ultra-generate-preview-06-06",
      "name": "Imagen 4 Ultra Preview (06-06)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text-in",
        "image-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-06-24",
      "deprecated_on": null,
      "retires_on": "2026-02-17",
      "replacement": "imagen-4.0-ultra-generate-001",
      "api_string": "imagen-4.0-ultra-generate-preview-06-06",
      "notes": "Preview endpoint for Imagen 4 Ultra, marked shut down by Google. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "veo-3-1-generate-preview",
      "name": "Veo 3.1",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-10-15",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "veo-3.1-generate-preview",
      "notes": "Google's current video-generation model, billed per second of generated video with audio: $0.40/s at 720p and 1080p, $0.60/s at 4K. Paid tier only, no free tier, so token price fields are N/A. Google states you are only charged if the video is generated successfully. No shutdown date announced. This is the recommended migration target for Veo 3 and Veo 2, alongside \"the GA models available through the Gemini Enterprise Agent Platform\".",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "veo-3-1-fast-generate-preview",
      "name": "Veo 3.1 Fast",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-10-15",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "veo-3.1-fast-generate-preview",
      "notes": "Faster Veo 3.1 tier, billed per second with audio: $0.10/s at 720p, $0.12/s at 1080p, $0.30/s at 4K. Paid tier only; token price fields are N/A. No shutdown date announced. Recommended migration target for veo-3.0-fast-generate-001.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "veo-3-1-lite-generate-preview",
      "name": "Veo 3.1 Lite",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-03-31",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "veo-3.1-lite-generate-preview",
      "notes": "Cheapest Veo tier, billed per second with audio: $0.05/s at 720p, $0.08/s at 1080p; 4K output is not supported. Paid tier only; token price fields are N/A. No shutdown date announced.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "veo-3-0-generate-001",
      "name": "Veo 3",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-09",
      "deprecated_on": null,
      "retires_on": "2026-06-30",
      "replacement": "veo-3.1-generate-preview",
      "api_string": "veo-3.0-generate-001",
      "notes": "Billed per second of video with audio at $0.40/s (paid tier only), so token price fields are N/A. Google's pricing-page banner: \"Veo 3 models (veo-3.0-generate-001, veo-3.0-fast-generate-001) are deprecated and will be shut down on June 30, 2026; migrate to Veo 3.1 Preview or the GA models available through the Gemini Enterprise Agent Platform to avoid service disruption.\" The stated date has passed and Google still writes \"will be shut down\" and has not marked the row as shut down, so the status stays deprecated. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "veo-3-0-fast-generate-001",
      "name": "Veo 3 Fast",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-09",
      "deprecated_on": null,
      "retires_on": "2026-06-30",
      "replacement": "veo-3.1-fast-generate-preview",
      "api_string": "veo-3.0-fast-generate-001",
      "notes": "Billed per second with audio: $0.10/s at 720p, $0.12/s at 1080p, $0.30/s at 4K (paid tier only); token price fields are N/A. Covered by the same 2026-06-30 Veo 3 shutdown banner, whose date has passed while Google still states the model is deprecated rather than shut down. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "veo-2-0-generate-001",
      "name": "Veo 2",
      "provider": "Google",
      "status": "deprecated",
      "modality": [
        "text-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-04-09",
      "deprecated_on": null,
      "retires_on": "2026-06-30",
      "replacement": "veo-3.1-generate-preview",
      "api_string": "veo-2.0-generate-001",
      "notes": "Billed per second of video at $0.35/s (paid tier only), so token price fields are N/A. Google's pricing-page banner: \"Veo 2 (veo-2.0-generate-001) is deprecated and will be shut down on June 30, 2026; migrate to Veo 3.1 Preview or the GA models available through the Gemini Enterprise Agent Platform to avoid service disruption.\" The date has passed and the row is not marked shut down, so the status stays deprecated. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "veo-3-0-generate-preview",
      "name": "Veo 3 Preview",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-31",
      "deprecated_on": null,
      "retires_on": "2025-11-12",
      "replacement": "veo-3.1-generate-preview",
      "api_string": "veo-3.0-generate-preview",
      "notes": "Preview endpoint for Veo 3, marked shut down by Google. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "veo-3-0-fast-generate-preview",
      "name": "Veo 3 Fast Preview",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text-in",
        "video-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-07-31",
      "deprecated_on": null,
      "retires_on": "2025-11-12",
      "replacement": "veo-3.1-fast-generate-preview",
      "api_string": "veo-3.0-fast-generate-preview",
      "notes": "Preview endpoint for Veo 3 Fast, marked shut down by Google. Google publishes shutdown dates as \"the earliest possible dates on which a model might be retired\"; already-shut-down models are marked with a grey row on the deprecations page.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "lyria-3-clip-preview",
      "name": "Lyria 3 Clip",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text-in",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-03-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "lyria-3-clip-preview",
      "notes": "Music generation. Billed per request at $0.04 per song for a 30-second clip (paid tier only, no free tier), so the per-token price fields are N/A. No shutdown date announced. Modality is audio-OUT: this model emits music, it does not accept audio input.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "lyria-3-pro-preview",
      "name": "Lyria 3 Pro",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text-in",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-03-25",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "lyria-3-pro-preview",
      "notes": "Music generation, full-song tier. Billed per request at $0.08 per song (paid tier only), so token price fields are N/A. No shutdown date announced. Modality is audio-OUT — it emits music, it does not accept audio input.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "lyria-realtime-exp",
      "name": "Lyria RealTime (Experimental)",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text-in",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05-20",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "lyria-realtime-exp",
      "notes": "Experimental realtime music-generation endpoint. Google lists it in the deprecations table with a release date and \"No shutdown date announced\", but does NOT price it on the pricing page — the null prices here are a sourced absence, not an omission on our side. Modality is audio-OUT.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-1-flash-lite-preview",
      "name": "Gemini 3.1 Flash-Lite Preview",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-03-03",
      "deprecated_on": null,
      "retires_on": "2026-05-25",
      "replacement": "gemini-3.1-flash-lite",
      "api_string": "gemini-3.1-flash-lite-preview",
      "notes": "Preview endpoint for Gemini 3.1 Flash-Lite, superseded by the stable gemini-3.1-flash-lite. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-pro-preview",
      "name": "Gemini 3 Pro Preview",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-11-18",
      "deprecated_on": null,
      "retires_on": "2026-03-09",
      "replacement": "gemini-3.1-pro-preview",
      "api_string": "gemini-3-pro-preview",
      "notes": "The first Gemini 3 Pro endpoint, superseded by gemini-3.1-pro-preview. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-pro-preview-03-25",
      "name": "Gemini 2.5 Pro Preview (03-25)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-03-03",
      "deprecated_on": null,
      "retires_on": "2025-12-02",
      "replacement": "gemini-3.1-pro-preview",
      "api_string": "gemini-2.5-pro-preview-03-25",
      "notes": "Dated preview snapshot of Gemini 2.5 Pro. Google lists three of them (03-25, 05-06, 06-05), all shut down on the same date and all pointing at gemini-3.1-pro-preview. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-pro-preview-05-06",
      "name": "Gemini 2.5 Pro Preview (05-06)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05-06",
      "deprecated_on": null,
      "retires_on": "2025-12-02",
      "replacement": "gemini-3.1-pro-preview",
      "api_string": "gemini-2.5-pro-preview-05-06",
      "notes": "Dated preview snapshot of Gemini 2.5 Pro. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-pro-preview-06-05",
      "name": "Gemini 2.5 Pro Preview (06-05)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-06-05",
      "deprecated_on": null,
      "retires_on": "2025-12-02",
      "replacement": "gemini-3.1-pro-preview",
      "api_string": "gemini-2.5-pro-preview-06-05",
      "notes": "Dated preview snapshot of Gemini 2.5 Pro, the last one before the stable gemini-2.5-pro. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-flash-preview-05-20",
      "name": "Gemini 2.5 Flash Preview (05-20)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05-20",
      "deprecated_on": null,
      "retires_on": "2025-11-18",
      "replacement": "gemini-3.6-flash",
      "api_string": "gemini-2.5-flash-preview-05-20",
      "notes": "Dated preview snapshot of Gemini 2.5 Flash. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-flash-preview-09-25",
      "name": "Gemini 2.5 Flash Preview (09-25)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-25",
      "deprecated_on": null,
      "retires_on": "2026-02-17",
      "replacement": "gemini-3.6-flash",
      "api_string": "gemini-2.5-flash-preview-09-25",
      "notes": "Dated preview snapshot of Gemini 2.5 Flash. Google's deprecations table spells this id gemini-2.5-flash-preview-09-25 while the models page uses gemini-2.5-flash-preview-09-2025 as its worked example of the \"preview\" naming convention; the id here is the one the deprecations table states. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-flash-lite-preview-09-2025",
      "name": "Gemini 2.5 Flash-Lite Preview (09-2025)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.1,
      "price_output_per_mtok": 0.4,
      "price_cached_input_per_mtok": 0.01,
      "knowledge_cutoff": null,
      "released": "2025-09-25",
      "deprecated_on": null,
      "retires_on": "2026-03-31",
      "replacement": "gemini-3.1-flash-lite",
      "api_string": "gemini-2.5-flash-lite-preview-09-2025",
      "notes": "Dated preview snapshot of Gemini 2.5 Flash-Lite. UNUSUAL, and recorded as Google states it rather than reconciled: the deprecations table marks this row gray (shut down, earliest date March 31, 2026) and the pricing page still carries a full live rate card for it. Both readings are first-party, taken the same day. The rates below are those still-published figures; input $0.10 (text/image/video) / $0.30 (audio); cached $0.01 (text/image/video) / $0.03 (audio) plus $1.00 per 1M tokens per hour storage; batch is half of each. Values shown are the text/image/video tier. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-0-flash-001",
      "name": "Gemini 2.0 Flash 001",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-02-05",
      "deprecated_on": null,
      "retires_on": "2026-06-01",
      "replacement": "gemini-3.6-flash",
      "api_string": "gemini-2.0-flash-001",
      "notes": "Pinned stable snapshot of gemini-2.0-flash; Google lists the alias and the snapshot as separate rows with the same dates. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-0-flash-lite-001",
      "name": "Gemini 2.0 Flash-Lite 001",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-02-25",
      "deprecated_on": null,
      "retires_on": "2026-06-01",
      "replacement": "gemini-3.1-flash-lite",
      "api_string": "gemini-2.0-flash-lite-001",
      "notes": "Pinned stable snapshot of gemini-2.0-flash-lite. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-0-flash-lite-preview",
      "name": "Gemini 2.0 Flash-Lite Preview",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-02-05",
      "deprecated_on": null,
      "retires_on": "2025-12-09",
      "replacement": "gemini-2.5-flash-lite",
      "api_string": "gemini-2.0-flash-lite-preview",
      "notes": "Preview endpoint for Gemini 2.0 Flash-Lite; Google lists both the undated alias and the dated -02-05 snapshot. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-0-flash-lite-preview-02-05",
      "name": "Gemini 2.0 Flash-Lite Preview (02-05)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-02-05",
      "deprecated_on": null,
      "retires_on": "2025-12-09",
      "replacement": "gemini-2.5-flash-lite",
      "api_string": "gemini-2.0-flash-lite-preview-02-05",
      "notes": "Dated preview snapshot of Gemini 2.0 Flash-Lite. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-0-flash-live-001",
      "name": "Gemini 2.0 Flash Live 001",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "audio",
        "video",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-04-09",
      "deprecated_on": null,
      "retires_on": "2025-12-09",
      "replacement": "gemini-3.1-flash-live-preview",
      "api_string": "gemini-2.0-flash-live-001",
      "notes": "Live API endpoint (bidirectional streaming voice/video), the first one Google shipped. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-live-2-5-flash-preview",
      "name": "Gemini Live 2.5 Flash Preview",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "audio",
        "video",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-06-17",
      "deprecated_on": null,
      "retires_on": "2025-12-09",
      "replacement": "gemini-3.1-flash-live-preview",
      "api_string": "gemini-live-2.5-flash-preview",
      "notes": "Live API preview endpoint. Note the id puts \"live\" before the version, unlike the 2.0 and 3.1 Live endpoints. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "text-embedding-004",
      "name": "Text Embedding 004",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2024-04-09",
      "deprecated_on": null,
      "retires_on": "2026-01-14",
      "replacement": "gemini-embedding-2",
      "api_string": "text-embedding-004",
      "notes": "Google's pre-Gemini text embedding endpoint. Not priced on the current pricing page and no dimension count is published for it there, so both are null. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "embedding-2-preview",
      "name": "Gemini Embedding 2 Preview",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video",
        "audio",
        "pdf"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-03-10",
      "deprecated_on": null,
      "retires_on": "2026-08-10",
      "replacement": "gemini-embedding-2",
      "api_string": "embedding-2-preview",
      "notes": "Preview endpoint for the multimodal Gemini Embedding 2, shut down nine days before this row was added; the stable gemini-embedding-2 carries the live rates. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "embedding-001",
      "name": "Embedding 001",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2025-10-30",
      "replacement": "gemini-embedding-2",
      "api_string": "embedding-001",
      "notes": "Legacy text embedding endpoint. Google's deprecations table leaves the Release date cell EMPTY for this row, so released is null rather than inferred. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "embedding-gecko-001",
      "name": "Embedding Gecko 001",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2025-10-30",
      "replacement": "gemini-embedding-2",
      "api_string": "embedding-gecko-001",
      "notes": "Legacy PaLM-era embedding endpoint. Release date cell empty on Google's table, so released is null. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-embedding-exp",
      "name": "Gemini Embedding Experimental",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2025-10-30",
      "replacement": "gemini-embedding-2",
      "api_string": "gemini-embedding-exp",
      "notes": "Experimental embedding endpoint. Release date cell empty on Google's table, so released is null. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-embedding-exp-03-07",
      "name": "Gemini Embedding Experimental (03-07)",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": "2025-10-30",
      "replacement": "gemini-embedding-2",
      "api_string": "gemini-embedding-exp-03-07",
      "notes": "Dated experimental embedding snapshot. Release date cell empty on Google's table, so released is null. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-robotics-er-1-5-preview",
      "name": "Gemini Robotics-ER 1.5",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "image",
        "video"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": null,
      "price_output_per_mtok": null,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-09-25",
      "deprecated_on": null,
      "retires_on": "2026-04-30",
      "replacement": "gemini-robotics-er-1.6-preview",
      "api_string": "gemini-robotics-er-1.5-preview",
      "notes": "Embodied-reasoning vision-language model for robotics, superseded by gemini-robotics-er-1.6-preview. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Not priced on the current pricing page, so token prices are null rather than guessed from the surviving family member. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/deprecations",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-1-flash-live-preview",
      "name": "Gemini 3.1 Flash Live",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "audio",
        "video",
        "image",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.75,
      "price_output_per_mtok": 4.5,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-03-11",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.1-flash-live-preview",
      "notes": "Live API model for real-time audio-to-audio dialogue; Google calls it \"our high-quality, low-latency audio-to-audio (A2A) model\". Rates differ per input type: input $0.75 (text) / $3.00 or $0.005 per minute (audio) / $1.00 or $0.002 per minute (image/video); output $4.50 (text) / $12.00 or $0.018 per minute (audio). Values shown are the TEXT tier, matching how the catalog carries the other multi-tier Gemini rows. Grounding with Google Search: 5,000 free requests per month shared across Gemini 3.x, then $14 per 1,000 requests. Google states \"No shutdown date announced\". Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-flash-native-audio-preview-12-2025",
      "name": "Gemini 2.5 Flash Live (Native Audio)",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "audio",
        "video",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 2,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-12-12",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "gemini-3.1-flash-live-preview",
      "api_string": "gemini-2.5-flash-native-audio-preview-12-2025",
      "notes": "Live API model with native-audio reasoning. Rates differ per input type: input $0.50 (text) / $3.00 (audio or video); output $2.00 (text) / $12.00 (audio). Values shown are the TEXT tier. Google states \"No shutdown date announced\" but already names gemini-3.1-flash-live-preview as the recommended replacement, so the pointer is recorded and the status stays preview (never infer a lifecycle change the provider has not declared). Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-5-live-translate-preview",
      "name": "Gemini 3.5 Live Translate",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "audio",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 3.5,
      "price_output_per_mtok": 21,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": null,
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.5-live-translate-preview",
      "notes": "Real-time speech-to-speech translation across 70+ languages. NOT on Google's deprecations table at all (checked 2026-08-19) — it was found by diffing the models and pricing pages against the catalog — so no release date and no lifecycle field is published for it anywhere first-party, and released stays null. Audio-only billing: input $3.50 or $0.0053 per minute, output $21.00 or $0.0315 per minute, at 25 tokens per second of audio, which Google states works out to roughly $0.0368 per minute. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-3-1-flash-tts-preview",
      "name": "Gemini 3.1 Flash TTS",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 20,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2026-04-13",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": null,
      "api_string": "gemini-3.1-flash-tts-preview",
      "notes": "Text-to-speech model: text in, audio out. Input $1.00 (text), output $20.00 (audio); batch $0.50 / $10.00. Audio tokens are billed at 25 tokens per second of audio. Google states \"No shutdown date announced\". Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-flash-preview-tts",
      "name": "Gemini 2.5 Flash TTS",
      "provider": "Google",
      "status": "preview",
      "modality": [
        "text",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 0.5,
      "price_output_per_mtok": 10,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05-20",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "gemini-3.1-flash-tts-preview",
      "api_string": "gemini-2.5-flash-preview-tts",
      "notes": "Text-to-speech model: text in, audio out. Input $0.50 (text), output $10.00 (audio); batch $0.25 / $5.00. Google states \"No shutdown date announced\" but names gemini-3.1-flash-tts-preview as the recommended replacement, so the pointer is recorded and the status stays preview. Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    },
    {
      "id": "gemini-2-5-pro-preview-tts",
      "name": "Gemini 2.5 Pro TTS",
      "provider": "Google",
      "status": "retired",
      "modality": [
        "text",
        "audio-out"
      ],
      "context_window": null,
      "max_output_tokens": null,
      "price_input_per_mtok": 1,
      "price_output_per_mtok": 20,
      "price_cached_input_per_mtok": null,
      "knowledge_cutoff": null,
      "released": "2025-05-20",
      "deprecated_on": null,
      "retires_on": null,
      "replacement": "gemini-3.1-flash-tts-preview",
      "api_string": "gemini-2.5-pro-preview-tts",
      "notes": "Text-to-speech model: text in, audio out. RECORDED EXACTLY AS GOOGLE STATES IT, because the two things it states do not agree: the deprecations row is GRAY (= already shut down) and its Shutdown date cell reads \"No shutdown date announced\", while the pricing page still carries a rate card (input $1.00 text, output $20.00 audio; batch $0.50 / $10.00) with the free tier marked \"Not available\". Status follows the gray marker, which is the page's own legend; retires_on stays null because no date is published. The still-published rates ARE carried in the price fields, the same way gemini-2.5-flash-lite-preview-09-2025 is handled: a rate Google is still publishing is a fact, and a consumer who wants only sellable models filters on status. Google marks already-shut-down models with a gray row on the deprecations table (\"Already-shutdown models are indicated with gray backgrounds\"), and this row is gray, so status is retired. Google states the listed shutdown dates are \"the earliest possible dates on which a model might be retired\". Google's current models page publishes no input/output token limits for this endpoint (read 2026-08-19), so context_window and max_output_tokens are null rather than carried over from a page that no longer states them. Lifecycle read first-party from Google's deprecations table on 2026-08-19.",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing",
      "open_weight": false,
      "embedding_dimensions": null,
      "price_per_1k_searches": null
    }
  ]
}