{
  "updated": "2026-09-12",
  "verifiedDate": "2026-09-02",
  "note": "Dataset updated 2026-09-02. Provider names, release dates, API identifiers, list prices, context limits, and provider-published benchmark figures are linked to primary sources in model-figures.json. Formal benchmark fields are null unless an exact published variant is recorded. Latency fields are null until benchr has a reproducible measurement protocol or a suitable verified provider source. The 0-100 capability profiles are labeled editorial decision aids, not lab measurements. Missing official figures remain null. Material corrections are recorded at /corrections and in history.json. DeepSeek's two rows carry the peak (list) rate; its off-peak rate is half that between 16:00 UTC on August 16, 2026 and any further change, and is recorded in the offpeak_* fields. Claude Fable 5.1 is an additional active model ID; its initial editorial profile intentionally matches Fable 5 rather than claiming an unsourced score increase.",
  "models": [
    {
      "id": "qwen-3-6-27b",
      "name": "Qwen3.6-27B",
      "company": "Alibaba (Qwen)",
      "type": "open",
      "license": "Apache-2.0",
      "released": "2026-04-22",
      "api_name": "Qwen/Qwen3.6-27B",
      "review_url": "articles/qwen-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "context": {
        "max_tokens": 262144,
        "effective_tokens": 200000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": 77.2,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 87.8,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 88,
        "reasoning": 86,
        "writing": 82,
        "vision": 82,
        "long_context": 84,
        "multilingual": 95
      },
      "best_for": [
        "Multilingual coding (Chinese, Japanese, Korean, Arabic)",
        "Local inference on consumer GPUs — dense 27B",
        "Tool-use agent loops at zero API cost"
      ],
      "skip_if": [
        "You want a managed hosted API — these are open weights you self-host",
        "Absolute deepest single-language reasoning"
      ],
      "figure_id": "qwen-3-6-27b",
      "benchmarks_estimated": []
    },
    {
      "id": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "company": "Anthropic",
      "type": "small",
      "license": "proprietary",
      "released": "2025-10-15",
      "api_name": "claude-haiku-4-5",
      "review_url": "articles/claude-haiku-4-5-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 1.0,
        "output_per_million": 5.0,
        "cache_input_per_million": 0.1,
        "batch_discount": 0.5,
        "batch_input": 0.5,
        "batch_output": 2.5
      },
      "context": {
        "max_tokens": 200000,
        "effective_tokens": 150000,
        "max_output_tokens": 64000
      },
      "benchmarks": {
        "swe_bench_verified": 73.3,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 75,
        "reasoning": 76,
        "writing": 80,
        "vision": 72,
        "long_context": 82,
        "multilingual": 78
      },
      "best_for": [
        "High-volume simple tasks",
        "Real-time chat",
        "Classification, routing, extraction"
      ],
      "skip_if": [
        "Complex reasoning needed",
        "Long-document analysis"
      ],
      "figure_id": "claude-haiku-4-5",
      "benchmarks_estimated": [],
      "ar": {
        "best_for": [
          "كانت المهام بسيطة وعالية الحجم",
          "كانت المحادثة لحظية",
          "كان العمل تصنيفاً وتوجيه طلبات واستخراج حقول"
        ],
        "skip_if": [
          "كنت تحتاج استدلالاً معقّداً",
          "كان عملك تحليل مستندات طويلة"
        ]
      }
    },
    {
      "id": "claude-sonnet-4-6",
      "name": "Claude Sonnet 4.6",
      "company": "Anthropic",
      "type": "mid",
      "license": "proprietary",
      "released": "2026-02-17",
      "api_name": "claude-sonnet-4-6",
      "review_url": "articles/claude-sonnet-4-6-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 3.0,
        "output_per_million": 15.0,
        "cache_input_per_million": 0.3,
        "batch_discount": 0.5,
        "batch_input": 1.5,
        "batch_output": 7.5
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 700000,
        "max_output_tokens": 64000
      },
      "benchmarks": {
        "swe_bench_verified": 79.6,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 89.9,
        "arc_agi_2": 58.3
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 88,
        "reasoning": 87,
        "writing": 89,
        "vision": 80,
        "long_context": 91,
        "multilingual": 86
      },
      "best_for": [
        "Existing Sonnet 4.6 pipelines not yet migrated",
        "Cost-effective coding",
        "Bulk content tasks",
        "Daily-driver API workloads"
      ],
      "skip_if": [
        "You need frontier-grade reasoning",
        "New work - Claude Sonnet 5 is newer and cheaper on both input and output"
      ],
      "figure_id": "claude-sonnet-4-6",
      "benchmarks_estimated": [],
      "ar": {
        "best_for": [
          "تحتاج نموذجاً افتراضياً للإنتاج",
          "تكتب كوداً ويهمّك العائد مقابل الكلفة",
          "تنجز مهام محتوى بالجملة",
          "تشغّل أحمالك اليومية على API"
        ],
        "skip_if": [
          "احتجت استدلالاً بمستوى الطبقة الأولى"
        ]
      },
      "opener": {
        "en": "The record carries a second output ceiling for this model — 300,000 tokens, marked beta — and says nothing about what enables it. Until it does, size a long-output job against the standard limit.",
        "ar": "يحمل السجل رقماً ثانياً لأقصى إخراج هذا النموذج: 300,000 توكن في بيتا، دون ذكر شرط إتاحته. وما دام الشرط غير مذكور، تُحسب المخرجات الطويلة على الحدّ القياسي."
      }
    },
    {
      "id": "claude-opus-4-7",
      "name": "Claude Opus 4.7",
      "company": "Anthropic",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-04-16",
      "api_name": "claude-opus-4-7",
      "review_url": "articles/claude-opus-4-7-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 5.0,
        "output_per_million": 25.0,
        "cache_input_per_million": 0.5,
        "batch_discount": 0.5,
        "batch_input": 2.5,
        "batch_output": 12.5
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 700000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": 87.6,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 94.2,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 96,
        "reasoning": 96,
        "writing": 90,
        "vision": 85,
        "long_context": 94,
        "multilingual": 89
      },
      "best_for": [
        "Complex coding tasks",
        "Long-document analysis",
        "Production agent loops",
        "Architecture decisions"
      ],
      "skip_if": [
        "You need cheap volume",
        "Simple summarization",
        "Sonnet covers your workload",
        "New work - Claude Opus 5 is the newer model at the same price"
      ],
      "figure_id": "claude-opus-4-7",
      "benchmarks_estimated": [],
      "ar": {
        "best_for": [
          "مهام البرمجة المعقّدة",
          "تحليل المستندات الطويلة",
          "حلقات الوكلاء في الإنتاج",
          "القرارات المعمارية"
        ],
        "skip_if": [
          "كنت تحتاج تشغيلاً عالي الحجم بتكلفة منخفضة",
          "كان العمل تلخيصاً بسيطاً",
          "كان Sonnet يغطي حِملك"
        ]
      }
    },
    {
      "id": "claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "company": "Anthropic",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-05-28",
      "api_name": "claude-opus-4-8",
      "review_url": "articles/claude-opus-4-8-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 5.0,
        "output_per_million": 25.0,
        "cache_input_per_million": 0.5,
        "batch_discount": 0.5,
        "fast_mode_input": 10.0,
        "fast_mode_output": 50.0,
        "batch_input": 2.5,
        "batch_output": 12.5
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 700000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": 88.6,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 93.6,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 97,
        "reasoning": 96,
        "writing": 91,
        "vision": 86,
        "long_context": 94,
        "multilingual": 90
      },
      "best_for": [
        "Highest-stakes coding tasks",
        "Complex multi-step agents",
        "Architecture decisions",
        "Production SWE-bench-level work"
      ],
      "skip_if": [
        "You need cheap volume",
        "Sonnet handles your workload",
        "Speed is the priority — use Fast Mode instead",
        "New work - Claude Opus 5 is the newer model at the same price"
      ],
      "figure_id": "claude-opus-4-8",
      "benchmarks_estimated": [],
      "ar": {
        "best_for": [
          "يكون العمل البرمجي عندك هو الأعلى مخاطرة",
          "تشغّل وكلاء معقّدين متعدّدي الخطوات",
          "تتّخذ قرارات معمارية",
          "يكون العمل الإنتاجي بصعوبة مهام SWE-bench"
        ],
        "skip_if": [
          "كنت تحتاج إلى تشغيل عالي الحجم بتكلفة منخفضة",
          "كان Sonnet يكفي لعبء عملك",
          "كانت السرعة هي الأولوية، فاستخدم الوضع السريع بدل القياسي"
        ]
      },
      "opener": {
        "en": "Anthropic's deprecation table sends three retired models here: Opus 4 and Opus 4.1, both priced at $15/$75 per million tokens, and Claude 3 Opus, whose retiring rate is not recorded.",
        "ar": "الوضع السريع هنا معدّل اختياري وليس السعر الأساسي: $10 للإدخال و$50 للإخراج لكل مليون توكن، مقابل سرعة إخراج أعلى بنحو 2.5×. أوقفت Anthropic هذا الوضع لنموذج Opus 4.7 في 24 يوليو 2026، ولا يزال متاحاً في Opus 4.8."
      }
    },
    {
      "id": "claude-fable-5",
      "benchmarks_estimated": [],
      "name": "Claude Fable 5",
      "company": "Anthropic",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-06-09",
      "api_name": "claude-fable-5",
      "review_url": "articles/claude-fable-5-launch",
      "deprecated": false,
      "pricing": {
        "input_per_million": 10.0,
        "output_per_million": 50.0,
        "cache_input_per_million": 1.0
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 700000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "swe_bench_pro": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 99,
        "reasoning": 98,
        "writing": 95,
        "vision": 94,
        "long_context": 95,
        "multilingual": 91
      },
      "best_for": [
        "The hardest long-horizon agentic coding",
        "Large-codebase migrations",
        "Frontier research and finance work",
        "Vision-driven agent loops"
      ],
      "skip_if": [
        "Price matters - Claude Opus 5 is half the cost",
        "Offensive-security or bio work — classifiers return an explicit refusal; another-model retry requires configured application logic",
        "Routine chat and drafting",
        "New work - Claude Fable 5.1 replaced it at the same base price with cheaper cache reads"
      ],
      "figure_id": "claude-fable-5"
    },
    {
      "id": "claude-sonnet-5",
      "benchmarks_estimated": [],
      "name": "Claude Sonnet 5",
      "company": "Anthropic",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-07-01",
      "api_name": "claude-sonnet-5",
      "review_url": "articles/claude-sonnet-5-launch",
      "deprecated": false,
      "pricing": {
        "input_per_million": 2.0,
        "output_per_million": 10.0,
        "cache_input_per_million": 0.2,
        "batch_discount": 0.5,
        "batch_input": 1.0,
        "batch_output": 5.0
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 800000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": 89.4,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 92.0,
        "arc_agi_2": 20.0
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 96,
        "reasoning": 95,
        "writing": 93,
        "vision": 88,
        "long_context": 93,
        "multilingual": 89
      },
      "best_for": [
        "Frontier coding at Sonnet-tier pricing",
        "128K max output — well above Sonnet 4.6's 64K",
        "Teams upgrading off Sonnet 4.6 without Opus pricing",
        "Production agents that need Mythos-class quality"
      ],
      "skip_if": [
        "Offensive-security or bio work — classifiers return an explicit refusal; another-model retry requires configured application logic",
        "Absolute cheapest volume — Haiku 4.5 remains cheaper",
        "You need Fable 5.1's higher ceiling on the hardest tasks"
      ],
      "figure_id": "claude-sonnet-5"
    },
    {
      "id": "claude-opus-5",
      "benchmarks_estimated": [],
      "name": "Claude Opus 5",
      "company": "Anthropic",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-07-24",
      "api_name": "claude-opus-5",
      "review_url": "articles/claude-opus-5-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 5.0,
        "output_per_million": 25.0,
        "cache_input_per_million": 0.5
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 800000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 97,
        "reasoning": 97,
        "writing": 92,
        "vision": 89,
        "long_context": 94,
        "multilingual": 90
      },
      "best_for": [
        "Complex agentic coding and enterprise work",
        "1M-context workloads that need 128K output",
        "Teams migrating from Claude Opus 4.8 at the same base API price"
      ],
      "skip_if": [
        "You need a provider-published benchmark comparison — none is recorded in the verified facts yet",
        "You need the lowest-cost Claude tier — Sonnet and Haiku remain cheaper"
      ],
      "figure_id": "claude-opus-5"
    },
    {
      "id": "claude-fable-5-1",
      "benchmarks_estimated": [],
      "name": "Claude Fable 5.1",
      "company": "Anthropic",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-09-01",
      "api_name": "claude-fable-5-1",
      "review_url": "articles/claude-fable-5-1-migration",
      "deprecated": false,
      "pricing": {
        "input_per_million": 10.0,
        "output_per_million": 50.0,
        "cache_input_per_million": 0.25,
        "batch_discount": 0.5,
        "batch_input": 5.0,
        "batch_output": 25.0,
        "cache_write_5m_per_million": 12.5,
        "cache_write_1h_per_million": 20.0
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 700000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "swe_bench_pro": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 99,
        "reasoning": 98,
        "writing": 95,
        "vision": 94,
        "long_context": 95,
        "multilingual": 91
      },
      "best_for": [
        "Production agentic coding with a 1M-token context window",
        "Repeated-prefix workloads where $0.25 per 1M cache reads materially change cost",
        "Fable 5 migrations whose tool choice remains auto or none",
        "Long-running assistant workflows that can use adaptive thinking"
      ],
      "skip_if": [
        "Your client forces tool_choice type any or a named tool — the 5.1 API returns HTTP 400",
        "You require Priority Tier — Anthropic does not support Fable 5.1 on it",
        "You require zero-data-retention access without Anthropic authorization",
        "You expect a published Fable 5.1 score in the tool's normalized benchmark columns"
      ],
      "figure_id": "claude-fable-5-1"
    },
    {
      "id": "deepseek-v4-pro",
      "name": "DeepSeek V4-Pro",
      "company": "DeepSeek",
      "type": "frontier-open",
      "license": "MIT",
      "released": "2026-04-24",
      "api_name": "deepseek-v4-pro",
      "review_url": "articles/deepseek-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 1.32,
        "output_per_million": 3.96,
        "cache_input_per_million": 0.044,
        "batch_discount": null,
        "self_hosted": true,
        "offpeak_input_per_million": 0.66,
        "offpeak_output_per_million": 1.98,
        "offpeak_cache_input_per_million": 0.022,
        "pricing_note": "Peak (list) rate. DeepSeek bills 01:00-04:00 and 06:00-10:00 UTC Monday to Friday at this rate and every other hour at half of it, effective 16:00 UTC on 2026-08-16."
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 700000,
        "max_output_tokens": 384000
      },
      "benchmarks": {
        "swe_bench_verified": 80.6,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 90.1,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 93,
        "reasoning": 92,
        "writing": 84,
        "vision": null,
        "long_context": 86,
        "multilingual": 87
      },
      "best_for": [
        "Frontier-grade open-weight coding",
        "Math-heavy work",
        "Self-hosted production, where the August 2026 API price rise does not apply"
      ],
      "skip_if": [
        "You need vision/multimodal",
        "You can't manage GPU hosting"
      ],
      "figure_id": "deepseek-v4-pro",
      "benchmarks_estimated": [],
      "ar": {
        "best_for": [
          "تحتاج إلى نموذج برمجة مفتوح الأوزان",
          "تكون الرياضيات محور المهمة",
          "تشغّله في الإنتاج باستضافة ذاتية، إذ لا تسري عليها زيادة أسعار API في أغسطس 2026"
        ],
        "skip_if": [
          "كان عملك يحتاج إلى الرؤية أو إلى مدخلات متعددة الوسائط",
          "لم تستطع إدارة استضافة على GPU"
        ]
      }
    },
    {
      "id": "deepseek-v4-1-flash",
      "name": "DeepSeek-V4.1-Flash",
      "company": "DeepSeek",
      "type": "open",
      "license": null,
      "released": "2026-09-10",
      "api_name": "deepseek-flash",
      "review_url": null,
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.3,
        "output_per_million": 1.2,
        "cache_input_per_million": 0.006,
        "batch_discount": null,
        "self_hosted": false,
        "offpeak_input_per_million": 0.15,
        "offpeak_output_per_million": 0.6,
        "offpeak_cache_input_per_million": 0.003,
        "pricing_note": "Peak (list) rate. DeepSeek bills 01:00-04:00 and 06:00-10:00 UTC Monday to Friday at this rate and every other hour at half of it."
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": null,
        "max_output_tokens": 384000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": null,
        "reasoning": null,
        "writing": null,
        "vision": null,
        "long_context": null,
        "multilingual": null
      },
      "best_for": [
        "Replacing the retired DeepSeek V4-Flash on the same API, at a lower listed rate",
        "Batchable work scheduled outside DeepSeek's peak hours, where every rate halves to among the lowest hosted rates among models tracked by benchr - not a market-wide guarantee",
        "Long-context work that needs a 1M-token window and up to 384K output tokens"
      ],
      "skip_if": [
        "You need published benchmark figures - DeepSeek released none that benchr records",
        "You need a stated license or open weights for this release - DeepSeek published neither",
        "Your traffic is fixed to European or Asian business hours, which sit inside the peak window"
      ],
      "figure_id": "deepseek-v4-1-flash",
      "benchmarks_estimated": []
    },
    {
      "id": "gemini-3-1-pro",
      "name": "Gemini 3.1 Pro",
      "company": "Google",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-02-19",
      "api_name": "gemini-3.1-pro-preview",
      "review_url": "articles/gemini-3-1-pro-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 2.0,
        "output_per_million": 12.0,
        "input_per_million_over_200k": 4.0,
        "output_per_million_over_200k": 18.0,
        "cache_input_per_million": 0.2,
        "batch_discount": null,
        "batch_input": 1.0,
        "batch_output": 6.0
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 800000,
        "max_output_tokens": 64000
      },
      "benchmarks": {
        "swe_bench_verified": 80.6,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 94.3,
        "arc_agi_2": 77.1
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 84,
        "reasoning": 90,
        "writing": 84,
        "vision": 95,
        "long_context": 92,
        "multilingual": 91
      },
      "best_for": [
        "Deep reasoning in the Gemini family",
        "Long-context vision work",
        "Workspace integration"
      ],
      "skip_if": [
        "Coding agents — Flash is faster and cheaper",
        "Cost-sensitive workloads — note the over-200K price bump"
      ],
      "figure_id": "gemini-3-1-pro",
      "benchmarks_estimated": [],
      "ar": {
        "best_for": [
          "تحتاج إلى استدلال عميق وتعمل ضمن عائلة Gemini",
          "يجمع عملك بين الرؤية والسياق الطويل",
          "تحتاج إلى تكامل مع Workspace"
        ],
        "skip_if": [
          "كنت تشغّل وكلاء البرمجة، فسعر Flash أقل",
          "كان عبء عملك حسّاساً للتكلفة، فسعر الإدخال يتضاعف فوق 200 ألف توكن"
        ]
      },
      "opener": {
        "en": "The >200K-token tier applies to batch pricing as well: batch input doubles from $1 to $2 per 1M and batch output rises from $6 to $9.",
        "ar": "فوق 200 ألف توكن ينتقل النموذج إلى تسعيرة ثانية: يتضاعف سعر الإدخال إلى 4$ ويرتفع سعر الإخراج إلى 18$ لكل مليون توكن، وسعر الإخراج يشمل توكنات التفكير."
      }
    },
    {
      "id": "gemini-3-5-flash",
      "benchmarks_estimated": [],
      "name": "Gemini 3.5 Flash",
      "company": "Google",
      "type": "mid",
      "license": "proprietary",
      "released": "2026-05-19",
      "api_name": "gemini-3.5-flash",
      "review_url": "articles/gemini-3-5-flash-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 1.5,
        "output_per_million": 9.0,
        "cache_input_per_million": 0.15,
        "batch_discount": null,
        "free_tier": true,
        "batch_input": 0.75,
        "batch_output": 4.5
      },
      "context": {
        "max_tokens": 1048576,
        "effective_tokens": 700000,
        "max_output_tokens": 65536
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 88,
        "reasoning": 86,
        "writing": 84,
        "vision": 92,
        "long_context": 90,
        "multilingual": 91
      },
      "best_for": [
        "Coding agents at speed",
        "Parallel agent execution",
        "Multimodal tasks",
        "Existing Gemini 3.5 Flash integrations"
      ],
      "skip_if": [
        "You need the deepest single-call reasoning — use Gemini 3.1 Pro",
        "New work - Gemini 3.8 Flash is newer and cheaper through 2026"
      ],
      "figure_id": "gemini-3-5-flash"
    },
    {
      "id": "gemini-3-5-flash-lite",
      "benchmarks_estimated": [],
      "name": "Gemini 3.5 Flash-Lite",
      "company": "Google",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-07-21",
      "review_url": "articles/gemini-3-5-flash-lite-review",
      "api_name": "gemini-3.5-flash-lite",
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.3,
        "output_per_million": 2.5
      },
      "context": {
        "max_tokens": 1048576,
        "effective_tokens": 750000,
        "max_output_tokens": 65536
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 84,
        "reasoning": 83,
        "writing": 78,
        "vision": 83,
        "long_context": 92,
        "multilingual": 84
      },
      "best_for": [
        "High-volume automation and document extraction",
        "Low-cost subagents with a 1M-token context window",
        "Multimodal input workflows that need structured text output"
      ],
      "skip_if": [
        "You need a provider-published score for the benchmarks shown in benchr's table",
        "You need the higher capability ceiling of Gemini 3.6 Flash"
      ],
      "figure_id": "gemini-3-5-flash-lite"
    },
    {
      "id": "gemini-3-6-flash",
      "benchmarks_estimated": [],
      "name": "Gemini 3.6 Flash",
      "company": "Google",
      "type": "mid",
      "license": "proprietary",
      "released": "2026-07-21",
      "api_name": "gemini-3.6-flash",
      "review_url": "articles/gemini-3-6-flash-launch",
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.75,
        "output_per_million": 3.75,
        "cache_input_per_million": 0.075,
        "batch_discount": null,
        "batch_input": 0.375,
        "batch_output": 1.875,
        "free_tier": true
      },
      "context": {
        "max_tokens": 1048576,
        "effective_tokens": 720000,
        "max_output_tokens": 65536
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 89,
        "reasoning": 87,
        "writing": 84,
        "vision": 92,
        "long_context": 90,
        "multilingual": 91
      },
      "best_for": [
        "Output-heavy coding agents",
        "Google Search and Maps grounded workflows",
        "Stable Flash migrations from 2.x and preview endpoints",
        "Multimodal input with text output"
      ],
      "skip_if": [
        "You need published official benchmark tables before procurement",
        "Gemini 3.8 Flash costs the same and is the newer model",
        "You need image generation or Live API voice output"
      ],
      "figure_id": "gemini-3-6-flash"
    },
    {
      "id": "gemini-3-7-flash",
      "benchmarks_estimated": [],
      "name": "Gemini 3.7 Flash",
      "company": "Google",
      "type": "mid",
      "license": "proprietary",
      "released": "2026-08-13",
      "api_name": "gemini-3.7-flash",
      "review_url": "articles/gemini-3-7-flash-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.75,
        "output_per_million": 3.75,
        "cache_input_per_million": 0.075,
        "batch_discount": null,
        "batch_input": 0.375,
        "batch_output": 1.875,
        "free_tier": true
      },
      "context": {
        "max_tokens": 1048576,
        "effective_tokens": 720000,
        "max_output_tokens": 65536
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 91,
        "reasoning": 88,
        "writing": 85,
        "vision": 92,
        "long_context": 90,
        "multilingual": 91
      },
      "best_for": [
        "Coding and agent work at Flash prices",
        "Teams already on Gemini 3.6 Flash — same price, newer model",
        "Long-context multimodal input with text output",
        "Free-tier prototyping before a paid rollout"
      ],
      "skip_if": [
        "You need published official benchmark tables — Google shipped none",
        "You want the cheapest subagent tier — use Gemini 3.5 Flash-Lite",
        "You need image generation, audio output, or the Live API",
        "Gemini 3.8 Flash costs the same and is the newer model"
      ],
      "figure_id": "gemini-3-7-flash"
    },
    {
      "id": "gemini-3-8-flash",
      "benchmarks_estimated": [],
      "name": "Gemini 3.8 Flash",
      "company": "Google",
      "type": "mid",
      "license": "proprietary",
      "released": "2026-09-02",
      "api_name": "gemini-3.8-flash",
      "review_url": null,
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.75,
        "output_per_million": 3.75,
        "cache_input_per_million": 0.075,
        "batch_discount": null,
        "batch_input": 0.375,
        "batch_output": 1.875,
        "free_tier": true,
        "intro_through": "2026-12-31",
        "standard_input_per_million_from_2027_01_01": 1.5,
        "standard_output_per_million_from_2027_01_01": 7.5
      },
      "context": {
        "max_tokens": 1048576,
        "effective_tokens": null,
        "max_output_tokens": 65536
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": null,
        "reasoning": null,
        "writing": null,
        "vision": null,
        "long_context": null,
        "multilingual": null
      },
      "best_for": [
        "Teams already on 3.7 Flash - identical price, newer model",
        "Agentic and multi-step coding work at Flash prices",
        "1M-token context with tunable thinking levels",
        "Free-tier prototyping before a paid rollout"
      ],
      "skip_if": [
        "You need the figures benchr has measured - this record is days old and unmeasured",
        "You want published SWE-bench Verified or GPQA tables - Google published neither",
        "You are budgeting past 2026 - the price doubles on January 1, 2027",
        "You need the Cyber variant - it is Fairwind Program only, not a public API model"
      ],
      "figure_id": "gemini-3-8-flash"
    },
    {
      "id": "llama-4-maverick",
      "benchmarks_estimated": [],
      "name": "Llama 4 Maverick",
      "company": "Meta",
      "type": "frontier-open",
      "license": "Llama 4 Community License",
      "released": "2025-04-05",
      "api_name": "meta-llama/Llama-4-Maverick-17B-128E-Instruct",
      "review_url": "articles/llama-4-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 700000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 69.8,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 80,
        "reasoning": 81,
        "writing": 76,
        "vision": 80,
        "long_context": 85,
        "multilingual": 80
      },
      "best_for": [
        "Self-hosting under the Llama 4 Community License with a 1M-token window",
        "Self-hosted production at zero licensing cost",
        "Multimodal at no API cost"
      ],
      "skip_if": [
        "You need the very best reasoning or coding",
        "You can't manage GPU infrastructure"
      ],
      "figure_id": "llama-4-maverick"
    },
    {
      "id": "llama-4-scout",
      "benchmarks_estimated": [],
      "name": "Llama 4 Scout",
      "company": "Meta",
      "type": "open",
      "license": "Llama 4 Community License",
      "released": "2025-04-05",
      "api_name": "meta-llama/Llama-4-Scout-17B-16E",
      "review_url": "articles/llama-4-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "context": {
        "max_tokens": 10000000,
        "effective_tokens": 2000000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 57.2,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 73,
        "reasoning": 74,
        "writing": 70,
        "vision": 75,
        "long_context": 92,
        "multilingual": 78
      },
      "best_for": [
        "Ultra-long-context tasks (10M token window)",
        "Fast self-hosted inference",
        "Free multimodal at scale"
      ],
      "skip_if": [
        "Deep reasoning tasks",
        "When frontier coding quality is needed"
      ],
      "figure_id": "llama-4-scout"
    },
    {
      "id": "phi-4",
      "benchmarks_estimated": [],
      "name": "Phi-4",
      "company": "Microsoft",
      "type": "small-open",
      "license": "MIT",
      "released": "2024-12-12",
      "api_name": "microsoft/phi-4",
      "review_url": null,
      "deprecated": false,
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "context": {
        "max_tokens": 16000,
        "effective_tokens": 14000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": 84.8,
        "humaneval": 82.6,
        "math": 80.4,
        "gpqa_diamond": 56.1,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 70,
        "reasoning": 78,
        "writing": 74,
        "vision": null,
        "long_context": 60,
        "multilingual": 72
      },
      "best_for": [
        "Local inference on consumer hardware",
        "Edge deployment",
        "Reasoning at tiny scale"
      ],
      "skip_if": [
        "Long-context tasks",
        "Production-quality writing or coding"
      ],
      "figure_id": "phi-4"
    },
    {
      "id": "minimax-m3",
      "benchmarks_estimated": [],
      "name": "MiniMax M3",
      "company": "MiniMax",
      "type": "frontier-open",
      "license": "proprietary hosted API",
      "released": "2026-06-01",
      "api_name": "MiniMax-M3",
      "review_url": "pricing/minimax-m3",
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.3,
        "output_per_million": 1.2,
        "cache_input_per_million": 0.06,
        "tier_prices_per_million": [
          0.3,
          1.2,
          0.6,
          2.4,
          0.45,
          1.8,
          0.9,
          3.6
        ],
        "batch_discount": null,
        "input_per_million_over_200k": 0.6,
        "output_per_million_over_200k": 2.4
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 750000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 90,
        "reasoning": 88,
        "writing": 80,
        "vision": 85,
        "long_context": 94,
        "multilingual": 84
      },
      "best_for": [
        "Very low-cost long-context coding agents",
        "1M-context workflows under $1/M input",
        "Multimodal coding and tool-use experiments"
      ],
      "skip_if": [
        "You need a mature Western provider ecosystem",
        "You need official benchmark tables before rollout"
      ],
      "figure_id": "minimax-m3"
    },
    {
      "id": "mistral-large-3",
      "benchmarks_estimated": [],
      "name": "Mistral Large 3",
      "company": "Mistral AI",
      "type": "frontier-open",
      "license": "Apache-2.0",
      "released": "2025-12-02",
      "api_name": "mistral-large-2512",
      "review_url": "articles/mistral-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.5,
        "output_per_million": 1.5,
        "cache_input_per_million": null,
        "batch_discount": null,
        "self_hosted": true
      },
      "context": {
        "max_tokens": 256000,
        "effective_tokens": 200000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 78,
        "reasoning": 79,
        "writing": 78,
        "vision": null,
        "long_context": 76,
        "multilingual": 88
      },
      "best_for": [
        "Apache-licensed production workloads",
        "European data residency",
        "Very cheap inference with decent reasoning"
      ],
      "skip_if": [
        "Coding at frontier quality",
        "Vision or multimodal workflows"
      ],
      "figure_id": "mistral-large-3"
    },
    {
      "id": "mistral-medium-3-5",
      "benchmarks_estimated": [],
      "name": "Mistral Medium 3.5",
      "company": "Mistral AI",
      "type": "frontier-open",
      "license": "Modified MIT",
      "released": "2026-04-28",
      "api_name": "mistral-medium-3-5",
      "review_url": null,
      "deprecated": false,
      "pricing": {
        "input_per_million": 1.5,
        "output_per_million": 7.5,
        "cache_input_per_million": null,
        "self_hosted": true
      },
      "context": {
        "max_tokens": 256000,
        "effective_tokens": 200000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 86,
        "reasoning": 87,
        "writing": 86,
        "vision": 84,
        "long_context": 84,
        "multilingual": 92
      },
      "best_for": [
        "European data residency",
        "Multimodal + reasoning in one model",
        "Self-hosted on 4 GPUs"
      ],
      "skip_if": [
        "You need MoE efficiency for ultra-cheap inference"
      ],
      "figure_id": "mistral-medium-3-5"
    },
    {
      "id": "kimi-k2-6",
      "name": "Kimi K2.6",
      "company": "Moonshot AI",
      "type": "frontier-open",
      "license": "Modified MIT",
      "released": null,
      "api_name": "kimi-k2.6",
      "review_url": "articles/kimi-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.95,
        "output_per_million": 4.0,
        "cache_input_per_million": 0.16,
        "batch_discount": null,
        "self_hosted": true
      },
      "context": {
        "max_tokens": 262144,
        "effective_tokens": 200000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": 80.2,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 90.5,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 80,
        "reasoning": 78,
        "writing": 74,
        "vision": null,
        "long_context": 78,
        "multilingual": 82
      },
      "best_for": [
        "Mid-range open-weight option",
        "Multilingual tasks",
        "Cost-efficient API with self-hosting option"
      ],
      "skip_if": [
        "You need the deepest reasoning or best coding",
        "Vision workloads",
        "You want Moonshot's newest - Kimi K3"
      ],
      "figure_id": "kimi-k2-6",
      "benchmarks_estimated": []
    },
    {
      "id": "kimi-k3",
      "benchmarks_estimated": [],
      "name": "Kimi K3",
      "company": "Moonshot AI",
      "type": "frontier-open",
      "license": "Kimi K3 License",
      "released": "2026-07-16",
      "api_name": "kimi-k3",
      "review_url": "articles/kimi-k3-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 3.0,
        "output_per_million": 15.0,
        "cache_input_per_million": 0.3,
        "batch_discount": null,
        "self_hosted": true
      },
      "context": {
        "max_tokens": 1048576,
        "effective_tokens": null,
        "max_output_tokens": 1048576
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 93.5,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 95,
        "reasoning": 93,
        "writing": 86,
        "vision": 94,
        "long_context": 96,
        "multilingual": 90
      },
      "best_for": [
        "Open-weight long-horizon coding and multimodal research workflows",
        "One-million-token workloads with strong automatic cache economics",
        "Teams that can preserve full reasoning history across agent turns"
      ],
      "skip_if": [
        "Your harness cannot return the complete reasoning and tool-call history unchanged",
        "You need non-thinking mode or low uncached-output cost"
      ],
      "figure_id": "kimi-k3"
    },
    {
      "id": "gpt-5",
      "name": "GPT-5",
      "company": "OpenAI",
      "type": "frontier",
      "license": "proprietary",
      "released": "2025-08-07",
      "api_name": "gpt-5",
      "review_url": "articles/gpt-5-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 1.25,
        "output_per_million": 10.0,
        "cache_input_per_million": 0.125,
        "batch_discount": 0.5
      },
      "context": {
        "max_tokens": 400000,
        "effective_tokens": 300000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": 74.9,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 88,
        "reasoning": 88,
        "writing": 86,
        "vision": 85,
        "long_context": 78,
        "multilingual": 87
      },
      "best_for": [
        "Production workhorse at a rational price",
        "Breadth tasks",
        "Coding agents",
        "The cheapest full GPT-5-class rate still on OpenAI's API"
      ],
      "skip_if": [
        "You are starting fresh - GPT-5.6 Terra is OpenAI's newer general tier",
        "Very large context windows"
      ],
      "figure_id": "gpt-5",
      "benchmarks_estimated": []
    },
    {
      "id": "gpt-5-mini",
      "benchmarks_estimated": [],
      "name": "GPT-5 Mini",
      "company": "OpenAI",
      "type": "small",
      "license": "proprietary",
      "released": "2025-08-07",
      "api_name": "gpt-5-mini",
      "review_url": null,
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.25,
        "output_per_million": 2.0,
        "cache_input_per_million": 0.025,
        "batch_discount": 0.5,
        "batch_input": 0.125,
        "batch_output": 1.0
      },
      "context": {
        "max_tokens": 400000,
        "effective_tokens": 200000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 72,
        "reasoning": 74,
        "writing": 75,
        "vision": 72,
        "long_context": 70,
        "multilingual": 80
      },
      "best_for": [
        "Cheap chat at scale",
        "Simple extraction and routing",
        "High-volume classification"
      ],
      "skip_if": [
        "Production-quality code review",
        "Complex reasoning",
        "Long documents",
        "You can switch models - GPT-5.6 Luna costs less per token and has a 1.05M window"
      ],
      "figure_id": "gpt-5-mini"
    },
    {
      "id": "gpt-5-4",
      "benchmarks_estimated": [],
      "name": "GPT-5.4",
      "company": "OpenAI",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-03-05",
      "api_name": "gpt-5.4",
      "review_url": "articles/gpt-5-4-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 2.5,
        "output_per_million": 15.0,
        "cache_input_per_million": 0.25
      },
      "context": {
        "max_tokens": 1050000,
        "effective_tokens": 600000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "osworld_verified": 75.0,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 90,
        "reasoning": 91,
        "writing": 87,
        "vision": 88,
        "long_context": 85,
        "multilingual": 88
      },
      "best_for": [
        "Financial modeling and spreadsheet work",
        "Computer-use pipelines",
        "Professional knowledge work",
        "Long documents at a mid-tier price"
      ],
      "skip_if": [
        "You want OpenAI's strongest - that is GPT-6 Astra as of September 2026",
        "Cheap volume — GPT-5 Mini costs a tenth",
        "Simple chat — GPT-5 is cheaper",
        "New projects - GPT-5.6 Terra is newer and cheaper on both input and output"
      ],
      "figure_id": "gpt-5-4"
    },
    {
      "id": "gpt-5-5",
      "benchmarks_estimated": [],
      "name": "GPT-5.5",
      "company": "OpenAI",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-04-23",
      "api_name": "gpt-5.5",
      "review_url": "articles/gpt-5-5-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 5.0,
        "output_per_million": 30.0,
        "cache_input_per_million": 0.5,
        "batch_discount": 0.5,
        "batch_input": 2.5,
        "batch_output": 15.0,
        "input_per_million_over_200k": 10.0,
        "output_per_million_over_200k": 45.0,
        "cache_input_per_million_over_200k": 1.0
      },
      "context": {
        "max_tokens": 1050000,
        "effective_tokens": 700000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 93,
        "reasoning": 95,
        "writing": 88,
        "vision": 89,
        "long_context": 86,
        "multilingual": 90
      },
      "best_for": [
        "Frontier math and reasoning",
        "Computer use",
        "Multi-step agents",
        "Vision + reasoning tasks"
      ],
      "skip_if": [
        "New projects - GPT-5.6 Sol is newer and cheaper on both input and output",
        "Quick chat — the price is hard to justify"
      ],
      "figure_id": "gpt-5-5"
    },
    {
      "id": "gpt-5-6",
      "benchmarks_estimated": [],
      "name": "GPT-5.6",
      "company": "OpenAI",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-07-09",
      "api_name": "gpt-5.6-sol",
      "review_url": "articles/gpt-5-6-launch",
      "deprecated": false,
      "pricing": {
        "input_per_million": 4.0,
        "output_per_million": 20.0,
        "cache_input_per_million": 0.4,
        "batch_discount": 0.5,
        "tier_prices_per_million": [
          2.0,
          10.0,
          8.0,
          30.0
        ],
        "batch_input": 2.0,
        "batch_output": 10.0,
        "input_per_million_over_200k": 8.0,
        "output_per_million_over_200k": 30.0,
        "cache_input_per_million_over_200k": 0.8
      },
      "context": {
        "max_tokens": 1050000,
        "effective_tokens": 750000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": 89.8,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": 91.2,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 95,
        "reasoning": 96,
        "writing": 89,
        "vision": 90,
        "long_context": 87,
        "multilingual": 91
      },
      "best_for": [
        "Generally available via OpenAI API since July 9, 2026",
        "New SOTA agentic command-line work (Terminal-Bench)",
        "Frontier math and reasoning",
        "Computer use and multi-step agents",
        "Frontier coding and agents below GPT-6 Astra's price"
      ],
      "skip_if": [
        "Cost-sensitive workloads — Terra is still five times cheaper on input",
        "You need OpenAI's highest ceiling - GPT-6 Astra launched above it on September 3, 2026"
      ],
      "figure_id": "gpt-5-6-sol"
    },
    {
      "id": "glm-5-2",
      "benchmarks_estimated": [],
      "name": "GLM-5.2",
      "company": "Z.AI",
      "type": "frontier-open",
      "license": "open-weight / hosted API",
      "released": "2026-06-16",
      "api_name": "glm-5.2",
      "review_url": "articles/glm-5-2-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 1.4,
        "output_per_million": 4.4,
        "cache_input_per_million": 0.26,
        "batch_discount": null,
        "self_hosted": true
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 750000,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 91,
        "reasoning": 89,
        "writing": 82,
        "vision": null,
        "long_context": 94,
        "multilingual": 86
      },
      "best_for": [
        "Long-horizon coding with 1M context",
        "Teams that want hosted API plus open-weight optionality",
        "Cost-sensitive agent engineering"
      ],
      "skip_if": [
        "You need broad Western enterprise platform support",
        "You need official public benchmark numbers for every metric",
        "New work - GLM-5.3 is newer at the same price"
      ],
      "figure_id": "glm-5-2"
    },
    {
      "id": "glm-5-3",
      "benchmarks_estimated": [],
      "name": "GLM-5.3",
      "company": "Z.AI",
      "type": "frontier",
      "license": "proprietary API for now; Z.AI says open weights will follow after safety hardening",
      "released": "2026-08-18",
      "api_name": "glm-5.3",
      "review_url": "articles/glm-5-3-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 1.4,
        "output_per_million": 4.4,
        "cache_input_per_million": 0.26,
        "batch_discount": null,
        "self_hosted": false
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": null,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 94,
        "reasoning": 91,
        "writing": 83,
        "vision": null,
        "long_context": 95,
        "multilingual": 87
      },
      "best_for": [
        "Long-horizon code and terminal work with 1M context and 128K output",
        "Cost-sensitive agent deployments using compatible API protocols",
        "Defensive software-security analysis with explicit human review"
      ],
      "skip_if": [
        "You need multimodal input",
        "You require downloadable weights before Z.AI completes its safety-hardening release"
      ],
      "figure_id": "glm-5-3"
    },
    {
      "id": "grok-4-3",
      "benchmarks_estimated": [],
      "name": "Grok 4.3",
      "company": "xAI",
      "type": "frontier",
      "license": "proprietary",
      "released": null,
      "api_name": "grok-4.3",
      "review_url": "articles/grok-4-3-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 1.25,
        "output_per_million": 2.5,
        "cache_input_per_million": 0.2,
        "batch_discount": null,
        "batch_input": 1.0,
        "batch_output": 2.0,
        "input_per_million_over_200k": 2.5,
        "output_per_million_over_200k": 5.0,
        "cache_input_per_million_over_200k": 0.4,
        "long_context_threshold_tokens": 200000
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": 800000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 82,
        "reasoning": 84,
        "writing": 80,
        "vision": 78,
        "long_context": 85,
        "multilingual": 75
      },
      "best_for": [
        "Real-time data access via X",
        "Very cheap output tokens",
        "Long-context tasks on a budget"
      ],
      "skip_if": [
        "Coding-first agent work — others edge it out",
        "Non-English-heavy workloads"
      ],
      "figure_id": "grok-4-3"
    },
    {
      "id": "grok-4-5",
      "benchmarks_estimated": [],
      "name": "Grok 4.5",
      "company": "xAI",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-07-08",
      "api_name": "grok-4.5",
      "review_url": "articles/grok-4-5-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 2.0,
        "output_per_million": 6.0,
        "cache_input_per_million": 0.3,
        "batch_discount": null,
        "input_per_million_over_200k": 4.0,
        "output_per_million_over_200k": 12.0,
        "cache_input_per_million_over_200k": 0.6,
        "long_context_threshold_tokens": 200000
      },
      "context": {
        "max_tokens": 500000,
        "effective_tokens": 400000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 93,
        "reasoning": 92,
        "writing": 87,
        "vision": 85,
        "long_context": 82,
        "multilingual": 81
      },
      "best_for": [
        "Coding and agentic workflows on xAI",
        "Lower output cost than GPT-5.5 class models",
        "Search-augmented knowledge work"
      ],
      "skip_if": [
        "You need 1M context — Grok 4.3 has the larger window",
        "You need official benchmark tables before procurement",
        "New work - Grok 4.6 is newer at the same base rate"
      ],
      "figure_id": "grok-4-5"
    },
    {
      "id": "grok-4-6",
      "benchmarks_estimated": [],
      "name": "Grok 4.6",
      "company": "xAI",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-08-12",
      "api_name": "grok-4.6",
      "review_url": "articles/grok-4-6-review",
      "deprecated": false,
      "pricing": {
        "input_per_million": 2.0,
        "output_per_million": 6.0,
        "cache_input_per_million": 0.5,
        "batch_discount": null,
        "input_per_million_over_200k": 4.0,
        "output_per_million_over_200k": 12.0,
        "cache_input_per_million_over_200k": 1.0,
        "long_context_threshold_tokens": 200000
      },
      "context": {
        "max_tokens": 500000,
        "effective_tokens": 400000,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": 94,
        "reasoning": 93,
        "writing": 88,
        "vision": 86,
        "long_context": 82,
        "multilingual": 81
      },
      "best_for": [
        "Long-running coding and agent sessions on xAI",
        "Interactive and visual build work",
        "Same $2/$6 rate as Grok 4.5 with a stronger launch table"
      ],
      "skip_if": [
        "You need 1M context — Grok 4.3 has the larger window",
        "Cache-heavy pipelines — cached input rose to $0.50 from Grok 4.5's $0.30",
        "You need third-party benchmark confirmation before procurement"
      ],
      "figure_id": "grok-4-6"
    },
    {
      "id": "gpt-6-astra",
      "benchmarks_estimated": [],
      "name": "GPT-6 Astra",
      "company": "OpenAI",
      "type": "frontier",
      "license": "proprietary",
      "released": "2026-09-03",
      "api_name": "gpt-6-astra",
      "review_url": null,
      "deprecated": false,
      "pricing": {
        "input_per_million": 10.0,
        "output_per_million": 50.0,
        "cache_input_per_million": 1.0,
        "batch_discount": null,
        "batch_input": 5.0,
        "batch_output": 25.0,
        "free_tier": false
      },
      "context": {
        "max_tokens": 1050000,
        "effective_tokens": null,
        "max_output_tokens": 128000
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": null,
        "reasoning": null,
        "writing": null,
        "vision": null,
        "long_context": null,
        "multilingual": null
      },
      "best_for": [
        "Long end-to-end work OpenAI aims this model at - reasoning, coding, computer use, research",
        "Prompts that need more than the 400K window of the GPT-5.6 family",
        "Teams already on the Responses API, which this model requires for tool calling"
      ],
      "skip_if": [
        "You send custom temperature, top_p or logprobs - this model accepts none of them",
        "You rely on the `none` reasoning-effort level, which it does not support",
        "You call tools through Chat Completions rather than the Responses API",
        "Cost is the constraint: input is 2.5x GPT-5.6 Sol and output is 2.5x its rate",
        "You need benchr-held benchmark figures - OpenAI published none with this release"
      ],
      "figure_id": "gpt-6-astra"
    },
    {
      "id": "qwen-3-8-max",
      "benchmarks_estimated": [],
      "name": "Qwen3.8-Max",
      "company": "Alibaba (Qwen)",
      "type": "frontier",
      "license": "proprietary API; Alibaba said weights would follow, and benchr has verified no licence page",
      "released": "2026-08-03",
      "api_name": "qwen3.8-max",
      "review_url": null,
      "deprecated": false,
      "pricing": {
        "input_per_million": 2.0,
        "output_per_million": 6.0,
        "cache_input_per_million": null,
        "batch_discount": null,
        "batch_input": null,
        "batch_output": null,
        "free_tier": false
      },
      "context": {
        "max_tokens": 1000000,
        "effective_tokens": null,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": null,
        "reasoning": null,
        "writing": null,
        "vision": null,
        "long_context": null,
        "multilingual": null
      },
      "best_for": [
        "Frontier-tier work at a rate well under the US flagships",
        "Text and vision in one model, on a context Alibaba describes as up to 1M tokens",
        "Workloads already running in Alibaba Cloud Model Studio"
      ],
      "skip_if": [
        "You need published benchmark scores - Alibaba gave arena placements, not numbers",
        "Your licence review needs a named licence, which no official page states",
        "You are outside the Singapore region, where the published rate may not apply"
      ],
      "figure_id": "qwen-3-8-max"
    },
    {
      "id": "qwen-3-8-flash",
      "benchmarks_estimated": [],
      "name": "Qwen3.8-Flash",
      "company": "Alibaba (Qwen)",
      "type": "mid",
      "license": "hosted API; the announcement calls it open-weight but names no licence, and benchr has verified none",
      "released": "2026-08-27",
      "api_name": "qwen3.8-flash",
      "review_url": null,
      "deprecated": false,
      "pricing": {
        "input_per_million": 0.15,
        "output_per_million": 0.47,
        "cache_input_per_million": null,
        "batch_discount": null,
        "batch_input": null,
        "batch_output": null,
        "free_tier": false
      },
      "context": {
        "max_tokens": 262144,
        "effective_tokens": null,
        "max_output_tokens": null
      },
      "benchmarks": {
        "swe_bench_verified": null,
        "lmsys_arena": null,
        "mmlu": null,
        "humaneval": null,
        "math": null,
        "gpqa_diamond": null,
        "arc_agi_2": null
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null
      },
      "capabilities": {
        "coding": null,
        "reasoning": null,
        "writing": null,
        "vision": null,
        "long_context": null,
        "multilingual": null
      },
      "best_for": [
        "High-volume multimodal work - among the lowest published rates benchr records",
        "Agent loops that fit the native 262K window",
        "Cost-first prototyping before committing to a frontier tier"
      ],
      "skip_if": [
        "You need the 1M context the announcement mentions - benchr records the native 262K",
        "You need scores: Alibaba named six benchmarks and published no result for any of them",
        "Your licence review needs a named licence"
      ],
      "figure_id": "qwen-3-8-flash"
    }
  ],
  "dimensions": [
    {
      "id": "pricing",
      "label": "Pricing",
      "icon": "$",
      "sub_dimensions": [
        {
          "id": "input_per_million",
          "label": "Input / 1M tokens",
          "unit": "$",
          "format": "currency",
          "lower_better": true
        },
        {
          "id": "output_per_million",
          "label": "Output / 1M tokens",
          "unit": "$",
          "format": "currency",
          "lower_better": true
        },
        {
          "id": "cache_input_per_million",
          "label": "Cached input / 1M tokens",
          "unit": "$",
          "format": "currency",
          "lower_better": true
        }
      ]
    },
    {
      "id": "context",
      "label": "Context window",
      "icon": "↔",
      "sub_dimensions": [
        {
          "id": "max_tokens",
          "label": "Max tokens",
          "unit": "tokens",
          "format": "number",
          "lower_better": false
        },
        {
          "id": "effective_tokens",
          "label": "Effective retrieval zone",
          "unit": "tokens",
          "format": "number",
          "lower_better": false
        },
        {
          "id": "max_output_tokens",
          "label": "Max output tokens",
          "unit": "tokens",
          "format": "number",
          "lower_better": false
        }
      ]
    },
    {
      "id": "benchmarks",
      "label": "Benchmarks",
      "icon": "★",
      "sub_dimensions": [
        {
          "id": "swe_bench_verified",
          "label": "SWE-bench Verified",
          "unit": "%",
          "format": "percent",
          "lower_better": false
        },
        {
          "id": "lmsys_arena",
          "label": "LMSYS Arena",
          "unit": "score",
          "format": "number",
          "lower_better": false
        },
        {
          "id": "mmlu",
          "label": "MMLU",
          "unit": "%",
          "format": "percent",
          "lower_better": false
        },
        {
          "id": "humaneval",
          "label": "HumanEval",
          "unit": "%",
          "format": "percent",
          "lower_better": false
        },
        {
          "id": "math",
          "label": "MATH",
          "unit": "%",
          "format": "percent",
          "lower_better": false
        },
        {
          "id": "gpqa_diamond",
          "label": "GPQA Diamond",
          "unit": "%",
          "format": "percent",
          "lower_better": false
        },
        {
          "id": "arc_agi_2",
          "label": "ARC-AGI 2",
          "unit": "%",
          "format": "percent",
          "lower_better": false
        }
      ]
    },
    {
      "id": "latency",
      "label": "Speed",
      "icon": "⚡",
      "sub_dimensions": [
        {
          "id": "first_token_ms",
          "label": "First token (ms)",
          "unit": "ms",
          "format": "number",
          "lower_better": true
        },
        {
          "id": "tokens_per_second",
          "label": "Tokens / second",
          "unit": "tok/s",
          "format": "number",
          "lower_better": false
        }
      ]
    },
    {
      "id": "capabilities",
      "label": "Capabilities (0–10)",
      "icon": "◉",
      "sub_dimensions": [
        {
          "id": "coding",
          "label": "Coding",
          "unit": "/10",
          "format": "number",
          "lower_better": false
        },
        {
          "id": "reasoning",
          "label": "Reasoning",
          "unit": "/10",
          "format": "number",
          "lower_better": false
        },
        {
          "id": "writing",
          "label": "Writing",
          "unit": "/10",
          "format": "number",
          "lower_better": false
        },
        {
          "id": "vision",
          "label": "Vision",
          "unit": "/10",
          "format": "number",
          "lower_better": false
        },
        {
          "id": "long_context",
          "label": "Long context",
          "unit": "/10",
          "format": "number",
          "lower_better": false
        },
        {
          "id": "multilingual",
          "label": "Multilingual",
          "unit": "/10",
          "format": "number",
          "lower_better": false
        }
      ]
    }
  ]
}
