{
  "count": 164,
  "providers": [
    "openai",
    "anthropic",
    "google",
    "deepseek",
    "meta",
    "xai",
    "mistral",
    "qwen",
    "cohere",
    "amazon",
    "microsoft",
    "moonshot",
    "zai",
    "minimax",
    "perplexity",
    "baidu",
    "nvidia",
    "xiaomi",
    "bytedance",
    "stepfun",
    "meituan",
    "tencent",
    "upstage"
  ],
  "data": [
    {
      "id": "gpt-6-astra",
      "name": "GPT-6 Astra",
      "providerId": "openai",
      "releaseDate": "2026-09-03",
      "contextWindow": 1050000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 10,
        "outputPerMTokens": 50,
        "cachedInputPerMTokens": 1,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-6-astra",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 97.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 93.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 95.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 75.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AutomationBench",
          "score": 41.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-6-astra/",
          "measuredAt": "2026-09-03"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 96,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-6-astra/",
          "measuredAt": "2026-09-03"
        },
        {
          "benchmark": "Terminal-Bench Science 0.1",
          "score": 64.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-6-astra/",
          "measuredAt": "2026-09-03"
        }
      ]
    },
    {
      "id": "gpt-5-6-sol",
      "name": "GPT-5.6 Sol",
      "providerId": "openai",
      "releaseDate": "2026-07-09",
      "contextWindow": 1050000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 4,
        "outputPerMTokens": 20,
        "cachedInputPerMTokens": 0.4,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 82.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 89.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 93.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 69.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 72.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 94.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 64.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        }
      ]
    },
    {
      "id": "gpt-5-6-terra",
      "name": "GPT-5.6 Terra",
      "providerId": "openai",
      "releaseDate": "2026-07-09",
      "contextWindow": 1050000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 12,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 70.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 86,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 93.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 99.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 43.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 69.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 92.9,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 63.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        }
      ]
    },
    {
      "id": "gpt-5-6-luna",
      "name": "GPT-5.6 Luna",
      "providerId": "openai",
      "releaseDate": "2026-07-09",
      "contextWindow": 1050000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.2,
        "outputPerMTokens": 1.2,
        "cachedInputPerMTokens": 0.02,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 61,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 82.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 91.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 98.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 41,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 67.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 92.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 62.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        }
      ]
    },
    {
      "id": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "providerId": "anthropic",
      "releaseDate": "2026-09-01",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 10,
        "outputPerMTokens": 50,
        "cachedInputPerMTokens": 0.25,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 87.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 90.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 70.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "CursorBench 3.2.0",
          "score": 73.4,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "GDPval-AA v2",
          "score": 1853,
          "unit": "elo",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 60.9,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "Terminal-Bench Science 0.1",
          "score": 52.6,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        }
      ]
    },
    {
      "id": "claude-opus-5",
      "name": "Claude Opus 5",
      "providerId": "anthropic",
      "releaseDate": "2026-07-24",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 5,
        "outputPerMTokens": 25,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 73.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 85.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 93.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 98.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 59.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "CursorBench 3.2.0",
          "score": 70,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "GDPval-AA v2",
          "score": 1824,
          "unit": "elo",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 56.6,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "Terminal-Bench Science 0.1",
          "score": 29,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        }
      ]
    },
    {
      "id": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "providerId": "anthropic",
      "releaseDate": "2026-06-30",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 10,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 29.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 65.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 94.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 33.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 43.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/283ef97c476cf442c91d9a37d5b214242a55bb92/Claude%20Sonnet%205%20System%20Card.pdf",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 57.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/283ef97c476cf442c91d9a37d5b214242a55bb92/Claude%20Sonnet%205%20System%20Card.pdf",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 85.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/283ef97c476cf442c91d9a37d5b214242a55bb92/Claude%20Sonnet%205%20System%20Card.pdf",
          "measuredAt": "2026-06-30"
        }
      ]
    },
    {
      "id": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "providerId": "anthropic",
      "releaseDate": "2025-10-15",
      "contextWindow": 200000,
      "maxOutput": 64000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1,
        "outputPerMTokens": 5,
        "cachedInputPerMTokens": 0.1,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 71.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 66.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 13.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 73.3,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-haiku-4-5",
          "measuredAt": "2025-10-15"
        }
      ]
    },
    {
      "id": "gemini-3-8-flash",
      "name": "Gemini 3.8 Flash",
      "providerId": "google",
      "releaseDate": "2026-09-02",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 0.75,
        "outputPerMTokens": 3.75,
        "cachedInputPerMTokens": 0.075,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 22,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 68.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 95.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 98.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 69.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 95.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-6-astra/",
          "measuredAt": "2026-09-03"
        },
        {
          "benchmark": "CharXiv Reasoning (no tools)",
          "score": 86.2,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 73.7,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        },
        {
          "benchmark": "GDPVal-AA v2",
          "score": 1545,
          "unit": "elo",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        },
        {
          "benchmark": "Harvey's Legal Agent Benchmark",
          "score": 10,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        },
        {
          "benchmark": "HLE-Verified",
          "score": 54.9,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        },
        {
          "benchmark": "LABBench2",
          "score": 86.2,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        },
        {
          "benchmark": "OSWorld-2.0",
          "score": 59,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        },
        {
          "benchmark": "Terminal-bench 2.1",
          "score": 89.4,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        },
        {
          "benchmark": "Terminal-bench 4.0",
          "score": 19.1,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        },
        {
          "benchmark": "Vals Finance Agent v2",
          "score": 61.4,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
          "measuredAt": "2026-09-02"
        }
      ]
    },
    {
      "id": "gemini-3-1-pro-preview",
      "name": "Gemini 3.1 Pro Preview",
      "providerId": "google",
      "releaseDate": "2026-02-19",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 12,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 98.3,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 26.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 59.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 94.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 95.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 73.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1480.1,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 94.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 54.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-5-6/",
          "measuredAt": "2026-07-09"
        },
        {
          "benchmark": "ARC-AGI-2",
          "score": 77.1,
          "unit": "percent",
          "sourceUrl": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-1-pro/",
          "measuredAt": "2026-02-19"
        }
      ]
    },
    {
      "id": "gemini-3-5-flash-lite",
      "name": "Gemini 3.5 Flash-Lite",
      "providerId": "google",
      "releaseDate": "2026-07-21",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 0.3,
        "outputPerMTokens": 2.5,
        "cachedInputPerMTokens": 0.03,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 0,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 26,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 83.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 71.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1435.5,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "deepseek-v4-1-flash",
      "name": "DeepSeek V4.1 Flash",
      "providerId": "deepseek",
      "releaseDate": "2026-09-10",
      "contextWindow": 1000000,
      "maxOutput": 384000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.14,
        "outputPerMTokens": 0.28,
        "cachedInputPerMTokens": 0.0028,
        "sourceUrl": "https://api-docs.deepseek.com/quick_start/pricing/",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "DeepSWE v1.1",
          "score": 74.2,
          "unit": "percent",
          "sourceUrl": "https://api-docs.deepseek.com/updates/",
          "measuredAt": "2026-09-10"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 90.9,
          "unit": "percent",
          "sourceUrl": "https://api-docs.deepseek.com/updates/",
          "measuredAt": "2026-09-10"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 90.6,
          "unit": "percent",
          "sourceUrl": "https://api-docs.deepseek.com/updates/",
          "measuredAt": "2026-09-10"
        }
      ]
    },
    {
      "id": "deepseek-v4-pro",
      "name": "DeepSeek V4 Pro",
      "providerId": "deepseek",
      "releaseDate": "2026-08-13",
      "contextWindow": 1000000,
      "maxOutput": 384000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.435,
        "outputPerMTokens": 0.87,
        "cachedInputPerMTokens": 0.003625,
        "sourceUrl": "https://api-docs.deepseek.com/quick_start/pricing/",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 96.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 2.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 45.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 96.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 47,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 77.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1450.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 62.7,
          "unit": "percent",
          "sourceUrl": "https://api-docs.deepseek.com/updates/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 60,
          "unit": "percent",
          "sourceUrl": "https://api-docs.deepseek.com/updates/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 87.9,
          "unit": "percent",
          "sourceUrl": "https://api-docs.deepseek.com/updates/",
          "measuredAt": "2026-08-13"
        }
      ]
    },
    {
      "id": "llama-4-maverick",
      "name": "Llama 4 Maverick",
      "providerId": "meta",
      "releaseDate": "2025-04-05",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/",
        "updatedAt": "2025-04-05"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond",
          "score": 69.8,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
          "measuredAt": "2025-04-05"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 80.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
          "measuredAt": "2025-04-05"
        }
      ]
    },
    {
      "id": "llama-4-scout",
      "name": "Llama 4 Scout",
      "providerId": "meta",
      "releaseDate": "2025-04-05",
      "contextWindow": 10000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/",
        "updatedAt": "2025-04-05"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond",
          "score": 57.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
          "measuredAt": "2025-04-05"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 74.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
          "measuredAt": "2025-04-05"
        }
      ]
    },
    {
      "id": "grok-4-6",
      "name": "Grok 4.6",
      "providerId": "xai",
      "releaseDate": "2026-08-12",
      "contextWindow": 500000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 6,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://docs.x.ai/developers/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 31.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 66,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 94,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 99.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 49.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AA-Briefcase",
          "score": 1577,
          "unit": "score",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "APEX-Agents",
          "score": 57.5,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "APEX-SWE",
          "score": 56.4,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "Artificial Analysis Intelligence Index",
          "score": 61,
          "unit": "score",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "CursorBench 3.2",
          "score": 69.9,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 65.9,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "FrontierCode v1.1 (Extended)",
          "score": 61.3,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "GDPVal-AA v2",
          "score": 1753,
          "unit": "score",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "Harvey LAB (Vals)",
          "score": 15.8,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "Terminal-Bench 3.0",
          "score": 26,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-6",
          "measuredAt": "2026-08-12"
        }
      ]
    },
    {
      "id": "grok-4-3",
      "name": "Grok 4.3",
      "providerId": "xai",
      "releaseDate": null,
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.25,
        "outputPerMTokens": 2.5,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://docs.x.ai/developers/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 14.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 42.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 88.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 93.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 33.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1397.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "mistral-medium-3-5",
      "name": "Mistral Medium 3.5",
      "providerId": "mistral",
      "releaseDate": "2026-04-28",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.5,
        "outputPerMTokens": 7.5,
        "cachedInputPerMTokens": 0.15,
        "sourceUrl": "https://docs.mistral.ai/inference/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1420.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 77.6,
          "unit": "percent",
          "sourceUrl": "https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5/",
          "measuredAt": "2026-05-22"
        },
        {
          "benchmark": "Tau3 Telecom",
          "score": 91.4,
          "unit": "score",
          "sourceUrl": "https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5/",
          "measuredAt": "2026-05-22"
        }
      ]
    },
    {
      "id": "mistral-small-4",
      "name": "Mistral Small 4",
      "providerId": "mistral",
      "releaseDate": "2026-03-16",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.15,
        "outputPerMTokens": 0.6,
        "cachedInputPerMTokens": 0.015,
        "sourceUrl": "https://docs.mistral.ai/inference/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "Artificial Analysis Long Context Reasoning",
          "score": 0.72,
          "unit": "score",
          "sourceUrl": "https://mistral.ai/news/mistral-small-4/",
          "measuredAt": "2026-03-16"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 71.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Mistral-Small-4-119B-2603",
          "measuredAt": "2026-03-16"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 78,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Mistral-Small-4-119B-2603",
          "measuredAt": "2026-03-16"
        }
      ]
    },
    {
      "id": "qwen3-8-max",
      "name": "Qwen3.8-Max",
      "providerId": "qwen",
      "releaseDate": "2026-08-02",
      "contextWindow": 1000000,
      "maxOutput": 131072,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.65,
        "outputPerMTokens": 4.951,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 46.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 74.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 92.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 99.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 45.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1480.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 92.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 67.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 86.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
          "measuredAt": "2026-08-12"
        }
      ]
    },
    {
      "id": "qwen3-8-27b",
      "name": "Qwen3.8 27B",
      "providerId": "qwen",
      "releaseDate": "2026-08-14",
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.424,
        "outputPerMTokens": 1.696,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1439.3,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 89.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-27B",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 61.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-27B",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 73,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-27B",
          "measuredAt": "2026-08-14"
        }
      ]
    },
    {
      "id": "gpt-5-5",
      "name": "GPT-5.5",
      "providerId": "openai",
      "releaseDate": "2026-04-23",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 5,
        "outputPerMTokens": 30,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://developers.openai.com/api/docs/models",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 72.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 85.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 84.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 63,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1465.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "ARC-AGI-1 (Verified)",
          "score": 95,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 85,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "BrowseComp",
          "score": 84.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "FrontierMath Tier 1-3",
          "score": 51.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "FrontierMath Tier 4",
          "score": 35.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "GDPval (wins or ties)",
          "score": 84.9,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 75.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "MMMU-Pro (no tools)",
          "score": 81.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 83.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 78.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 58.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 98,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 82.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "Toolathlon",
          "score": 55.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        }
      ]
    },
    {
      "id": "gpt-5-5-pro",
      "name": "GPT-5.5 Pro",
      "providerId": "openai",
      "releaseDate": "2026-04-23",
      "contextWindow": 1050000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 30,
        "outputPerMTokens": 180,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.5-pro",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 78,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 87.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "BrowseComp",
          "score": 90.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "FrontierMath Tier 1-3",
          "score": 52.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "FrontierMath Tier 4",
          "score": 39.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "GDPval (wins or ties)",
          "score": 82.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 43.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 57.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
          "measuredAt": "2026-04-23"
        }
      ]
    },
    {
      "id": "gpt-5-4",
      "name": "GPT-5.4",
      "providerId": "openai",
      "releaseDate": "2026-03-05",
      "contextWindow": 1050000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 2.5,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.25,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.4",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 99.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 49,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 78.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 93.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 97.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 45.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 76.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1452.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "BrowseComp",
          "score": 82.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "GDPval (wins or ties)",
          "score": 83,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 92.8,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 39.8,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 52.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 67.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "MMMU-Pro (no tools)",
          "score": 81.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 82.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 75,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 57.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 98.9,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 75.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "Toolathlon",
          "score": 54.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        }
      ]
    },
    {
      "id": "gpt-5-4-pro",
      "name": "GPT-5.4 Pro",
      "providerId": "openai",
      "releaseDate": "2026-03-05",
      "contextWindow": 1050000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 30,
        "outputPerMTokens": 180,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.4-pro",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 58.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 82.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 94.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 46.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "ARC-AGI-1 (Verified)",
          "score": 94.5,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 83.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "BrowseComp",
          "score": 89.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "FrontierMath Tier 1-3",
          "score": 50,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "FrontierMath Tier 4",
          "score": 38,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "GDPval (wins or ties)",
          "score": 82,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 94.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 42.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 58.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
          "measuredAt": "2026-03-05"
        }
      ]
    },
    {
      "id": "gpt-5-4-mini",
      "name": "GPT-5.4 mini",
      "providerId": "openai",
      "releaseDate": "2026-03-17",
      "contextWindow": 400000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.75,
        "outputPerMTokens": 4.5,
        "cachedInputPerMTokens": 0.075,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.4-mini",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 9.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 51.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 86.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 88.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 29.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 88,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 72.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 54.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 60,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        },
        {
          "benchmark": "Toolathlon",
          "score": 42.9,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        }
      ]
    },
    {
      "id": "gpt-5-4-nano",
      "name": "GPT-5.4 nano",
      "providerId": "openai",
      "releaseDate": "2026-03-17",
      "contextWindow": 400000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.2,
        "outputPerMTokens": 1.25,
        "cachedInputPerMTokens": 0.02,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.4-nano",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 12.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 44.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 78.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 87.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 11.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 82.8,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 39,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 52.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 46.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        },
        {
          "benchmark": "Toolathlon",
          "score": 35.5,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        }
      ]
    },
    {
      "id": "gpt-5-2",
      "name": "GPT-5.2",
      "providerId": "openai",
      "releaseDate": "2025-12-11",
      "contextWindow": 400000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.75,
        "outputPerMTokens": 14,
        "cachedInputPerMTokens": 0.175,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.2",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 98.3,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 31.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 67.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 91.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 96.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 37.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 73.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1412.4,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AIME 2025 (no tools)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "ARC-AGI-1 (Verified)",
          "score": 86.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 52.9,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "GDPval (wins or ties)",
          "score": 70.9,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 92.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 55.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 80,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        }
      ]
    },
    {
      "id": "gpt-5-2-pro",
      "name": "GPT-5.2 Pro",
      "providerId": "openai",
      "releaseDate": "2025-12-11",
      "contextWindow": 400000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 21,
        "outputPerMTokens": 168,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.2-pro",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 46,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 74,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2025 (no tools)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "ARC-AGI-1 (Verified)",
          "score": 90.5,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 54.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "BrowseComp",
          "score": 77.9,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "GDPval (wins or ties)",
          "score": 74.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 93.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 36.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 50,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
          "measuredAt": "2025-12-11"
        }
      ]
    },
    {
      "id": "o3-pro",
      "name": "o3-pro",
      "providerId": "openai",
      "releaseDate": "2025-06-10",
      "contextWindow": 200000,
      "maxOutput": 100000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 20,
        "outputPerMTokens": 80,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://developers.openai.com/api/docs/models/o3-pro",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "gpt-oss-120b",
      "name": "gpt-oss-120b",
      "providerId": "openai",
      "releaseDate": "2025-08-05",
      "contextWindow": 131072,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-oss-120b",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 90,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 75.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 88.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1365.9,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 80.1,
          "unit": "percent",
          "sourceUrl": "https://cdn.openai.com/pdf/419b6906-9da6-406c-a19d-1bb078ac7637/oai_gpt-oss_model_card.pdf",
          "measuredAt": "2025-08-05"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 62.4,
          "unit": "percent",
          "sourceUrl": "https://cdn.openai.com/pdf/419b6906-9da6-406c-a19d-1bb078ac7637/oai_gpt-oss_model_card.pdf",
          "measuredAt": "2025-08-05"
        }
      ]
    },
    {
      "id": "gpt-oss-20b",
      "name": "gpt-oss-20b",
      "providerId": "openai",
      "releaseDate": "2025-08-05",
      "contextWindow": 131072,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-oss-20b",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 89.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 60.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 65.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1287.3,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 71.5,
          "unit": "percent",
          "sourceUrl": "https://cdn.openai.com/pdf/419b6906-9da6-406c-a19d-1bb078ac7637/oai_gpt-oss_model_card.pdf",
          "measuredAt": "2025-08-05"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 60.7,
          "unit": "percent",
          "sourceUrl": "https://cdn.openai.com/pdf/419b6906-9da6-406c-a19d-1bb078ac7637/oai_gpt-oss_model_card.pdf",
          "measuredAt": "2025-08-05"
        }
      ]
    },
    {
      "id": "claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "providerId": "anthropic",
      "releaseDate": "2026-05-28",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 5,
        "outputPerMTokens": 25,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://platform.claude.com/docs/en/models/opus-4-8/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 56.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 80,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 91,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 98.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 53,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1452.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "BrowseComp",
          "score": 84.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 93.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 49.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 57.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 83.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Multilingual",
          "score": 84.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Multimodal",
          "score": 38.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 69.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 88.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 74.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        }
      ]
    },
    {
      "id": "claude-opus-4-7",
      "name": "Claude Opus 4.7",
      "providerId": "anthropic",
      "releaseDate": "2026-04-16",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 5,
        "outputPerMTokens": 25,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://platform.claude.com/docs/en/models/opus-4-7/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 95.8,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 31.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 70.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 97.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 51.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 83.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1483.4,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "BrowseComp",
          "score": 79.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 94.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 46.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 54.7,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 82.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Multilingual",
          "score": 80.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Multimodal",
          "score": 34.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 64.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 87.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 66.1,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        }
      ]
    },
    {
      "id": "claude-opus-4-6",
      "name": "Claude Opus 4.6",
      "providerId": "anthropic",
      "releaseDate": "2026-02-05",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 5,
        "outputPerMTokens": 25,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://platform.claude.com/docs/en/models/opus-4-6/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 96.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 26.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 66,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 94.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 47,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 78.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1497.5,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 68.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 91.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 59.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "MMMLU",
          "score": 91.1,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "MMMU-Pro (no tools)",
          "score": 73.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 77.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 72.7,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 80.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "Tau2-bench Retail",
          "score": 91.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 99.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 65.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        }
      ]
    },
    {
      "id": "claude-sonnet-4-6",
      "name": "Claude Sonnet 4.6",
      "providerId": "anthropic",
      "releaseDate": "2026-02-17",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://platform.claude.com/docs/en/models/sonnet-4-6/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 87.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 85.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 35.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 75.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1458.3,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 58.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "GDPval-AA",
          "score": 1633,
          "unit": "elo",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 89.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 33.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 49,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 61.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMLU",
          "score": 89.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMU-Pro (no tools)",
          "score": 74.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 75.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 72.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 79.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Tau2-bench Retail",
          "score": 91.7,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 97.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 59.1,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        }
      ]
    },
    {
      "id": "claude-sonnet-4-5",
      "name": "Claude Sonnet 4.5",
      "providerId": "anthropic",
      "releaseDate": "2025-09-29",
      "contextWindow": 200000,
      "maxOutput": 64000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://www.anthropic.com/news/claude-sonnet-4-5",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 84.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 2.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 23.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 82.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 77.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 30.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 71.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 13.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "GDPval-AA",
          "score": 1276,
          "unit": "elo",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 83.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 17.7,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 33.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 43.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMLU",
          "score": 89.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMU-Pro (no tools)",
          "score": 63.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 68.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 61.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 77.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Tau2-bench Retail",
          "score": 86.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 98,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 51,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        }
      ]
    },
    {
      "id": "gemma-4-31b",
      "name": "Gemma 4 31B",
      "providerId": "google",
      "releaseDate": "2026-04-02",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1441.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AIME 2026 (no tools)",
          "score": 89.2,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 84.3,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 19.5,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 80,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 85.2,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        }
      ]
    },
    {
      "id": "gemma-4-26b-a4b",
      "name": "Gemma 4 26B A4B",
      "providerId": "google",
      "releaseDate": "2026-04-02",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1434.5,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AIME 2026 (no tools)",
          "score": 88.3,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 82.3,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 8.7,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 77.1,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 82.6,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        }
      ]
    },
    {
      "id": "gemma-4-12b",
      "name": "Gemma 4 12B Unified",
      "providerId": "google",
      "releaseDate": "2026-06-03",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (no tools)",
          "score": 77.5,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-06-03"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 78.8,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-06-03"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 5.2,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-06-03"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 72,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-06-03"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 77.2,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-06-03"
        }
      ]
    },
    {
      "id": "gemma-4-e4b",
      "name": "Gemma 4 E4B",
      "providerId": "google",
      "releaseDate": "2026-04-02",
      "contextWindow": 128000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (no tools)",
          "score": 42.5,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 58.6,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 52,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 69.4,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        }
      ]
    },
    {
      "id": "gemma-4-e2b",
      "name": "Gemma 4 E2B",
      "providerId": "google",
      "releaseDate": "2026-04-02",
      "contextWindow": 128000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (no tools)",
          "score": 37.5,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 43.4,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 44,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 60,
          "unit": "percent",
          "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
          "measuredAt": "2026-04-02"
        }
      ]
    },
    {
      "id": "gpt-5-1",
      "name": "GPT-5.1",
      "providerId": "openai",
      "releaseDate": "2025-11-13",
      "contextWindow": 400000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.25,
        "outputPerMTokens": 10,
        "cachedInputPerMTokens": 0.125,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.1",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 94.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 87.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 88.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 48,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 68,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1422.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "gpt-5",
      "name": "GPT-5",
      "providerId": "openai",
      "releaseDate": "2025-08-07",
      "contextWindow": 400000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.25,
        "outputPerMTokens": 10,
        "cachedInputPerMTokens": 0.125,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 95,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 22,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 55.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 86.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 91.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 50.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 73.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2025 (no tools)",
          "score": 94.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5/",
          "measuredAt": "2025-08-07"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 74.9,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5/",
          "measuredAt": "2025-08-07"
        }
      ]
    },
    {
      "id": "gpt-5-mini",
      "name": "GPT-5 mini",
      "providerId": "openai",
      "releaseDate": "2025-08-07",
      "contextWindow": 400000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.25,
        "outputPerMTokens": 2,
        "cachedInputPerMTokens": 0.025,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5-mini",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 87.5,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 12.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 46.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 75,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 86.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 21.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 64.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 81.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
          "measuredAt": "2026-03-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 71,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-for-developers/",
          "measuredAt": "2025-08-07"
        }
      ]
    },
    {
      "id": "gpt-5-nano",
      "name": "GPT-5 nano",
      "providerId": "openai",
      "releaseDate": "2025-08-07",
      "contextWindow": 400000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.05,
        "outputPerMTokens": 0.4,
        "cachedInputPerMTokens": 0.005,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5-nano",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 85,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 2.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 20,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 69.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 81.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 11.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 54.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-for-developers/",
          "measuredAt": "2025-08-07"
        }
      ]
    },
    {
      "id": "gpt-5-pro",
      "name": "GPT-5 Pro",
      "providerId": "openai",
      "releaseDate": "2025-10-06",
      "contextWindow": 400000,
      "maxOutput": 272000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 15,
        "outputPerMTokens": 120,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5-pro",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 19.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 55.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 88.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5/",
          "measuredAt": "2025-08-07"
        }
      ]
    },
    {
      "id": "o3",
      "name": "o3",
      "providerId": "openai",
      "releaseDate": "2025-04-16",
      "contextWindow": 200000,
      "maxOutput": 100000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 8,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://developers.openai.com/api/docs/models/o3",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 89.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 33.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 81.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 84.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 49.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 62.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 69.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/introducing-gpt-5-for-developers/",
          "measuredAt": "2025-08-07"
        }
      ]
    },
    {
      "id": "gpt-4-1",
      "name": "GPT-4.1",
      "providerId": "openai",
      "releaseDate": "2025-04-14",
      "contextWindow": 1047576,
      "maxOutput": 32768,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 8,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-4.1",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 66.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 38.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 31.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 48.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2024",
          "score": 48.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 66.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "MMLU",
          "score": 90.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 54.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "Video-MME (long, no subtitles)",
          "score": 72,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        }
      ]
    },
    {
      "id": "gpt-4-1-mini",
      "name": "GPT-4.1 mini",
      "providerId": "openai",
      "releaseDate": "2025-04-14",
      "contextWindow": 1047576,
      "maxOutput": 32768,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.4,
        "outputPerMTokens": 1.6,
        "cachedInputPerMTokens": 0.1,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-4.1-mini",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 6.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 65.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 44.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 12.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2024",
          "score": 49.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 65,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "MMLU",
          "score": 87.5,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        }
      ]
    },
    {
      "id": "gpt-4-1-nano",
      "name": "GPT-4.1 nano",
      "providerId": "openai",
      "releaseDate": "2025-04-14",
      "contextWindow": 1047576,
      "maxOutput": 32768,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.1,
        "outputPerMTokens": 0.4,
        "cachedInputPerMTokens": 0.025,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-4.1-nano",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 48.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 28.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "Aider Polyglot",
          "score": 9.8,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "AIME 2024",
          "score": 29.4,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 50.3,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "MMLU",
          "score": 80.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        }
      ]
    },
    {
      "id": "gpt-4o",
      "name": "GPT-4o",
      "providerId": "openai",
      "releaseDate": "2024-05-13",
      "contextWindow": 128000,
      "maxOutput": 16384,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 2.5,
        "outputPerMTokens": 10,
        "cachedInputPerMTokens": 1.25,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-4o",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 11.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 0.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 49.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 6.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 26,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 31,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2024",
          "score": 13.1,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 46,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "MMLU",
          "score": 85.7,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        }
      ]
    },
    {
      "id": "gpt-4o-mini",
      "name": "GPT-4o mini",
      "providerId": "openai",
      "releaseDate": "2024-07-18",
      "contextWindow": 128000,
      "maxOutput": 16384,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.15,
        "outputPerMTokens": 0.6,
        "cachedInputPerMTokens": 0.075,
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-4o-mini",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 0.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 37.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 6.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 8.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2024",
          "score": 8.6,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 40.2,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        },
        {
          "benchmark": "MMLU",
          "score": 82,
          "unit": "percent",
          "sourceUrl": "https://openai.com/index/gpt-4-1/",
          "measuredAt": "2025-04-14"
        }
      ]
    },
    {
      "id": "claude-opus-4-1",
      "name": "Claude Opus 4.1",
      "providerId": "anthropic",
      "releaseDate": "2025-08-05",
      "contextWindow": 200000,
      "maxOutput": 32000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 15,
        "outputPerMTokens": 75,
        "cachedInputPerMTokens": 1.5,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 2.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 12.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 77.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 68.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 73.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 74.5,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-opus-4-1",
          "measuredAt": "2025-08-05"
        }
      ]
    },
    {
      "id": "claude-opus-4",
      "name": "Claude Opus 4",
      "providerId": "anthropic",
      "releaseDate": "2025-05-22",
      "contextWindow": 200000,
      "maxOutput": 32000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 15,
        "outputPerMTokens": 75,
        "cachedInputPerMTokens": 1.5,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 76.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 64.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 70.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2024 (no extended thinking)",
          "score": 33.9,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 72.5,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        },
        {
          "benchmark": "Terminal-bench",
          "score": 43.2,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        }
      ]
    },
    {
      "id": "claude-sonnet-4",
      "name": "Claude Sonnet 4",
      "providerId": "anthropic",
      "releaseDate": "2025-05-22",
      "contextWindow": 200000,
      "maxOutput": 64000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 79.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 71.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2024 (no extended thinking)",
          "score": 33.1,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 72.7,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        }
      ]
    },
    {
      "id": "claude-sonnet-3-7",
      "name": "Claude 3.7 Sonnet",
      "providerId": "anthropic",
      "releaseDate": "2025-02-24",
      "contextWindow": 200000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://www.anthropic.com/news/claude-3-7-sonnet",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 49.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 79.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 57.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 61,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 70.3,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-3-7-sonnet",
          "measuredAt": "2025-02-24"
        }
      ]
    },
    {
      "id": "claude-sonnet-3-5",
      "name": "Claude 3.5 Sonnet",
      "providerId": "anthropic",
      "releaseDate": "2024-10-22",
      "contextWindow": 200000,
      "maxOutput": 8192,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://www.anthropic.com/news/claude-3-5-sonnet",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 3.3,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 55.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 8.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 49,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-3-5-sonnet",
          "measuredAt": "2024-10-22"
        }
      ]
    },
    {
      "id": "gemini-2-5-pro",
      "name": "Gemini 2.5 Pro",
      "providerId": "google",
      "releaseDate": "2025-06-17",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 1.25,
        "outputPerMTokens": 10,
        "cachedInputPerMTokens": 0.125,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 88.3,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 0,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 24.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 85.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 84.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 57.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1457.8,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "Aider Polyglot",
          "score": 82.2,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "AIME 2025",
          "score": 88,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 86.4,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 21.6,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "LiveCodeBench",
          "score": 74.2,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "MMMU",
          "score": 82,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "SWE-bench Verified (single attempt)",
          "score": 59.6,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        }
      ]
    },
    {
      "id": "gemini-2-5-flash",
      "name": "Gemini 2.5 Flash",
      "providerId": "google",
      "releaseDate": "2025-06-17",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 0.3,
        "outputPerMTokens": 2.5,
        "cachedInputPerMTokens": 0.03,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 70.8,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1417.3,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "Aider Polyglot",
          "score": 56.7,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "AIME 2025",
          "score": 72,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 82.8,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 11,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "LiveCodeBench",
          "score": 59.3,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "MMMU",
          "score": 79.7,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "SWE-bench Verified (single attempt)",
          "score": 48.9,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
          "measuredAt": "2025-06-17"
        }
      ]
    },
    {
      "id": "gemini-2-5-flash-lite",
      "name": "Gemini 2.5 Flash-Lite",
      "providerId": "google",
      "releaseDate": "2025-07-22",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 0.1,
        "outputPerMTokens": 0.4,
        "cachedInputPerMTokens": 0.01,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025",
          "score": 63.1,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Lite-Model-Card.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 66.7,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Lite-Model-Card.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 6.9,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Lite-Model-Card.pdf",
          "measuredAt": "2025-06-17"
        },
        {
          "benchmark": "SWE-bench Verified (single attempt)",
          "score": 27.6,
          "unit": "percent",
          "sourceUrl": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Lite-Model-Card.pdf",
          "measuredAt": "2025-06-17"
        }
      ]
    },
    {
      "id": "deepseek-v3-2",
      "name": "DeepSeek V3.2",
      "providerId": "deepseek",
      "releaseDate": "2025-12-01",
      "contextWindow": 131072,
      "maxOutput": 65536,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://api-docs.deepseek.com/news/news251201/",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 94.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 94.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1424.8,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AIME 2025",
          "score": 93.1,
          "unit": "percent",
          "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
          "measuredAt": "2025-12-01"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 82.4,
          "unit": "percent",
          "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
          "measuredAt": "2025-12-01"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 25.1,
          "unit": "percent",
          "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
          "measuredAt": "2025-12-01"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 85,
          "unit": "percent",
          "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
          "measuredAt": "2025-12-01"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 73.1,
          "unit": "percent",
          "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
          "measuredAt": "2025-12-01"
        }
      ]
    },
    {
      "id": "deepseek-r1",
      "name": "DeepSeek R1",
      "providerId": "deepseek",
      "releaseDate": "2025-01-20",
      "contextWindow": 128000,
      "maxOutput": 32768,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-R1",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 70,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 71.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 53.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1372.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 71.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-R1/blob/main/README.md",
          "measuredAt": "2025-01-20"
        },
        {
          "benchmark": "MMLU",
          "score": 90.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-R1/blob/main/README.md",
          "measuredAt": "2025-01-20"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 84,
          "unit": "percent",
          "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-R1/blob/main/README.md",
          "measuredAt": "2025-01-20"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 49.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-R1/blob/main/README.md",
          "measuredAt": "2025-01-20"
        }
      ]
    },
    {
      "id": "deepseek-v3",
      "name": "DeepSeek V3",
      "providerId": "deepseek",
      "releaseDate": "2024-12-26",
      "contextWindow": 128000,
      "maxOutput": 8192,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-V3",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 25,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 56.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 15.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1332.4,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 59.1,
          "unit": "percent",
          "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-V3",
          "measuredAt": "2024-12-26"
        },
        {
          "benchmark": "MMLU",
          "score": 88.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-V3",
          "measuredAt": "2024-12-26"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 75.9,
          "unit": "percent",
          "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-V3",
          "measuredAt": "2024-12-26"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 42,
          "unit": "percent",
          "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-V3",
          "measuredAt": "2024-12-26"
        }
      ]
    },
    {
      "id": "mistral-large-3",
      "name": "Mistral Large 3",
      "providerId": "mistral",
      "releaseDate": "2025-12-02",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.5,
        "outputPerMTokens": 1.5,
        "cachedInputPerMTokens": 0.05,
        "sourceUrl": "https://docs.mistral.ai/inference/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1427,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "codestral",
      "name": "Codestral",
      "providerId": "mistral",
      "releaseDate": "2025-01-13",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.3,
        "outputPerMTokens": 0.9,
        "cachedInputPerMTokens": 0.03,
        "sourceUrl": "https://docs.mistral.ai/inference/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "HumanEval",
          "score": 86.6,
          "unit": "percent",
          "sourceUrl": "https://mistral.ai/news/codestral-2501/",
          "measuredAt": "2025-01-13"
        }
      ]
    },
    {
      "id": "ministral-3-14b",
      "name": "Ministral 3 14B",
      "providerId": "mistral",
      "releaseDate": "2025-12-02",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.2,
        "outputPerMTokens": 0.2,
        "cachedInputPerMTokens": 0.02,
        "sourceUrl": "https://docs.mistral.ai/inference/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025",
          "score": 85,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-14B-Instruct-2512-BF16",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 71.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-14B-Instruct-2512-BF16",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "LiveCodeBench",
          "score": 64.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-14B-Instruct-2512-BF16",
          "measuredAt": "2025-12-02"
        }
      ]
    },
    {
      "id": "qwen3-max",
      "name": "Qwen3-Max",
      "providerId": "qwen",
      "releaseDate": "2025-09-23",
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1.2,
        "outputPerMTokens": 6,
        "cachedInputPerMTokens": 0.12,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 18.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 72.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 73.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 48.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 69.6,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?from=research.latest-advancements-list&id=241398b9cd6353de490b0f82806c7848c5d2777d",
          "measuredAt": "2025-09-24"
        }
      ]
    },
    {
      "id": "qwen3-coder-plus",
      "name": "Qwen3-Coder-Plus",
      "providerId": "qwen",
      "releaseDate": "2025-07-22",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.574,
        "outputPerMTokens": 2.294,
        "cachedInputPerMTokens": 0.0574,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "SWE-bench Verified",
          "score": 69.6,
          "unit": "percent",
          "sourceUrl": "https://www.linkedin.com/posts/qwen_were-excited-to-announce-the-upgrade-of-activity-7376348152515260416-b4oF",
          "measuredAt": "2025-09-23"
        }
      ]
    },
    {
      "id": "qwen3-235b-a22b",
      "name": "Qwen3 235B-A22B",
      "providerId": "qwen",
      "releaseDate": "2025-04-29",
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://github.com/QwenLM/Qwen3",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 80.8,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 70.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1366.1,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "ARC-AGI-1 (pass@1)",
          "score": 40.75,
          "unit": "percent",
          "sourceUrl": "https://github.com/QwenLM/Qwen3/blob/main/eval/README.md",
          "measuredAt": "2025-07-21"
        }
      ]
    },
    {
      "id": "grok-4-5",
      "name": "Grok 4.5",
      "providerId": "xai",
      "releaseDate": "2026-07-08",
      "contextWindow": 500000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 6,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://docs.x.ai/developers/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 24.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 57.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 93.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 97.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 48.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1450.1,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "llama-3-3-70b-instruct",
      "name": "Llama 3.3 70B Instruct",
      "providerId": "meta",
      "releaseDate": "2024-12-06",
      "contextWindow": 131072,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.llama.com/docs/model-cards-and-prompt-formats/llama3_3/",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 47.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 5.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1274,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 50.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
          "measuredAt": "2025-04-05"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 68.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
          "measuredAt": "2025-04-05"
        }
      ]
    },
    {
      "id": "solar-pro-4",
      "name": "Solar Pro 4",
      "providerId": "upstage",
      "releaseDate": "2026-08-20",
      "contextWindow": 512000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.3,
        "outputPerMTokens": 1.2,
        "cachedInputPerMTokens": 0.06,
        "sourceUrl": "https://console.upstage.ai/docs/models/compare/generative-intelligence",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1377.3,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "Artificial Analysis Intelligence Index",
          "score": 42,
          "unit": "score",
          "sourceUrl": "https://www.upstage.ai/news/upstage-ai-unveils-solar-pro-4-scoring-42-on-artificial-analysis-index-to-rank",
          "measuredAt": "2026-08-20"
        }
      ]
    },
    {
      "id": "solar-open-2-250b",
      "name": "Solar Open 2 250B-A15B",
      "providerId": "upstage",
      "releaseDate": "2026-07-24",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AA-LCR",
          "score": 62.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "AIME 2026",
          "score": 95.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "APEX-Agents",
          "score": 16.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "ArtifactsBench",
          "score": 55.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "GDPval-AA v2",
          "score": 1128,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 86.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "HMMT 2026-02",
          "score": 93.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 28.8,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "IFBench",
          "score": 80,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 92.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 58.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 86.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "Multi-Challenge",
          "score": 61,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 70.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "Tau3-Bench (banking)",
          "score": 19.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        },
        {
          "benchmark": "Terminal Bench Hard",
          "score": 28.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
          "measuredAt": "2026-07-24"
        }
      ]
    },
    {
      "id": "hy3",
      "name": "Hy3",
      "providerId": "tencent",
      "releaseDate": "2026-07-06",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://cloud.tencent.com/document/product/1823/130055",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1440.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AA-LCR",
          "score": 73.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Apex-Agent",
          "score": 25.6,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "ArxivMath",
          "score": 52.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "BrowseComp",
          "score": 84.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "CL-bench",
          "score": 23.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "CL-bench Life",
          "score": 17,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "ClawEval",
          "score": 68.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "CMT-Benchmark",
          "score": 37.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "DeepSearchQA",
          "score": 91,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "DeepSWE",
          "score": 28,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "e-bench",
          "score": 50.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "FrontierScience-Olympiad",
          "score": 74.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "FrontierScience-Research",
          "score": 21.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 90.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "HLE (with tools, text-only)",
          "score": 53.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "HorizonMath",
          "score": 7.1,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 37,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Hy-Backend 2.0",
          "score": 25,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Hy-CompanyBench",
          "score": 41.7,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Hy-Euler Pro",
          "score": 24.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Hy-FinModelBench",
          "score": 69,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Hy-Math",
          "score": 60.9,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Hy-SkillsWorld",
          "score": 45.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Hy-SWE Max",
          "score": 49,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "IMOAnswerBench",
          "score": 90,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "MathArena Apex",
          "score": 38.7,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 79.1,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "NL2Repo",
          "score": 45.6,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "PHYBench",
          "score": 77.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "ProdBench",
          "score": 23,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "SkillsBench",
          "score": 55.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "SuperChem",
          "score": 54.9,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "SWE-bench Multilingual",
          "score": 75.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "SWE-bench Pro",
          "score": 57.9,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 78,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 71.7,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "Toolathlon",
          "score": 48.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "USAMO 2026",
          "score": 72,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "WideSearch",
          "score": 76.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        },
        {
          "benchmark": "WildClawBench",
          "score": 53.6,
          "unit": "percent",
          "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
          "measuredAt": "2026-07-06"
        }
      ]
    },
    {
      "id": "longcat-2-0",
      "name": "LongCat 2.0",
      "providerId": "meituan",
      "releaseDate": "2026-06-30",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://longcat.chat/platform/docs/ChangeLog.html",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "BrowseComp",
          "score": 79.9,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "FORTE",
          "score": 73.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 88.9,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "IFEval",
          "score": 90,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "IMO-AnswerBench",
          "score": 81.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "RWSearch",
          "score": 78.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "SWE-bench Multilingual",
          "score": 77.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "SWE-bench Pro",
          "score": 59.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 70.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "Writing Bench",
          "score": 83.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
          "measuredAt": "2026-06-30"
        }
      ]
    },
    {
      "id": "longcat-flash-chat",
      "name": "LongCat Flash Chat",
      "providerId": "meituan",
      "releaseDate": "2025-09-01",
      "contextWindow": 128000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://longcat.chat/platform/docs/ChangeLog.html",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1422.9,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AceBench",
          "score": 76.1,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "AIME 2024",
          "score": 70.42,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "AIME 2025",
          "score": 61.25,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "ArenaHard-V2",
          "score": 86.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "BeyondAIME",
          "score": 43,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "C-Eval",
          "score": 90.44,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "CMMLU",
          "score": 84.34,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "COLLIE",
          "score": 57.1,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "DROP",
          "score": 79.06,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 73.23,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "GraphWalks-128k",
          "score": 51.05,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "HumanEval+",
          "score": 88.41,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "IFEval",
          "score": 89.65,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "LiveCodeBench",
          "score": 48.02,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "MATH500",
          "score": 96.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "MBPP+",
          "score": 79.63,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "Meeseeks-zh",
          "score": 43.03,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "MMLU",
          "score": 89.71,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 82.68,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "Safety: Criminal",
          "score": 91.24,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "Safety: Harmful",
          "score": 83.98,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "Safety: Misinformation",
          "score": 81.72,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "Safety: Privacy",
          "score": 93.98,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 60.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "Tau2-Bench (airline)",
          "score": 58,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "Tau2-Bench (retail)",
          "score": 71.27,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "Tau2-Bench (telecom)",
          "score": 73.68,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "TerminalBench",
          "score": 39.51,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "VitaBench",
          "score": 24.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        },
        {
          "benchmark": "ZebraLogic",
          "score": 89.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
          "measuredAt": "2025-09-01"
        }
      ]
    },
    {
      "id": "step-3-7-flash",
      "name": "Step 3.7 Flash",
      "providerId": "stepfun",
      "releaseDate": "2026-05-29",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://platform.stepfun.ai/docs/en/guides/models/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 95,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AA-LCR",
          "score": 63.9,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "BrowseComp",
          "score": 75.8,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "ClawEval v1.1",
          "score": 67.1,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "DeepSearchQA Accuracy",
          "score": 81.7,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "DeepSearchQA F1",
          "score": 92.8,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "GDPval-Stirrup",
          "score": 1415.8,
          "unit": "elo",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "GDPval-Stirrup Intelligence Index",
          "score": 45.8,
          "unit": "score",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "HLE (with tools, text-only)",
          "score": 49.7,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "HLE (with tools)",
          "score": 47.2,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "ResearchRubrics",
          "score": 71.7,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "SimpleVQA (with tools)",
          "score": 79.2,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "SWE-bench Multilingual",
          "score": 72.4,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "SWE-bench Pro",
          "score": 56.3,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 76.5,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 59.6,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "Toolathlon",
          "score": 49.5,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "V* (with Python)",
          "score": 95.3,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        },
        {
          "benchmark": "WorldVQA (with visual search)",
          "score": 58.1,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
          "measuredAt": "2026-05-29"
        }
      ]
    },
    {
      "id": "step-3-5-flash",
      "name": "Step 3.5 Flash",
      "providerId": "stepfun",
      "releaseDate": "2026-02-12",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://platform.stepfun.ai/docs/en/guides/models/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 98.3,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 96.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1403.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AIME 2025",
          "score": 97.3,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "BrowseComp",
          "score": 51.6,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "BrowseComp (with context manager)",
          "score": 69,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "BrowseComp-ZH",
          "score": 66.9,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "BrowseComp-ZH (with context manager)",
          "score": 73.7,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "GAIA (no file)",
          "score": 84.5,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "HMMT 2025 (Feb/Nov average)",
          "score": 96.2,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "IMOAnswerBench",
          "score": 85.4,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 86.4,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "ResearchRubrics",
          "score": 65.3,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 74.4,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "Tau2-Bench",
          "score": 88.2,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 51,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "xbench-DeepSearch 2025.05",
          "score": 83.7,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "xbench-DeepSearch 2025.10",
          "score": 56.3,
          "unit": "percent",
          "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
          "measuredAt": "2026-02-12"
        }
      ]
    },
    {
      "id": "seed-2-1-pro",
      "name": "Seed2.1 Pro",
      "providerId": "bytedance",
      "releaseDate": "2026-07-13",
      "contextWindow": null,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision",
        "video"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "Agent Startup Bench",
          "score": 68.8,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "BabyVision",
          "score": 73.7,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "BeyondAIME",
          "score": 87,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "CharXiv-RQ",
          "score": 85.4,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "CharXiv-RQ (with tools)",
          "score": 86.4,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "ERQA",
          "score": 72,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "KINA",
          "score": 48.3,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MathVision",
          "score": 92.6,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MathVision (with tools)",
          "score": 94.5,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "Minerva Video Reasoning",
          "score": 70.7,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MMLongBench 128K",
          "score": 78.3,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MMMU-Pro",
          "score": 81.6,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 82.7,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "NL2Repo-Bench",
          "score": 47,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "OVOBench",
          "score": 80.7,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "SuperGPQA",
          "score": 70.8,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "SWE-Atlas",
          "score": 35.2,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 71,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "TOMATO",
          "score": 79.5,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "VideoMME",
          "score": 89.2,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "VideoSimpleQA",
          "score": 76.4,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "Workspace Bench",
          "score": 53,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "WorldVQA",
          "score": 53,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "xDailyBench",
          "score": 61,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "ZEROBench",
          "score": 18,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "ZEROBench (with tools)",
          "score": 22,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        }
      ]
    },
    {
      "id": "dola-seed-2-1-turbo",
      "name": "Dola Seed 2.1 Turbo",
      "providerId": "bytedance",
      "releaseDate": "2026-07-13",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision",
        "video"
      ],
      "pricing": {
        "inputPerMTokens": 0.5,
        "outputPerMTokens": 2.5,
        "cachedInputPerMTokens": 0.1,
        "sourceUrl": "https://www.byteplus.com/en/product/modelark/leaderboard",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "Agent Startup Bench",
          "score": 54,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "BabyVision",
          "score": 62.9,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "BeyondAIME",
          "score": 88,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "CharXiv-RQ",
          "score": 82.5,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "CharXiv-RQ (with tools)",
          "score": 83.6,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "ERQA",
          "score": 71.3,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "KINA",
          "score": 46.6,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MathVision",
          "score": 90.1,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MathVision (with tools)",
          "score": 92.7,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "Minerva Video Reasoning",
          "score": 65.9,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MMLongBench 128K",
          "score": 76.9,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MMMU-Pro",
          "score": 80.1,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 82.2,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "NL2Repo-Bench",
          "score": 43.7,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "OVOBench",
          "score": 79.2,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "SuperGPQA",
          "score": 67.4,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "SWE-Atlas",
          "score": 30.6,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 67.6,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "TOMATO",
          "score": 56.8,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "VideoMME",
          "score": 89,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "VideoSimpleQA",
          "score": 71.4,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "Workspace Bench",
          "score": 54.7,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "WorldVQA",
          "score": 48.6,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "xDailyBench",
          "score": 56.4,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "ZEROBench",
          "score": 11,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        },
        {
          "benchmark": "ZEROBench (with tools)",
          "score": 20,
          "unit": "percent",
          "sourceUrl": "https://seed.bytedance.com/en/seed2_1",
          "measuredAt": "2026-07-13"
        }
      ]
    },
    {
      "id": "dola-seed-2-0-pro",
      "name": "Dola Seed 2.0 Pro",
      "providerId": "bytedance",
      "releaseDate": "2026-02-16",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision",
        "video"
      ],
      "pricing": {
        "inputPerMTokens": 0.5,
        "outputPerMTokens": 3,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.byteplus.com/en/product/modelark/leaderboard",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1447.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AIME 2025 (no tools)",
          "score": 99,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "BFCL v4",
          "score": 73.4,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "BrowseComp",
          "score": 77.3,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "DeepResearchBench",
          "score": 53.3,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 92.4,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "HLE-Verified",
          "score": 73.6,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 33.3,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 90.1,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "SciCode",
          "score": 48.5,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "SWE-bench Multilingual",
          "score": 71.7,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "SWE-bench Pro",
          "score": 46.9,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 76.5,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 55.8,
          "unit": "percent",
          "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
          "measuredAt": "2026-02-16"
        }
      ]
    },
    {
      "id": "minimax-m3",
      "name": "MiniMax M3",
      "providerId": "minimax",
      "releaseDate": "2026-06-01",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.3,
        "outputPerMTokens": 1.2,
        "cachedInputPerMTokens": 0.06,
        "sourceUrl": "https://platform.minimax.io/subscribe/token-plan?tab=api-enterprise",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 71.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1433.5,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "minimax-m2-7",
      "name": "MiniMax M2.7",
      "providerId": "minimax",
      "releaseDate": "2026-03-18",
      "contextWindow": 204800,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.3,
        "outputPerMTokens": 1.2,
        "cachedInputPerMTokens": 0.06,
        "sourceUrl": "https://platform.minimax.io/docs/api-reference/anthropic-api-compatible-cache",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1404.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "minimax-m2-5",
      "name": "MiniMax M2.5",
      "providerId": "minimax",
      "releaseDate": "2026-02-12",
      "contextWindow": 204800,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.3,
        "outputPerMTokens": 1.2,
        "cachedInputPerMTokens": 0.03,
        "sourceUrl": "https://platform.minimax.io/docs/api-reference/anthropic-api-compatible-cache",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1358.9,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "sonar",
      "name": "Sonar",
      "providerId": "perplexity",
      "releaseDate": "2025-01-21",
      "contextWindow": 128000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1,
        "outputPerMTokens": 1,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.perplexity.ai/docs/sonar/models/sonar",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "SimpleQA F-score",
          "score": 77.3,
          "unit": "percent",
          "sourceUrl": "https://www.perplexity.ai/hub/blog/introducing-the-sonar-pro-api",
          "measuredAt": "2025-01-21"
        }
      ]
    },
    {
      "id": "sonar-pro",
      "name": "Sonar Pro",
      "providerId": "perplexity",
      "releaseDate": "2025-01-21",
      "contextWindow": 200000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.perplexity.ai/docs/sonar/models/sonar-pro",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "SimpleQA F-score",
          "score": 85.8,
          "unit": "percent",
          "sourceUrl": "https://www.perplexity.ai/hub/blog/introducing-the-sonar-pro-api",
          "measuredAt": "2025-01-21"
        }
      ]
    },
    {
      "id": "sonar-reasoning-pro",
      "name": "Sonar Reasoning Pro",
      "providerId": "perplexity",
      "releaseDate": null,
      "contextWindow": 128000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 8,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.perplexity.ai/docs/sonar/models/sonar-reasoning-pro",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "sonar-deep-research",
      "name": "Sonar Deep Research",
      "providerId": "perplexity",
      "releaseDate": "2025-02-14",
      "contextWindow": 128000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 8,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.perplexity.ai/docs/sonar/models/sonar-deep-research",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "DRACO Breadth & Depth (Opus 4.6 base)",
          "score": 73.1,
          "unit": "percent",
          "sourceUrl": "https://r2cdn.perplexity.ai/pplx-draco.pdf",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "DRACO Citation Quality (Opus 4.6 base)",
          "score": 64.6,
          "unit": "percent",
          "sourceUrl": "https://r2cdn.perplexity.ai/pplx-draco.pdf",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "DRACO Factual Accuracy (Opus 4.6 base)",
          "score": 67.9,
          "unit": "percent",
          "sourceUrl": "https://r2cdn.perplexity.ai/pplx-draco.pdf",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "DRACO Normalized Score (Opus 4.6 base)",
          "score": 70.5,
          "unit": "percent",
          "sourceUrl": "https://r2cdn.perplexity.ai/pplx-draco.pdf",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "DRACO Pass Rate (Opus 4.6 base)",
          "score": 72.8,
          "unit": "percent",
          "sourceUrl": "https://r2cdn.perplexity.ai/pplx-draco.pdf",
          "measuredAt": "2026-02-12"
        },
        {
          "benchmark": "DRACO Presentation Quality (Opus 4.6 base)",
          "score": 90.3,
          "unit": "percent",
          "sourceUrl": "https://r2cdn.perplexity.ai/pplx-draco.pdf",
          "measuredAt": "2026-02-12"
        }
      ]
    },
    {
      "id": "command-a-plus",
      "name": "Command A+",
      "providerId": "cohere",
      "releaseDate": "2026-05-20",
      "contextWindow": 128000,
      "maxOutput": 64000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/command-a-plus",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025",
          "score": 90,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "CharXiv Reasoning",
          "score": 52.7,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "IFBench",
          "score": 74,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "MathVista",
          "score": 80.6,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "MMMU",
          "score": 75.1,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "MMMU-Pro",
          "score": 63,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "SciCode",
          "score": 38,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 85,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "Terminal-Bench Hard",
          "score": 25,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        }
      ]
    },
    {
      "id": "command-a-reasoning",
      "name": "Command A Reasoning",
      "providerId": "cohere",
      "releaseDate": "2025-08-21",
      "contextWindow": 256000,
      "maxOutput": 32000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/command-a-reasoning",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025",
          "score": 57,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "IFBench",
          "score": 36,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "SciCode",
          "score": 30,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 37,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        },
        {
          "benchmark": "Terminal-Bench Hard",
          "score": 3,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/command-a-plus",
          "measuredAt": "2026-05-20"
        }
      ]
    },
    {
      "id": "command-a-vision",
      "name": "Command A Vision",
      "providerId": "cohere",
      "releaseDate": "2025-07-31",
      "contextWindow": 128000,
      "maxOutput": 8000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/command-a-vision",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "command-a-translate",
      "name": "Command A Translate",
      "providerId": "cohere",
      "releaseDate": "2025-08-28",
      "contextWindow": 16000,
      "maxOutput": 8000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/command-a-translate",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "north-mini-code",
      "name": "North Mini Code",
      "providerId": "cohere",
      "releaseDate": "2026-06-09",
      "contextWindow": 256000,
      "maxOutput": 64000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/north-mini-code-1.0",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LiveCodeBench v6",
          "score": 70.3,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/north-mini-code",
          "measuredAt": "2026-06-09"
        },
        {
          "benchmark": "SciCode",
          "score": 38.2,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/north-mini-code",
          "measuredAt": "2026-06-09"
        },
        {
          "benchmark": "SWE-bench Pro",
          "score": 40.2,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/north-mini-code",
          "measuredAt": "2026-06-09"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 67.6,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/north-mini-code",
          "measuredAt": "2026-06-09"
        },
        {
          "benchmark": "Terminal-Bench Hard",
          "score": 31.1,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/north-mini-code",
          "measuredAt": "2026-06-09"
        },
        {
          "benchmark": "Terminal-Bench v2",
          "score": 36,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/blog/north-mini-code",
          "measuredAt": "2026-06-09"
        }
      ]
    },
    {
      "id": "north-small-translate",
      "name": "North Small Translate",
      "providerId": "cohere",
      "releaseDate": null,
      "contextWindow": 16000,
      "maxOutput": 16000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/models",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "command-a",
      "name": "Command A",
      "providerId": "cohere",
      "releaseDate": "2025-03-13",
      "contextWindow": 256000,
      "maxOutput": 8000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 2.5,
        "outputPerMTokens": 10,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/command-a",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "BFCL Overall",
          "score": 63.8,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "GPQA",
          "score": 50.8,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "IFEval",
          "score": 90.9,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "MBPP+",
          "score": 86.2,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "MMLU",
          "score": 85.5,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 69.6,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "RepoQA",
          "score": 92.6,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        }
      ]
    },
    {
      "id": "command-r7b",
      "name": "Command R7B",
      "providerId": "cohere",
      "releaseDate": "2024-12-16",
      "contextWindow": 128000,
      "maxOutput": 4000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.0375,
        "outputPerMTokens": 0.15,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/command-r7b",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "BFCL Overall",
          "score": 52.2,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "GPQA",
          "score": 26.3,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "IFEval",
          "score": 77.9,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "MBPP+",
          "score": 72,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "MMLU",
          "score": 65.2,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 42.4,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        },
        {
          "benchmark": "RepoQA",
          "score": 69.6,
          "unit": "percent",
          "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
          "measuredAt": "2025-03-13"
        }
      ]
    },
    {
      "id": "command-r",
      "name": "Command R",
      "providerId": "cohere",
      "releaseDate": "2024-08-30",
      "contextWindow": 128000,
      "maxOutput": 4000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.15,
        "outputPerMTokens": 0.6,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/command-r",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1163,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "command-r-plus",
      "name": "Command R+",
      "providerId": "cohere",
      "releaseDate": "2024-08-30",
      "contextWindow": 128000,
      "maxOutput": 4000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 2.5,
        "outputPerMTokens": 10,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.cohere.com/docs/command-r-plus",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1203.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "amazon-nova-2-lite",
      "name": "Amazon Nova 2 Lite",
      "providerId": "amazon",
      "releaseDate": "2025-12-02",
      "contextWindow": 1000000,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.aws.amazon.com/nova/latest/nova2-userguide/what-is-nova-2.html",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025",
          "score": 91,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "BFCL v4",
          "score": 60.3,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 79.6,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "IF-Bench",
          "score": 70.8,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "LongCodeBench 1M",
          "score": 84,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 24.6,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 80.9,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "MultiChallenge",
          "score": 76.6,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "RealKIE-FCC Verified",
          "score": 62.1,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "tau2-bench Airline Verified",
          "score": 64.8,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "tau2-bench Retail Verified",
          "score": 76.5,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "tau2-bench Telecom",
          "score": 76,
          "unit": "percent",
          "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
          "measuredAt": "2025-12-02"
        }
      ]
    },
    {
      "id": "amazon-nova-premier",
      "name": "Amazon Nova Premier",
      "providerId": "amazon",
      "releaseDate": "2025-04-30",
      "contextWindow": 1000000,
      "maxOutput": 10000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.aws.amazon.com/nova/latest/userguide/what-is-nova.html",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "BFCL",
          "score": 63.7,
          "unit": "percent",
          "sourceUrl": "https://aws.amazon.com/about-aws/whats-new/2025/04/amazon-nova-premier-complex-tasks-model-distillation/",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "CharXiv",
          "score": 84.6,
          "unit": "percent",
          "sourceUrl": "https://aws.amazon.com/about-aws/whats-new/2025/04/amazon-nova-premier-complex-tasks-model-distillation/",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "MATH-500",
          "score": 82,
          "unit": "percent",
          "sourceUrl": "https://aws.amazon.com/about-aws/whats-new/2025/04/amazon-nova-premier-complex-tasks-model-distillation/",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "MMLU",
          "score": 87.4,
          "unit": "percent",
          "sourceUrl": "https://aws.amazon.com/about-aws/whats-new/2025/04/amazon-nova-premier-complex-tasks-model-distillation/",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "SimpleQA with RAG",
          "score": 86.3,
          "unit": "percent",
          "sourceUrl": "https://aws.amazon.com/about-aws/whats-new/2025/04/amazon-nova-premier-complex-tasks-model-distillation/",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 42.4,
          "unit": "percent",
          "sourceUrl": "https://aws.amazon.com/about-aws/whats-new/2025/04/amazon-nova-premier-complex-tasks-model-distillation/",
          "measuredAt": "2025-04-30"
        }
      ]
    },
    {
      "id": "amazon-nova-pro",
      "name": "Amazon Nova Pro",
      "providerId": "amazon",
      "releaseDate": "2024-12-02",
      "contextWindow": 300000,
      "maxOutput": 10000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.8,
        "outputPerMTokens": 3.2,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://aws.amazon.com/blogs/machine-learning/demystifying-amazon-bedrock-pricing-for-a-chatbot-assistant/",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "ARC-C",
          "score": 94.8,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "BBH",
          "score": 86.9,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "DROP",
          "score": 85.4,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "GPQA",
          "score": 46.9,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "GSM8K",
          "score": 94.8,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "IFEval",
          "score": 92.1,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "MATH",
          "score": 76.6,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "MMLU",
          "score": 85.9,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        }
      ]
    },
    {
      "id": "amazon-nova-lite",
      "name": "Amazon Nova Lite",
      "providerId": "amazon",
      "releaseDate": "2024-12-02",
      "contextWindow": 300000,
      "maxOutput": 10000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.06,
        "outputPerMTokens": 0.24,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://aws.amazon.com/blogs/machine-learning/demystifying-amazon-bedrock-pricing-for-a-chatbot-assistant/",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "ARC-C",
          "score": 92.4,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "BBH",
          "score": 82.4,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "DROP",
          "score": 80.2,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "GPQA",
          "score": 42,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "GSM8K",
          "score": 94.5,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "IFEval",
          "score": 89.7,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "MATH",
          "score": 73.3,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        },
        {
          "benchmark": "MMLU",
          "score": 80.5,
          "unit": "percent",
          "sourceUrl": "https://assets.amazon.science/10/0a/0b61d39a4e9aaec16f71ad3d9168/the-amazon-nova-family-of-models-technical-report-and-model-card2-26.pdf",
          "measuredAt": "2024-12-03"
        }
      ]
    },
    {
      "id": "ministral-3-8b",
      "name": "Ministral 3 8B",
      "providerId": "mistral",
      "releaseDate": "2025-12-02",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.15,
        "outputPerMTokens": 0.15,
        "cachedInputPerMTokens": 0.015,
        "sourceUrl": "https://docs.mistral.ai/inference/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025",
          "score": 78.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-8B-Instruct-2512-BF16",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 66.8,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-8B-Instruct-2512-BF16",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "LiveCodeBench",
          "score": 61.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-8B-Instruct-2512-BF16",
          "measuredAt": "2025-12-02"
        }
      ]
    },
    {
      "id": "ministral-3-3b",
      "name": "Ministral 3 3B",
      "providerId": "mistral",
      "releaseDate": "2025-12-02",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.1,
        "outputPerMTokens": 0.1,
        "cachedInputPerMTokens": 0.01,
        "sourceUrl": "https://docs.mistral.ai/inference/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025",
          "score": 72.1,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-3B-Instruct-2512",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 53.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-3B-Instruct-2512",
          "measuredAt": "2025-12-02"
        },
        {
          "benchmark": "LiveCodeBench",
          "score": 54.8,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-3B-Instruct-2512",
          "measuredAt": "2025-12-02"
        }
      ]
    },
    {
      "id": "grok-4-fast-reasoning",
      "name": "Grok 4 Fast Reasoning",
      "providerId": "xai",
      "releaseDate": "2025-09-19",
      "contextWindow": 2000000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.2,
        "outputPerMTokens": 0.5,
        "cachedInputPerMTokens": 0.05,
        "sourceUrl": "https://x.ai/news/grok-4-fast",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (no tools)",
          "score": 92,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "BrowseComp",
          "score": 44.9,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "BrowseComp (zh)",
          "score": 51.2,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 85.7,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "HMMT 2025 (no tools)",
          "score": 93.3,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 20,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "LiveCodeBench (Jan-May)",
          "score": 80,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "Reka Research Eval",
          "score": 66,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "Search Arena Elo",
          "score": 1163,
          "unit": "elo",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "SimpleQA",
          "score": 95,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "X Bench Deepsearch (zh)",
          "score": 74,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        },
        {
          "benchmark": "X Browse",
          "score": 58,
          "unit": "percent",
          "sourceUrl": "https://x.ai/news/grok-4-fast",
          "measuredAt": "2025-09-19"
        }
      ]
    },
    {
      "id": "grok-4-20-reasoning",
      "name": "Grok 4.20 Reasoning",
      "providerId": "xai",
      "releaseDate": "2026-03-09",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.25,
        "outputPerMTokens": 2.5,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://docs.x.ai/developers/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 17.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 44.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 89.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 92.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 30.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        }
      ]
    },
    {
      "id": "grok-4-20-non-reasoning",
      "name": "Grok 4.20 Non-Reasoning",
      "providerId": "xai",
      "releaseDate": "2026-03-09",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.25,
        "outputPerMTokens": 2.5,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://docs.x.ai/developers/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "grok-4-20-multi-agent",
      "name": "Grok 4.20 Multi-Agent",
      "providerId": "xai",
      "releaseDate": "2026-03-09",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.25,
        "outputPerMTokens": 2.5,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://docs.x.ai/developers/models/grok-4.20-multi-agent-0309",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "grok-build-0-1",
      "name": "Grok Build 0.1",
      "providerId": "xai",
      "releaseDate": "2026-05-19",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1,
        "outputPerMTokens": 2,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://docs.x.ai/developers/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "mai-thinking-1",
      "name": "MAI-Thinking-1",
      "providerId": "microsoft",
      "releaseDate": "2026-08-12",
      "contextWindow": 256000,
      "maxOutput": 64000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.azure.com/catalog/publishers/microsoft",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AdvancedIF",
          "score": 85,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "AIME 2025",
          "score": 97,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "AIME 2026",
          "score": 94.5,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "BFCL v3",
          "score": 72,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 84.2,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "GraphWalks (<=128K)",
          "score": 90,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "HMMT February 2026",
          "score": 84.9,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "IFBench",
          "score": 69,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 87.7,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 85,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "MultiChallenge",
          "score": 53,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "SimpleQA Verified",
          "score": 31,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "SWE-bench Pro",
          "score": 52.8,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 73.5,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 46,
          "unit": "percent",
          "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
          "measuredAt": "2026-08-12"
        }
      ]
    },
    {
      "id": "phi-4",
      "name": "Phi-4",
      "providerId": "microsoft",
      "releaseDate": "2024-12-12",
      "contextWindow": 16384,
      "maxOutput": 16384,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.azure.com/catalog/models/Phi-4",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 56.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 13.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1216.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "ArenaHard",
          "score": 75.4,
          "unit": "percent",
          "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
          "measuredAt": "2024-12-12"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 56.1,
          "unit": "percent",
          "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
          "measuredAt": "2024-12-12"
        },
        {
          "benchmark": "HumanEval",
          "score": 82.6,
          "unit": "percent",
          "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
          "measuredAt": "2024-12-12"
        },
        {
          "benchmark": "HumanEval+",
          "score": 82.8,
          "unit": "percent",
          "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
          "measuredAt": "2024-12-12"
        },
        {
          "benchmark": "IFEval",
          "score": 63,
          "unit": "percent",
          "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
          "measuredAt": "2024-12-12"
        },
        {
          "benchmark": "MATH",
          "score": 80.4,
          "unit": "percent",
          "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
          "measuredAt": "2024-12-12"
        },
        {
          "benchmark": "MMLU",
          "score": 84.8,
          "unit": "percent",
          "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
          "measuredAt": "2024-12-12"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 70.4,
          "unit": "percent",
          "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
          "measuredAt": "2024-12-12"
        }
      ]
    },
    {
      "id": "phi-4-reasoning",
      "name": "Phi-4 Reasoning",
      "providerId": "microsoft",
      "releaseDate": "2025-04-30",
      "contextWindow": 32768,
      "maxOutput": 32768,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.azure.com/catalog/publishers/microsoft",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025",
          "score": 62.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-reasoning",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 65.8,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-reasoning",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "HumanEval+",
          "score": 92.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-reasoning",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 74.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-reasoning",
          "measuredAt": "2025-04-30"
        }
      ]
    },
    {
      "id": "phi-4-mini-reasoning",
      "name": "Phi-4 Mini Reasoning",
      "providerId": "microsoft",
      "releaseDate": "2025-04-30",
      "contextWindow": 128000,
      "maxOutput": 128000,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.azure.com/catalog/publishers/microsoft",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2024",
          "score": 57.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-mini-reasoning",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 52,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-mini-reasoning",
          "measuredAt": "2025-04-30"
        },
        {
          "benchmark": "MATH-500",
          "score": 94.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-mini-reasoning",
          "measuredAt": "2025-04-30"
        }
      ]
    },
    {
      "id": "phi-4-multimodal-instruct",
      "name": "Phi-4 Multimodal Instruct",
      "providerId": "microsoft",
      "releaseDate": "2025-02-26",
      "contextWindow": 128000,
      "maxOutput": 4096,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.azure.com/catalog/models/Phi-4-multimodal-instruct?publisher=microsoft",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AI2D",
          "score": 82.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "BLINK",
          "score": 61.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "ChartQA",
          "score": 81.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "DocVQA",
          "score": 93.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "InfoVQA",
          "score": 72.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "InterGPS",
          "score": 48.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "MathVista testmini",
          "score": 62.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "MMBench (dev-en)",
          "score": 86.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "MMMU",
          "score": 55.1,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "MMMU-Pro (standard/vision)",
          "score": 38.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "OCRBench",
          "score": 84.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "POPE",
          "score": 85.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "ScienceQA Visual",
          "score": 97.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "Speech AI2D",
          "score": 68.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "Speech ChartQA",
          "score": 69,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "Speech DocVQA",
          "score": 87.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "Speech InfoVQA",
          "score": 63.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "TextVQA (val)",
          "score": 75.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "Video-MME (16 frames)",
          "score": 55,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        }
      ]
    },
    {
      "id": "kimi-k3",
      "name": "Kimi K3",
      "providerId": "moonshot",
      "releaseDate": "2026-07-16",
      "contextWindow": 1000000,
      "maxOutput": 1048576,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://www.kimi.com/en/blog/kimi-k3",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 96.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 39,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 72.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 93.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 97.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 50.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AA-Briefcase",
          "score": 1548,
          "unit": "elo",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "AA-LCR",
          "score": 74.7,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Agents' Last Exam",
          "score": 28.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "APEX-Agents",
          "score": 41,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "AutomationBench",
          "score": 30.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "BabyVision (with Python)",
          "score": 85.7,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "BrowseComp (1M context, no compaction)",
          "score": 90.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "BrowseComp (context compaction)",
          "score": 91.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "CharXiv Reasoning (no tools)",
          "score": 84.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "CharXiv Reasoning (with Python)",
          "score": 91.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "CorpFin v2",
          "score": 71.6,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "CritPt",
          "score": 23.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "DeepSearchQA (F1)",
          "score": 95,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "DeepSWE v1.1 (Kimi Code)",
          "score": 67.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Finance Agent v2",
          "score": 54.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "FrontierSWE",
          "score": 81.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "GDPval-AA v2",
          "score": 1686,
          "unit": "elo",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 93.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Harvey Lab-AA",
          "score": 94.6,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 43.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 56,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "JobBench",
          "score": 54.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Kimi Code Bench 2.0",
          "score": 72.9,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Legal Research Bench",
          "score": 44.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "MathVision (no tools)",
          "score": 94.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "MathVision (with Python)",
          "score": 97.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "MCP-Atlas",
          "score": 84.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "MCPMark-Verified",
          "score": 94.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "MLS-Bench-Lite",
          "score": 48.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "MMMU-Pro (no tools)",
          "score": 81.6,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "MMMU-Pro (with Python)",
          "score": 83.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "MMVU",
          "score": 82.1,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "OfficeQA Pro",
          "score": 63.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "OmniDocBench",
          "score": 91.1,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "OSWorld 2.0",
          "score": 58.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 84.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "PerceptionBench",
          "score": 58.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "PostTrainBench",
          "score": 36.6,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "ProgramBench",
          "score": 77.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "ResearchRubrics",
          "score": 76.2,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "SaaS-Bench",
          "score": 60.1,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "SciCode",
          "score": 58.7,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "SpreadsheetBench 2",
          "score": 34.8,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "SWE-Marathon",
          "score": 42,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "tau3-Banking",
          "score": 33.4,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 88.3,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Toolathlon-Verified",
          "score": 76.5,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "Video-MME (with subtitles)",
          "score": 90,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "WorldVQA ForceAnswer",
          "score": 51,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "ZeroBench pass@5 (no tools)",
          "score": 23,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "ZeroBench pass@5 (with Python)",
          "score": 41,
          "unit": "percent",
          "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
          "measuredAt": "2026-07-27"
        },
        {
          "benchmark": "BrowseComp (1M context, no compaction)",
          "score": 90.4,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/en/blog/kimi-k3",
          "measuredAt": "2026-07-16"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 67.3,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/en/blog/kimi-k3",
          "measuredAt": "2026-07-16"
        }
      ]
    },
    {
      "id": "kimi-k2-7-code",
      "name": "Kimi K2.7 Code",
      "providerId": "moonshot",
      "releaseDate": "2026-06-12",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://platform.kimi.ai/docs/pricing/chat",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 12.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 54,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 87.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 95.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 36.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "Kimi Claw 24/7 Bench",
          "score": 46.9,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "Kimi Code Bench v2",
          "score": 62,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "MCP-Atlas",
          "score": 76,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "MCPMark-Verified",
          "score": 81.1,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "MLS-Bench-Lite",
          "score": 35.1,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "ProgramBench",
          "score": 53.6,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        }
      ]
    },
    {
      "id": "kimi-k2-6",
      "name": "Kimi K2.6",
      "providerId": "moonshot",
      "releaseDate": "2026-04-20",
      "contextWindow": 256000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://platform.kimi.ai/docs/pricing/chat",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 95.8,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 25.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 57.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 96.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 34.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 76.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1454.9,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "Kimi Claw 24/7 Bench",
          "score": 42.9,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "Kimi Code Bench v2",
          "score": 50.9,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "MCP-Atlas",
          "score": 69.4,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "MCPMark-Verified",
          "score": 72.8,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "MLS-Bench-Lite",
          "score": 26.7,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        },
        {
          "benchmark": "ProgramBench",
          "score": 48.3,
          "unit": "percent",
          "sourceUrl": "https://www.kimi.com/resources/kimi-k2-7-code",
          "measuredAt": "2026-06-12"
        }
      ]
    },
    {
      "id": "glm-5-3",
      "name": "GLM-5.3",
      "providerId": "zai",
      "releaseDate": "2026-08-14",
      "contextWindow": 1000000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1.4,
        "outputPerMTokens": 4.4,
        "cachedInputPerMTokens": 0.26,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 29.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 68.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 91.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 41,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "Agents' Last Exam",
          "score": 28.5,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "CyberGym",
          "score": 84.5,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 66.9,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "ExploitBench",
          "score": 54.4,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "Terminal-Bench 3.0",
          "score": 28.3,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "Agents' Last Exam",
          "score": 28.5,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Agents' Last Exam (ALE-CLI)",
          "score": 28.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "AutomationBench v1.0.6",
          "score": 48.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "CyberGym",
          "score": 84.5,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "CyberGym",
          "score": 84.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 66.9,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 66.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "ExploitBench",
          "score": 54.4,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "ExploitBench",
          "score": 54.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "ExploitGym 2h",
          "score": 105,
          "unit": "score",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "ExploitGym 6h",
          "score": 130,
          "unit": "score",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "FrontierSWE",
          "score": 78.1,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "GDPval-AA v2",
          "score": 1769,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "HLE (with tools)",
          "score": 62.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "NL2Repo",
          "score": 58,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "PostTrainBench",
          "score": 39.8,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "ProgramBench (Almost Solved)",
          "score": 19,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "SWE-Marathon v1.1",
          "score": 42.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 88.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Terminal-Bench 3.0",
          "score": 28.3,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Terminal-Bench 3.0",
          "score": 28.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Toolathlon Verified",
          "score": 73,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Z.ai Code Bench (max effort)",
          "score": 34.5,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
          "measuredAt": "2026-08-14"
        }
      ]
    },
    {
      "id": "glm-5-3-flash",
      "name": "GLM-5.3-Flash",
      "providerId": "zai",
      "releaseDate": "2026-09-16",
      "contextWindow": 1000000,
      "maxOutput": 131072,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.15,
        "outputPerMTokens": 0.5,
        "cachedInputPerMTokens": 0.03,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 17.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 55.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 93.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "Agents' Last Exam (ALE-CLI)",
          "score": 26.3,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "AutomationBench v1.0.6",
          "score": 48.8,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "BabyVision",
          "score": 53.4,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "Chartography (with tools)",
          "score": 78,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "CharXiv Reasoning (with tools)",
          "score": 89.4,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 63.4,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "GDPval-AA v2",
          "score": 1773,
          "unit": "elo",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "HLE (with tools)",
          "score": 55.3,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "MMVU",
          "score": 80.5,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "MVBench",
          "score": 77.8,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "NL2Repo",
          "score": 56.3,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "OfficeQA Pro",
          "score": 62.4,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 84.3,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "Toolathlon Verified",
          "score": 78.4,
          "unit": "percent",
          "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1471.9,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "glm-5-2",
      "name": "GLM-5.2",
      "providerId": "zai",
      "releaseDate": "2026-06-24",
      "contextWindow": 1000000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1.4,
        "outputPerMTokens": 4.4,
        "cachedInputPerMTokens": 0.26,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 90,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 29.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 59.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 91.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 86.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 34.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 78.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "Agents' Last Exam (ALE-CLI)",
          "score": 23.8,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "AutomationBench v1.0.6",
          "score": 26.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "CyberGym",
          "score": 77.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 46.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "ExploitBench",
          "score": 24.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "ExploitGym 2h",
          "score": 29,
          "unit": "score",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "ExploitGym 6h",
          "score": 39,
          "unit": "score",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "FrontierSWE",
          "score": 67.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "GDPval-AA v2",
          "score": 1508,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "HLE (with tools)",
          "score": 54.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "NL2Repo",
          "score": 48.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "PostTrainBench",
          "score": 31.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "ProgramBench (Almost Solved)",
          "score": 9.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "SWE-Marathon v1.1",
          "score": 19.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 81,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Terminal-Bench 3.0",
          "score": 4.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "Toolathlon Verified",
          "score": 59.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
          "measuredAt": "2026-08-14"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 62.1,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.2",
          "measuredAt": "2026-06-24"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 81,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.2",
          "measuredAt": "2026-06-24"
        }
      ]
    },
    {
      "id": "glm-5",
      "name": "GLM-5",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 200000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1,
        "outputPerMTokens": 3.2,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 96.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 96.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 87.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 80,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 72.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1446.3,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 77.8,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5",
          "measuredAt": "2026-02-11"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 56.2,
          "unit": "percent",
          "sourceUrl": "https://docs.z.ai/guides/llm/glm-5",
          "measuredAt": "2026-02-11"
        }
      ]
    },
    {
      "id": "glm-4-7-flashx",
      "name": "GLM-4.7-FlashX",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 200000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.07,
        "outputPerMTokens": 0.4,
        "cachedInputPerMTokens": 0.01,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "gemini-3-7-flash",
      "name": "Gemini 3.7 Flash",
      "providerId": "google",
      "releaseDate": "2026-08-13",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 0.75,
        "outputPerMTokens": 3.75,
        "cachedInputPerMTokens": 0.075,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 36.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 71.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 94.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 97.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 69.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "Artificial Analysis Intelligence Index",
          "score": 56,
          "unit": "score",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "Code Arena",
          "score": 1588,
          "unit": "elo",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 65.3,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "FrontierCode 1.1",
          "score": 43.6,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "GDPVal-AA v2",
          "score": 1525,
          "unit": "elo",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "HLE-Verified",
          "score": 53.6,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "OSWorld-2.0",
          "score": 47.9,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "Terminal-bench 2.1",
          "score": 85.8,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "measuredAt": "2026-08-13"
        },
        {
          "benchmark": "Terminal-bench 3.0",
          "score": 14.9,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
          "measuredAt": "2026-08-13"
        }
      ]
    },
    {
      "id": "gemini-3-6-flash",
      "name": "Gemini 3.6 Flash",
      "providerId": "google",
      "releaseDate": "2026-07-21",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 0.75,
        "outputPerMTokens": 3.75,
        "cachedInputPerMTokens": 0.075,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 96.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 22,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 58.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 94.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 94.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 66.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "CharXiv Reasoning (no tools)",
          "score": 85.2,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
          "measuredAt": "2026-07-21"
        },
        {
          "benchmark": "DeepSWE v1.1",
          "score": 49,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
          "measuredAt": "2026-07-21"
        },
        {
          "benchmark": "GDM-MRCR v2 (128k average)",
          "score": 91.8,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
          "measuredAt": "2026-07-21"
        },
        {
          "benchmark": "GDPVal-AA v2",
          "score": 1421,
          "unit": "elo",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
          "measuredAt": "2026-07-21"
        },
        {
          "benchmark": "MLE-Bench",
          "score": 63.9,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
          "measuredAt": "2026-07-21"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 83,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
          "measuredAt": "2026-07-21"
        },
        {
          "benchmark": "SWE-Bench Pro (Public)",
          "score": 58.7,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
          "measuredAt": "2026-07-21"
        },
        {
          "benchmark": "Terminal-bench 2.1",
          "score": 78,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
          "measuredAt": "2026-07-21"
        }
      ]
    },
    {
      "id": "gemini-3-5-flash",
      "name": "Gemini 3.5 Flash",
      "providerId": "google",
      "releaseDate": "2026-05-19",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 1.5,
        "outputPerMTokens": 9,
        "cachedInputPerMTokens": 0.15,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 95,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 26.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 62.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 92.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 95.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 66.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 79.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "ARC-AGI-2",
          "score": 72.1,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
          "measuredAt": "2026-05-19"
        },
        {
          "benchmark": "GDPVal-AA",
          "score": 1656,
          "unit": "elo",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
          "measuredAt": "2026-05-19"
        },
        {
          "benchmark": "Humanity's Last Exam (full set, text + MM)",
          "score": 40.2,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
          "measuredAt": "2026-05-19"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 83.6,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
          "measuredAt": "2026-05-19"
        },
        {
          "benchmark": "MMMU-Pro",
          "score": 83.6,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
          "measuredAt": "2026-05-19"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 78.4,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
          "measuredAt": "2026-05-19"
        },
        {
          "benchmark": "SWE-Bench Pro (Public)",
          "score": 55.1,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
          "measuredAt": "2026-05-19"
        },
        {
          "benchmark": "Terminal-bench 2.1",
          "score": 76.2,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
          "measuredAt": "2026-05-19"
        }
      ]
    },
    {
      "id": "gemini-3-1-flash-lite",
      "name": "Gemini 3.1 Flash-Lite",
      "providerId": "google",
      "releaseDate": "2026-03-03",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": 0.25,
        "outputPerMTokens": 1.5,
        "cachedInputPerMTokens": 0.025,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 27.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 81.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 80,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 86.9,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
          "measuredAt": "2026-03-03"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 16,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
          "measuredAt": "2026-03-03"
        },
        {
          "benchmark": "LiveCodeBench",
          "score": 72,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
          "measuredAt": "2026-03-03"
        },
        {
          "benchmark": "MMMLU",
          "score": 88.9,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
          "measuredAt": "2026-03-03"
        },
        {
          "benchmark": "MMMU-Pro",
          "score": 76.8,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
          "measuredAt": "2026-03-03"
        },
        {
          "benchmark": "MRCR v2 (128k average)",
          "score": 60.1,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
          "measuredAt": "2026-03-03"
        },
        {
          "benchmark": "SimpleQA Verified",
          "score": 43.3,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
          "measuredAt": "2026-03-03"
        }
      ]
    },
    {
      "id": "gemini-3-flash-preview",
      "name": "Gemini 3 Flash Preview",
      "providerId": "google",
      "releaseDate": "2025-12-17",
      "contextWindow": 1048576,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision",
        "audio"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 17.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 51.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 89.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 95.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 66.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 75.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2025 (no tools)",
          "score": 95.2,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-flash/",
          "measuredAt": "2025-12-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 90.4,
          "unit": "percent",
          "sourceUrl": "https://blog.google/products-and-platforms/products/gemini/gemini-3-flash/",
          "measuredAt": "2025-12-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 33.7,
          "unit": "percent",
          "sourceUrl": "https://blog.google/products-and-platforms/products/gemini/gemini-3-flash/",
          "measuredAt": "2025-12-17"
        },
        {
          "benchmark": "MMMU-Pro",
          "score": 81.2,
          "unit": "percent",
          "sourceUrl": "https://blog.google/products-and-platforms/products/gemini/gemini-3-flash/",
          "measuredAt": "2025-12-17"
        },
        {
          "benchmark": "SimpleQA Verified",
          "score": 68.7,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/technologies/gemini/flash/",
          "measuredAt": "2025-12-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 78,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/technologies/gemini/flash/",
          "measuredAt": "2025-12-17"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 47.6,
          "unit": "percent",
          "sourceUrl": "https://deepmind.google/technologies/gemini/flash/",
          "measuredAt": "2025-12-17"
        }
      ]
    },
    {
      "id": "qwen3-8-flash",
      "name": "Qwen3.8-Flash",
      "providerId": "qwen",
      "releaseDate": "2026-08-26",
      "contextWindow": 1000000,
      "maxOutput": 131072,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.113,
        "outputPerMTokens": 0.382,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "qwen3-8-flash-next",
      "name": "Qwen3.8-Flash-Next",
      "providerId": "qwen",
      "releaseDate": null,
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "Agents' Last Exam Pass@1",
          "score": 24.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "Agents' Last Exam Score",
          "score": 51.2,
          "unit": "score",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "CoWorkBench",
          "score": 73.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "DeepSWE 1.1",
          "score": 58.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 91.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "Humanity's Last Exam",
          "score": 35.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "IFBench",
          "score": 81.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "JobBench",
          "score": 55.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 91.9,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "NL2Repo-Bench",
          "score": 48.1,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "SWE-bench Multilingual",
          "score": 81,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "SWE-bench Pro",
          "score": 62.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        },
        {
          "benchmark": "Toolathlon Verified",
          "score": 73.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
          "measuredAt": "2026-09-16"
        }
      ]
    },
    {
      "id": "qwen3-8-2-4t-a95b",
      "name": "Qwen3.8 2.4T-A95B",
      "providerId": "qwen",
      "releaseDate": "2026-08-12",
      "contextWindow": 1000000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1.65,
        "outputPerMTokens": 4.951,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "BabyVision",
          "score": 82,
          "unit": "percent",
          "sourceUrl": "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/qwen3-8-2-4t-a95b",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 92.6,
          "unit": "percent",
          "sourceUrl": "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/qwen3-8-2-4t-a95b",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "OSWorld",
          "score": 86.1,
          "unit": "percent",
          "sourceUrl": "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/qwen3-8-2-4t-a95b",
          "measuredAt": "2026-08-12"
        },
        {
          "benchmark": "PaperBench",
          "score": 93,
          "unit": "percent",
          "sourceUrl": "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/qwen3-8-2-4t-a95b",
          "measuredAt": "2026-08-12"
        }
      ]
    },
    {
      "id": "qwen3-coder-next",
      "name": "Qwen3-Coder-Next",
      "providerId": "qwen",
      "releaseDate": "2026-02-03",
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://huggingface.co/Qwen/Qwen3-Coder-Next",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "SWE-Bench Pro",
          "score": 44.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3-Coder-Next",
          "measuredAt": "2026-02-03"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 70.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3-Coder-Next",
          "measuredAt": "2026-02-03"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 36.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/Qwen/Qwen3-Coder-Next",
          "measuredAt": "2026-02-03"
        }
      ]
    },
    {
      "id": "qwen3-7-max",
      "name": "Qwen3.7-Max",
      "providerId": "qwen",
      "releaseDate": "2026-05-20",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1.65,
        "outputPerMTokens": 4.951,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "qwen3-7-plus",
      "name": "Qwen3.7-Plus",
      "providerId": "qwen",
      "releaseDate": "2026-05-26",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.276,
        "outputPerMTokens": 1.101,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 87.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 93.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1454.2,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "qwen3-7-flash",
      "name": "Qwen3.7-Flash",
      "providerId": "qwen",
      "releaseDate": "2026-07-15",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.028,
        "outputPerMTokens": 0.11,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 19.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 82.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 86.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        }
      ]
    },
    {
      "id": "qwen3-6-plus",
      "name": "Qwen3.6-Plus",
      "providerId": "qwen",
      "releaseDate": "2026-04-02",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.276,
        "outputPerMTokens": 1.651,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 38.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 88.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 93.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 44.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 57.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1436.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "qwen3-6-flash",
      "name": "Qwen3.6-Flash",
      "providerId": "qwen",
      "releaseDate": "2026-04-16",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.165,
        "outputPerMTokens": 0.99,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 17.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 83.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 84.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 15.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        }
      ]
    },
    {
      "id": "qwen3-6-27b",
      "name": "Qwen3.6 27B",
      "providerId": "qwen",
      "releaseDate": "2026-04-21",
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 35.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 85.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 91.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 87.8,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 86.2,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 77.2,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        }
      ]
    },
    {
      "id": "qwen3-6-35b-a3b",
      "name": "Qwen3.6 35B-A3B",
      "providerId": "qwen",
      "releaseDate": null,
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 17.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 83.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 86.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 86,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 85.2,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 73.4,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        }
      ]
    },
    {
      "id": "qwen3-5-plus",
      "name": "Qwen3.5-Plus",
      "providerId": "qwen",
      "releaseDate": "2026-02-15",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.115,
        "outputPerMTokens": 0.688,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 84.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 86.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 25.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        }
      ]
    },
    {
      "id": "qwen3-5-flash",
      "name": "Qwen3.5-Flash",
      "providerId": "qwen",
      "releaseDate": "2026-02-23",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.029,
        "outputPerMTokens": 0.287,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 9.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 82.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 84.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 20.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1397.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "qwen3-5-397b-a17b",
      "name": "Qwen3.5 397B-A17B",
      "providerId": "qwen",
      "releaseDate": "2026-02-15",
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.172,
        "outputPerMTokens": 1.032,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 94.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 29.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 85.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 88.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1438.3,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 88.4,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 87.8,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.5",
          "measuredAt": "2026-02-15"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 76.4,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.5",
          "measuredAt": "2026-02-15"
        }
      ]
    },
    {
      "id": "qwen3-5-122b-a10b",
      "name": "Qwen3.5 122B-A10B",
      "providerId": "qwen",
      "releaseDate": null,
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.115,
        "outputPerMTokens": 0.917,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1417.9,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "qwen3-5-35b-a3b",
      "name": "Qwen3.5 35B-A3B",
      "providerId": "qwen",
      "releaseDate": null,
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.057,
        "outputPerMTokens": 0.459,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 93.3,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 83.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 54.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1395.5,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "qwen3-5-27b",
      "name": "Qwen3.5 27B",
      "providerId": "qwen",
      "releaseDate": null,
      "contextWindow": 262144,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 0.086,
        "outputPerMTokens": 0.688,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://www.alibabacloud.com/help/en/model-studio/model-pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 91.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1408.1,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 85.5,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 86.1,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 75,
          "unit": "percent",
          "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
          "measuredAt": "2026-04-21"
        }
      ]
    },
    {
      "id": "glm-5-1",
      "name": "GLM-5.1",
      "providerId": "zai",
      "releaseDate": "2026-04-07",
      "contextWindow": 200000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1.4,
        "outputPerMTokens": 4.4,
        "cachedInputPerMTokens": 0.26,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 95.8,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 36.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 89.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 93.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 34,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 74.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1462.4,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "glm-5-turbo",
      "name": "GLM-5-Turbo",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 200000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1.2,
        "outputPerMTokens": 4,
        "cachedInputPerMTokens": 0.24,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "glm-4-7",
      "name": "GLM-4.7",
      "providerId": "zai",
      "releaseDate": "2025-12-22",
      "contextWindow": 200000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.6,
        "outputPerMTokens": 2.2,
        "cachedInputPerMTokens": 0.11,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 83.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 83.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 32.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1435.9,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "glm-4-7-flash",
      "name": "GLM-4.7-Flash",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 200000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0,
        "outputPerMTokens": 0,
        "cachedInputPerMTokens": 0,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 60.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 58.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1352.2,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "glm-4-6",
      "name": "GLM-4.6",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 200000,
      "maxOutput": 131072,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.6,
        "outputPerMTokens": 2.2,
        "cachedInputPerMTokens": 0.11,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 91.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1440.5,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "glm-4-5",
      "name": "GLM-4.5",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 128000,
      "maxOutput": 98304,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.6,
        "outputPerMTokens": 2.2,
        "cachedInputPerMTokens": 0.11,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 93.3,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1430.2,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "glm-4-5-x",
      "name": "GLM-4.5-X",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 128000,
      "maxOutput": 98304,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 2.2,
        "outputPerMTokens": 8.9,
        "cachedInputPerMTokens": 0.45,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "glm-4-5-air",
      "name": "GLM-4.5-Air",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 128000,
      "maxOutput": 98304,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.2,
        "outputPerMTokens": 1.1,
        "cachedInputPerMTokens": 0.03,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 83.3,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1383.6,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "glm-4-5-airx",
      "name": "GLM-4.5-AirX",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 128000,
      "maxOutput": 98304,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 1.1,
        "outputPerMTokens": 4.5,
        "cachedInputPerMTokens": 0.22,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "glm-4-5-flash",
      "name": "GLM-4.5-Flash",
      "providerId": "zai",
      "releaseDate": null,
      "contextWindow": 200000,
      "maxOutput": 98304,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0,
        "outputPerMTokens": 0,
        "cachedInputPerMTokens": 0,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "glm-4-32b-0414-128k",
      "name": "GLM-4 32B 0414 128K",
      "providerId": "zai",
      "releaseDate": "2025-04-14",
      "contextWindow": 131072,
      "maxOutput": 16384,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": 0.1,
        "outputPerMTokens": 0.1,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://docs.z.ai/guides/overview/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": []
    },
    {
      "id": "ernie-5-1",
      "name": "ERNIE 5.1",
      "providerId": "baidu",
      "releaseDate": "2026-05-09",
      "contextWindow": 128000,
      "maxOutput": 65536,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1468.2,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AIME 2026 (with tools)",
          "score": 99.6,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie-5.1-0508-release/",
          "measuredAt": "2026-05-09"
        }
      ]
    },
    {
      "id": "ernie-5-0",
      "name": "ERNIE 5.0",
      "providerId": "baidu",
      "releaseDate": "2026-02-06",
      "contextWindow": 128000,
      "maxOutput": 65536,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://cloud.baidu.com/doc/qianfan/s/wmh4sv6ya",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "ACEBench Chinese",
          "score": 89.6,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "ACEBench English",
          "score": 87.7,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "AIME 2025",
          "score": 89.06,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "BBEH",
          "score": 66.63,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "BFCL v4",
          "score": 66.47,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "BrowseComp-ZH",
          "score": 64.71,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "ChineseSimpleQA",
          "score": 86.03,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 86.36,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "HMMT 2025",
          "score": 79.58,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "HumanEval+",
          "score": 94.48,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "Humanity's Last Exam",
          "score": 25.81,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "IFEval",
          "score": 93.35,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "LiveCodeBench v6",
          "score": 76.21,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "MBPP+",
          "score": 82.54,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 83.8,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "Multi-IF",
          "score": 85.56,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "MultiChallenge",
          "score": 65.98,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "SimpleQA",
          "score": 74.01,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "SpreadsheetBench",
          "score": 40.08,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "Tau2-Bench",
          "score": 78.79,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        },
        {
          "benchmark": "ZebraLogic",
          "score": 96.5,
          "unit": "percent",
          "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
          "measuredAt": "2026-02-06"
        }
      ]
    },
    {
      "id": "nvidia-nemotron-3-5-lightning-30b-a3b-nvfp4",
      "name": "NVIDIA Nemotron 3.5 Lightning 30B-A3B NVFP4",
      "providerId": "nvidia",
      "releaseDate": "2026-08-11",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1330.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "BrowseComp",
          "score": 36.81,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "measuredAt": "2026-08-11"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 75.57,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "measuredAt": "2026-08-11"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 10.47,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "measuredAt": "2026-08-11"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 81.62,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "measuredAt": "2026-08-11"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 52.8,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "measuredAt": "2026-08-11"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 23.46,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
          "measuredAt": "2026-08-11"
        }
      ]
    },
    {
      "id": "nvidia-nemotron-3-ultra-550b-a55b-nvfp4",
      "name": "NVIDIA Nemotron 3 Ultra 550B-A55B NVFP4",
      "providerId": "nvidia",
      "releaseDate": "2026-06-04",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1444.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "BrowseComp",
          "score": 41.4,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
          "measuredAt": "2026-06-04"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 87.9,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
          "measuredAt": "2026-06-04"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 26.1,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
          "measuredAt": "2026-06-04"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 69.7,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
          "measuredAt": "2026-06-04"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 53.9,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
          "measuredAt": "2026-06-04"
        }
      ]
    },
    {
      "id": "nvidia-nemotron-3-super-120b-a12b",
      "name": "NVIDIA Nemotron 3 Super 120B-A12B",
      "providerId": "nvidia",
      "releaseDate": "2026-03-11",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1377.9,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AIME 2025 (no tools)",
          "score": 90.21,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
          "measuredAt": "2026-03-11"
        },
        {
          "benchmark": "GPQA (no tools)",
          "score": 79.23,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
          "measuredAt": "2026-03-11"
        },
        {
          "benchmark": "GPQA (with tools)",
          "score": 82.7,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
          "measuredAt": "2026-03-11"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 18.26,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
          "measuredAt": "2026-03-11"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 22.82,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
          "measuredAt": "2026-03-11"
        },
        {
          "benchmark": "LiveCodeBench",
          "score": 81.19,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
          "measuredAt": "2026-03-11"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 83.73,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
          "measuredAt": "2026-03-11"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 60.47,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
          "measuredAt": "2026-03-11"
        },
        {
          "benchmark": "Terminal-Bench Core 2.0",
          "score": 31,
          "unit": "percent",
          "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
          "measuredAt": "2026-03-11"
        }
      ]
    },
    {
      "id": "nvidia-nemotron-3-nano-30b-a3b-bf16",
      "name": "NVIDIA Nemotron 3 Nano 30B-A3B BF16",
      "providerId": "nvidia",
      "releaseDate": "2025-12-15",
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1348.4,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "AIME 2025 (no tools)",
          "score": 89.1,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
          "measuredAt": "2025-12-15"
        },
        {
          "benchmark": "AIME 2025 (with tools)",
          "score": 99.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
          "measuredAt": "2025-12-15"
        },
        {
          "benchmark": "GPQA (no tools)",
          "score": 73,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
          "measuredAt": "2025-12-15"
        },
        {
          "benchmark": "GPQA (with tools)",
          "score": 75,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
          "measuredAt": "2025-12-15"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 10.6,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
          "measuredAt": "2025-12-15"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 15.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
          "measuredAt": "2025-12-15"
        },
        {
          "benchmark": "LiveCodeBench",
          "score": 68.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
          "measuredAt": "2025-12-15"
        },
        {
          "benchmark": "MMLU-Pro",
          "score": 78.3,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
          "measuredAt": "2025-12-15"
        },
        {
          "benchmark": "SWE-Bench (OpenHands)",
          "score": 38.8,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
          "measuredAt": "2025-12-15"
        }
      ]
    },
    {
      "id": "mimo-v2-5-pro",
      "name": "MiMo-V2.5-Pro",
      "providerId": "xiaomi",
      "releaseDate": null,
      "contextWindow": null,
      "maxOutput": null,
      "modalities": [
        "text"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://platform.xiaomimimo.com/docs/en-US/news/v2.5-tts-release",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1464.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        }
      ]
    },
    {
      "id": "mimo-v2-5",
      "name": "MiMo-V2.5",
      "providerId": "xiaomi",
      "releaseDate": null,
      "contextWindow": 1000000,
      "maxOutput": null,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": null,
        "outputPerMTokens": null,
        "cachedInputPerMTokens": null,
        "sourceUrl": "https://platform.xiaomimimo.com/docs/en-US/news/v2.5-tts-release",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "LMArena Elo",
          "score": 1427.4,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "ChartQA",
          "score": 81.4,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "DocVQA",
          "score": 93.2,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "InfoVQA",
          "score": 72.7,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "MMMU",
          "score": 55.1,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "MMMU-Pro",
          "score": 38.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        },
        {
          "benchmark": "ScienceQA Visual",
          "score": 97.5,
          "unit": "percent",
          "sourceUrl": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct",
          "measuredAt": "2025-02-26"
        }
      ]
    }
  ],
  "meta": {
    "source": "https://llmmetric.com",
    "attribution": "Please cite llmmetric.com. Upstream records remain subject to their source terms.",
    "generatedAt": "2026-09-18T05:14:29.930Z"
  }
}