{
  "count": 14,
  "providers": [
    "openai",
    "anthropic",
    "google",
    "deepseek",
    "meta",
    "xai",
    "mistral",
    "qwen",
    "cohere",
    "amazon",
    "microsoft",
    "moonshot",
    "zai",
    "minimax",
    "perplexity",
    "baidu",
    "nvidia",
    "xiaomi",
    "bytedance",
    "stepfun",
    "meituan",
    "tencent",
    "upstage"
  ],
  "data": [
    {
      "id": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "providerId": "anthropic",
      "releaseDate": "2026-09-01",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 10,
        "outputPerMTokens": 50,
        "cachedInputPerMTokens": 0.25,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 87.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 90.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 70.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "CursorBench 3.2.0",
          "score": 73.4,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "GDPval-AA v2",
          "score": 1853,
          "unit": "elo",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 60.9,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "Terminal-Bench Science 0.1",
          "score": 52.6,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        }
      ]
    },
    {
      "id": "claude-opus-5",
      "name": "Claude Opus 5",
      "providerId": "anthropic",
      "releaseDate": "2026-07-24",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 5,
        "outputPerMTokens": 25,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 73.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 85.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 93.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 98.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 59.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "CursorBench 3.2.0",
          "score": 70,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "GDPval-AA v2",
          "score": 1824,
          "unit": "elo",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 56.6,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        },
        {
          "benchmark": "Terminal-Bench Science 0.1",
          "score": 29,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
          "measuredAt": "2026-09-01"
        }
      ]
    },
    {
      "id": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "providerId": "anthropic",
      "releaseDate": "2026-06-30",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 2,
        "outputPerMTokens": 10,
        "cachedInputPerMTokens": 0.2,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 29.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 65.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 94.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 33.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 43.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/283ef97c476cf442c91d9a37d5b214242a55bb92/Claude%20Sonnet%205%20System%20Card.pdf",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 57.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/283ef97c476cf442c91d9a37d5b214242a55bb92/Claude%20Sonnet%205%20System%20Card.pdf",
          "measuredAt": "2026-06-30"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 85.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/283ef97c476cf442c91d9a37d5b214242a55bb92/Claude%20Sonnet%205%20System%20Card.pdf",
          "measuredAt": "2026-06-30"
        }
      ]
    },
    {
      "id": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "providerId": "anthropic",
      "releaseDate": "2025-10-15",
      "contextWindow": 200000,
      "maxOutput": 64000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 1,
        "outputPerMTokens": 5,
        "cachedInputPerMTokens": 0.1,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 71.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 66.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 13.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 73.3,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-haiku-4-5",
          "measuredAt": "2025-10-15"
        }
      ]
    },
    {
      "id": "claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "providerId": "anthropic",
      "releaseDate": "2026-05-28",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 5,
        "outputPerMTokens": 25,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://platform.claude.com/docs/en/models/opus-4-8/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 100,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 56.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 80,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 91,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 98.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 53,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1452.7,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "BrowseComp",
          "score": 84.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 93.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 49.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 57.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 83.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Multilingual",
          "score": 84.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Multimodal",
          "score": 38.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 69.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 88.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 74.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        }
      ]
    },
    {
      "id": "claude-opus-4-7",
      "name": "Claude Opus 4.7",
      "providerId": "anthropic",
      "releaseDate": "2026-04-16",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 5,
        "outputPerMTokens": 25,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://platform.claude.com/docs/en/models/opus-4-7/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 95.8,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 31.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 70.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 97.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 51.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 83.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1483.4,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "BrowseComp",
          "score": 79.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 94.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 46.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 54.7,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 82.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Multilingual",
          "score": 80.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Multimodal",
          "score": 34.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-Bench Pro",
          "score": 64.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 87.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        },
        {
          "benchmark": "Terminal-Bench 2.1",
          "score": 66.1,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
          "measuredAt": "2026-05-28"
        }
      ]
    },
    {
      "id": "claude-opus-4-6",
      "name": "Claude Opus 4.6",
      "providerId": "anthropic",
      "releaseDate": "2026-02-05",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 5,
        "outputPerMTokens": 25,
        "cachedInputPerMTokens": 0.5,
        "sourceUrl": "https://platform.claude.com/docs/en/models/opus-4-6/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2026 (MathArena)",
          "score": 96.7,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2026",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 26.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 66,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 90.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 94.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 47,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 78.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1497.5,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 68.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 91.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 59.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "MMMLU",
          "score": 91.1,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "MMMU-Pro (no tools)",
          "score": 73.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 77.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 72.7,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 80.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "Tau2-bench Retail",
          "score": 91.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 99.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 65.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
          "measuredAt": "2026-02-05"
        }
      ]
    },
    {
      "id": "claude-sonnet-4-6",
      "name": "Claude Sonnet 4.6",
      "providerId": "anthropic",
      "releaseDate": "2026-02-17",
      "contextWindow": 1000000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://platform.claude.com/docs/en/models/sonnet-4-6/overview",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 87.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 85.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 35.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 75.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "LMArena Elo",
          "score": 1458.3,
          "unit": "elo",
          "sourceUrl": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset",
          "measuredAt": "2026-09-13"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 58.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "GDPval-AA",
          "score": 1633,
          "unit": "elo",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 89.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 33.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 49,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 61.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMLU",
          "score": 89.3,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMU-Pro (no tools)",
          "score": 74.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 75.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 72.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 79.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Tau2-bench Retail",
          "score": 91.7,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 97.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 59.1,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        }
      ]
    },
    {
      "id": "claude-sonnet-4-5",
      "name": "Claude Sonnet 4.5",
      "providerId": "anthropic",
      "releaseDate": "2025-09-29",
      "contextWindow": 200000,
      "maxOutput": 64000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://www.anthropic.com/news/claude-sonnet-4-5",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 84.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 2.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 23.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 82.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 77.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SimpleQA Verified (Epoch AI run)",
          "score": 30.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 71.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "ARC-AGI-2 (Verified)",
          "score": 13.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "GDPval-AA",
          "score": 1276,
          "unit": "elo",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "GPQA Diamond",
          "score": 83.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Humanity's Last Exam (no tools)",
          "score": 17.7,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Humanity's Last Exam (with tools)",
          "score": 33.6,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MCP Atlas",
          "score": 43.8,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMLU",
          "score": 89.5,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMU-Pro (no tools)",
          "score": 63.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "MMMU-Pro (with tools)",
          "score": 68.9,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "OSWorld-Verified",
          "score": 61.4,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 77.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Tau2-bench Retail",
          "score": 86.2,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Tau2-bench Telecom",
          "score": 98,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        },
        {
          "benchmark": "Terminal-Bench 2.0",
          "score": 51,
          "unit": "percent",
          "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
          "measuredAt": "2026-02-17"
        }
      ]
    },
    {
      "id": "claude-opus-4-1",
      "name": "Claude Opus 4.1",
      "providerId": "anthropic",
      "releaseDate": "2025-08-05",
      "contextWindow": 200000,
      "maxOutput": 32000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 15,
        "outputPerMTokens": 75,
        "cachedInputPerMTokens": 1.5,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "FrontierMath Tier 4 v2 (Epoch AI run)",
          "score": 2.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
          "score": 12.6,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 77.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 68.9,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 73.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 74.5,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-opus-4-1",
          "measuredAt": "2025-08-05"
        }
      ]
    },
    {
      "id": "claude-opus-4",
      "name": "Claude Opus 4",
      "providerId": "anthropic",
      "releaseDate": "2025-05-22",
      "contextWindow": 200000,
      "maxOutput": 32000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 15,
        "outputPerMTokens": 75,
        "cachedInputPerMTokens": 1.5,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 76.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 64.4,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 70.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2024 (no extended thinking)",
          "score": 33.9,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 72.5,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        },
        {
          "benchmark": "Terminal-bench",
          "score": 43.2,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        }
      ]
    },
    {
      "id": "claude-sonnet-4",
      "name": "Claude Sonnet 4",
      "providerId": "anthropic",
      "releaseDate": "2025-05-22",
      "contextWindow": 200000,
      "maxOutput": 64000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 79.2,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 71.1,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "AIME 2024 (no extended thinking)",
          "score": 33.1,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 72.7,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-4",
          "measuredAt": "2025-05-22"
        }
      ]
    },
    {
      "id": "claude-sonnet-3-7",
      "name": "Claude 3.7 Sonnet",
      "providerId": "anthropic",
      "releaseDate": "2025-02-24",
      "contextWindow": 200000,
      "maxOutput": 128000,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://www.anthropic.com/news/claude-3-7-sonnet",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 49.2,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 79.7,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 57.8,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified (Epoch AI run)",
          "score": 61,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/swe-bench-verified",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 70.3,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-3-7-sonnet",
          "measuredAt": "2025-02-24"
        }
      ]
    },
    {
      "id": "claude-sonnet-3-5",
      "name": "Claude 3.5 Sonnet",
      "providerId": "anthropic",
      "releaseDate": "2024-10-22",
      "contextWindow": 200000,
      "maxOutput": 8192,
      "modalities": [
        "text",
        "vision"
      ],
      "pricing": {
        "inputPerMTokens": 3,
        "outputPerMTokens": 15,
        "cachedInputPerMTokens": 0.3,
        "sourceUrl": "https://www.anthropic.com/news/claude-3-5-sonnet",
        "updatedAt": "2026-09-16"
      },
      "benchmarks": [
        {
          "benchmark": "AIME 2025 (MathArena)",
          "score": 3.3,
          "unit": "percent",
          "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "GPQA Diamond (Epoch AI run)",
          "score": 55.3,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "OTIS Mock AIME 2024-2025 (Epoch AI run)",
          "score": 8.5,
          "unit": "percent",
          "sourceUrl": "https://epoch.ai/benchmarks/otis-mock-aime-2024-2025",
          "measuredAt": "2026-09-17"
        },
        {
          "benchmark": "SWE-bench Verified",
          "score": 49,
          "unit": "percent",
          "sourceUrl": "https://www.anthropic.com/news/claude-3-5-sonnet",
          "measuredAt": "2024-10-22"
        }
      ]
    }
  ],
  "meta": {
    "source": "https://llmmetric.com",
    "attribution": "Please cite llmmetric.com. Upstream records remain subject to their source terms.",
    "generatedAt": "2026-09-18T05:27:24.150Z"
  }
}