{
  "data": {
    "version": "beta-1",
    "methodologyUrl": "/methodology#llmmetric-index",
    "minPeers": 3,
    "activeAxes": [
      {
        "id": "swe-verified",
        "name": "SWE-bench Verified",
        "category": "Coding",
        "peers": 48
      },
      {
        "id": "swe-pro",
        "name": "SWE-bench Pro",
        "category": "Coding",
        "peers": 22
      },
      {
        "id": "deepswe-1-1",
        "name": "DeepSWE v1.1",
        "category": "Coding",
        "peers": 16
      },
      {
        "id": "terminal-science-0-1",
        "name": "Terminal-Bench Science 0.1",
        "category": "Coding",
        "peers": 4
      },
      {
        "id": "terminal-4",
        "name": "Terminal-Bench 4.0",
        "category": "Coding",
        "peers": 3
      },
      {
        "id": "livecodebench-6",
        "name": "LiveCodeBench v6",
        "category": "Coding",
        "peers": 11
      },
      {
        "id": "gpqa",
        "name": "GPQA Diamond",
        "category": "Reasoning",
        "peers": 111
      },
      {
        "id": "mmlu-pro",
        "name": "MMLU-Pro",
        "category": "Reasoning",
        "peers": 29
      },
      {
        "id": "aime-2025",
        "name": "AIME 2025",
        "category": "Reasoning",
        "peers": 40
      },
      {
        "id": "hle-no-tools",
        "name": "Humanity's Last Exam (no tools)",
        "category": "Reasoning",
        "peers": 29
      },
      {
        "id": "hle-with-tools",
        "name": "Humanity's Last Exam (with tools)",
        "category": "Reasoning",
        "peers": 14
      },
      {
        "id": "frontiermath-1-3-epoch",
        "name": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
        "category": "Reasoning",
        "peers": 59
      },
      {
        "id": "automation-1-0-6",
        "name": "AutomationBench 1.0.6",
        "category": "Agents",
        "peers": 3
      },
      {
        "id": "browsecomp",
        "name": "BrowseComp",
        "category": "Agents",
        "peers": 15
      },
      {
        "id": "mcp-atlas",
        "name": "MCP Atlas",
        "category": "Agents",
        "peers": 9
      },
      {
        "id": "tau2-telecom",
        "name": "Tau2-bench Telecom",
        "category": "Agents",
        "peers": 8
      },
      {
        "id": "toolathlon",
        "name": "Toolathlon",
        "category": "Agents",
        "peers": 6
      },
      {
        "id": "osworld-verified",
        "name": "OSWorld-Verified",
        "category": "Agents",
        "peers": 12
      }
    ],
    "ranked": [
      {
        "modelId": "gpt-6-astra",
        "name": "GPT-6 Astra",
        "providerId": "openai",
        "releaseDate": "2026-09-03",
        "rank": 1,
        "score": 90.1,
        "categoryScores": {
          "Coding": 87.5,
          "Reasoning": 99.4,
          "Agents": 83.3
        },
        "coverage": 4,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "terminal-science-0-1",
            "benchmark": "Terminal-Bench Science 0.1",
            "category": "Coding",
            "rawScore": 64.6,
            "percentile": 87.5,
            "peerCount": 4,
            "sourceUrl": "https://openai.com/index/gpt-6-astra/",
            "measuredAt": "2026-09-03"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 96,
            "percentile": 99.5,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-6-astra/",
            "measuredAt": "2026-09-03"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 93.7,
            "percentile": 99.2,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "automation-1-0-6",
            "benchmark": "AutomationBench 1.0.6",
            "category": "Agents",
            "rawScore": 41.4,
            "percentile": 83.3,
            "peerCount": 3,
            "sourceUrl": "https://zapier.com/benchmarks",
            "measuredAt": "2026-09-03"
          }
        ]
      },
      {
        "modelId": "claude-opus-4-8",
        "name": "Claude Opus 4.8",
        "providerId": "anthropic",
        "releaseDate": "2026-05-28",
        "rank": 2,
        "score": 88.6,
        "categoryScores": {
          "Coding": 98.4,
          "Reasoning": 85.2,
          "Agents": 82.1
        },
        "coverage": 8,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 88.6,
            "percentile": 99,
            "peerCount": 48,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 69.2,
            "percentile": 97.7,
            "peerCount": 22,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 93.6,
            "percentile": 90.5,
            "peerCount": 111,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 49.8,
            "percentile": 91.4,
            "peerCount": 29,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 57.9,
            "percentile": 75,
            "peerCount": 14,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 80,
            "percentile": 83.9,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 84.3,
            "percentile": 76.7,
            "peerCount": 15,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 83.4,
            "percentile": 87.5,
            "peerCount": 12,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          }
        ]
      },
      {
        "modelId": "kimi-k3",
        "name": "Kimi K3",
        "providerId": "moonshot",
        "releaseDate": "2026-07-16",
        "rank": 3,
        "score": 77.1,
        "categoryScores": {
          "Coding": 59.4,
          "Reasoning": 76.2,
          "Agents": 95.8
        },
        "coverage": 6,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 67.3,
            "percentile": 59.4,
            "peerCount": 16,
            "sourceUrl": "https://www.kimi.com/en/blog/kimi-k3",
            "measuredAt": "2026-07-16"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 93.5,
            "percentile": 89.6,
            "peerCount": 111,
            "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
            "measuredAt": "2026-07-27"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 43.5,
            "percentile": 84.5,
            "peerCount": 29,
            "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
            "measuredAt": "2026-07-27"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 56,
            "percentile": 53.6,
            "peerCount": 14,
            "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
            "measuredAt": "2026-07-27"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 72.2,
            "percentile": 77.1,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 84.8,
            "percentile": 95.8,
            "peerCount": 12,
            "sourceUrl": "https://github.com/MoonshotAI/Kimi-K3",
            "measuredAt": "2026-07-27"
          }
        ]
      },
      {
        "modelId": "claude-opus-4-7",
        "name": "Claude Opus 4.7",
        "providerId": "anthropic",
        "releaseDate": "2026-04-16",
        "rank": 4,
        "score": 75.5,
        "categoryScores": {
          "Coding": 90.5,
          "Reasoning": 75.5,
          "Agents": 60.4
        },
        "coverage": 8,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 87.6,
            "percentile": 96.9,
            "peerCount": 48,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 64.3,
            "percentile": 84.1,
            "peerCount": 22,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 94.2,
            "percentile": 94.1,
            "peerCount": 111,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 46.9,
            "percentile": 87.9,
            "peerCount": 29,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 54.7,
            "percentile": 46.4,
            "peerCount": 14,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 70.2,
            "percentile": 73.7,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 79.8,
            "percentile": 50,
            "peerCount": 15,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 82.8,
            "percentile": 70.8,
            "peerCount": 12,
            "sourceUrl": "https://www-cdn.anthropic.com/0b4915911bb0d19eca5b5ee635c80fef830a37ea.pdf",
            "measuredAt": "2026-05-28"
          }
        ]
      },
      {
        "modelId": "claude-opus-4-6",
        "name": "Claude Opus 4.6",
        "providerId": "anthropic",
        "releaseDate": "2026-02-05",
        "rank": 5,
        "score": 73.8,
        "categoryScores": {
          "Coding": 92.7,
          "Reasoning": 72,
          "Agents": 56.7
        },
        "coverage": 6,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 80.8,
            "percentile": 92.7,
            "peerCount": 48,
            "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
            "measuredAt": "2026-02-05"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 91.3,
            "percentile": 77.9,
            "peerCount": 111,
            "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
            "measuredAt": "2026-02-05"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 66,
            "percentile": 66.1,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "mcp-atlas",
            "benchmark": "MCP Atlas",
            "category": "Agents",
            "rawScore": 59.5,
            "percentile": 38.9,
            "peerCount": 9,
            "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
            "measuredAt": "2026-02-05"
          },
          {
            "axisId": "tau2-telecom",
            "benchmark": "Tau2-bench Telecom",
            "category": "Agents",
            "rawScore": 99.3,
            "percentile": 93.8,
            "peerCount": 8,
            "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
            "measuredAt": "2026-02-05"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 72.7,
            "percentile": 37.5,
            "peerCount": 12,
            "sourceUrl": "https://www-cdn.anthropic.com/14e4fb01875d2a69f646fa5e574dea2b1c0ff7b5.pdf",
            "measuredAt": "2026-02-05"
          }
        ]
      },
      {
        "modelId": "claude-opus-5-5",
        "name": "Claude Opus 5.5",
        "providerId": "anthropic",
        "releaseDate": "2026-09-22",
        "rank": 6,
        "score": 73.1,
        "categoryScores": {
          "Coding": 72.9,
          "Reasoning": 96.4,
          "Agents": 50
        },
        "coverage": 4,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "terminal-science-0-1",
            "benchmark": "Terminal-Bench Science 0.1",
            "category": "Coding",
            "rawScore": 58.7,
            "percentile": 62.5,
            "peerCount": 4,
            "sourceUrl": "https://www.anthropic.com/claude-opus-5-5",
            "measuredAt": "2026-09-22"
          },
          {
            "axisId": "terminal-4",
            "benchmark": "Terminal-Bench 4.0",
            "category": "Coding",
            "rawScore": 66.4,
            "percentile": 83.3,
            "peerCount": 3,
            "sourceUrl": "https://www.anthropic.com/claude-opus-5-5",
            "measuredAt": "2026-09-22"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 67.7,
            "percentile": 96.4,
            "peerCount": 14,
            "sourceUrl": "https://www.anthropic.com/claude-opus-5-5",
            "measuredAt": "2026-09-22"
          },
          {
            "axisId": "automation-1-0-6",
            "benchmark": "AutomationBench 1.0.6",
            "category": "Agents",
            "rawScore": 40,
            "percentile": 50,
            "peerCount": 3,
            "sourceUrl": "https://zapier.com/benchmarks",
            "measuredAt": "2026-09-22"
          }
        ]
      },
      {
        "modelId": "gpt-5-5",
        "name": "GPT-5.5",
        "providerId": "openai",
        "releaseDate": "2026-04-23",
        "rank": 7,
        "score": 69.2,
        "categoryScores": {
          "Coding": 52.3,
          "Reasoning": 80.8,
          "Agents": 74.4
        },
        "coverage": 8,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 58.6,
            "percentile": 52.3,
            "peerCount": 22,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
            "measuredAt": "2026-04-23"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.7,
            "percentile": 72.5,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 85.3,
            "percentile": 89,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 84.4,
            "percentile": 83.3,
            "peerCount": 15,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
            "measuredAt": "2026-04-23"
          },
          {
            "axisId": "mcp-atlas",
            "benchmark": "MCP Atlas",
            "category": "Agents",
            "rawScore": 75.3,
            "percentile": 72.2,
            "peerCount": 9,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
            "measuredAt": "2026-04-23"
          },
          {
            "axisId": "tau2-telecom",
            "benchmark": "Tau2-bench Telecom",
            "category": "Agents",
            "rawScore": 98,
            "percentile": 62.5,
            "peerCount": 8,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
            "measuredAt": "2026-04-23"
          },
          {
            "axisId": "toolathlon",
            "benchmark": "Toolathlon",
            "category": "Agents",
            "rawScore": 55.6,
            "percentile": 91.7,
            "peerCount": 6,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
            "measuredAt": "2026-04-23"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 78.7,
            "percentile": 62.5,
            "peerCount": 12,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
            "measuredAt": "2026-04-23"
          }
        ]
      },
      {
        "modelId": "hy3",
        "name": "Hy3",
        "providerId": "tencent",
        "releaseDate": "2026-07-06",
        "rank": 8,
        "score": 66.8,
        "categoryScores": {
          "Coding": 66.6,
          "Reasoning": 68.8,
          "Agents": 65
        },
        "coverage": 7,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 78,
            "percentile": 85.4,
            "peerCount": 48,
            "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
            "measuredAt": "2026-07-06"
          },
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 57.9,
            "percentile": 47.7,
            "peerCount": 22,
            "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
            "measuredAt": "2026-07-06"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.4,
            "percentile": 70.3,
            "peerCount": 111,
            "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
            "measuredAt": "2026-07-06"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 37,
            "percentile": 67.2,
            "peerCount": 29,
            "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
            "measuredAt": "2026-07-06"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 84.2,
            "percentile": 70,
            "peerCount": 15,
            "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
            "measuredAt": "2026-07-06"
          },
          {
            "axisId": "mcp-atlas",
            "benchmark": "MCP Atlas",
            "category": "Agents",
            "rawScore": 79.1,
            "percentile": 83.3,
            "peerCount": 9,
            "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
            "measuredAt": "2026-07-06"
          },
          {
            "axisId": "toolathlon",
            "benchmark": "Toolathlon",
            "category": "Agents",
            "rawScore": 48.5,
            "percentile": 41.7,
            "peerCount": 6,
            "sourceUrl": "https://github.com/Tencent-Hunyuan/Hy3",
            "measuredAt": "2026-07-06"
          }
        ]
      },
      {
        "modelId": "step-3-5-flash",
        "name": "Step 3.5 Flash",
        "providerId": "stepfun",
        "releaseDate": "2026-02-12",
        "rank": 9,
        "score": 59.8,
        "categoryScores": {
          "Coding": 64.9,
          "Reasoning": 91.3,
          "Agents": 23.3
        },
        "coverage": 4,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 74.4,
            "percentile": 61.5,
            "peerCount": 48,
            "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
            "measuredAt": "2026-02-12"
          },
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 86.4,
            "percentile": 68.2,
            "peerCount": 11,
            "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
            "measuredAt": "2026-02-12"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 97.3,
            "percentile": 91.3,
            "peerCount": 40,
            "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
            "measuredAt": "2026-02-12"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 51.6,
            "percentile": 23.3,
            "peerCount": 15,
            "sourceUrl": "https://static.stepfun.com/blog/step-3.5-flash/",
            "measuredAt": "2026-02-12"
          }
        ]
      },
      {
        "modelId": "claude-sonnet-4-6",
        "name": "Claude Sonnet 4.6",
        "providerId": "anthropic",
        "releaseDate": "2026-02-17",
        "rank": 10,
        "score": 59.4,
        "categoryScores": {
          "Coding": 88.5,
          "Reasoning": 48.7,
          "Agents": 41
        },
        "coverage": 7,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 79.6,
            "percentile": 88.5,
            "peerCount": 48,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 89.9,
            "percentile": 67.6,
            "peerCount": 111,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 33.2,
            "percentile": 53.4,
            "peerCount": 29,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 49,
            "percentile": 25,
            "peerCount": 14,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "mcp-atlas",
            "benchmark": "MCP Atlas",
            "category": "Agents",
            "rawScore": 61.3,
            "percentile": 50,
            "peerCount": 9,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "tau2-telecom",
            "benchmark": "Tau2-bench Telecom",
            "category": "Agents",
            "rawScore": 97.9,
            "percentile": 43.8,
            "peerCount": 8,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 72.5,
            "percentile": 29.2,
            "peerCount": 12,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          }
        ]
      },
      {
        "modelId": "gpt-5-4",
        "name": "GPT-5.4",
        "providerId": "openai",
        "releaseDate": "2026-03-05",
        "rank": 11,
        "score": 59.3,
        "categoryScores": {
          "Coding": 43.2,
          "Reasoning": 69.5,
          "Agents": 65.3
        },
        "coverage": 10,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 57.7,
            "percentile": 43.2,
            "peerCount": 22,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 92.8,
            "percentile": 85.6,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 39.8,
            "percentile": 70.7,
            "peerCount": 29,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 52.1,
            "percentile": 39.3,
            "peerCount": 14,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 78.6,
            "percentile": 82.2,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 82.7,
            "percentile": 63.3,
            "peerCount": 15,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "mcp-atlas",
            "benchmark": "MCP Atlas",
            "category": "Agents",
            "rawScore": 67.2,
            "percentile": 61.1,
            "peerCount": 9,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "tau2-telecom",
            "benchmark": "Tau2-bench Telecom",
            "category": "Agents",
            "rawScore": 98.9,
            "percentile": 81.3,
            "peerCount": 8,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "toolathlon",
            "benchmark": "Toolathlon",
            "category": "Agents",
            "rawScore": 54.6,
            "percentile": 75,
            "peerCount": 6,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 75,
            "percentile": 45.8,
            "peerCount": 12,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          }
        ]
      },
      {
        "modelId": "longcat-2-0",
        "name": "LongCat 2.0",
        "providerId": "meituan",
        "releaseDate": "2026-06-30",
        "rank": 11,
        "score": 59.3,
        "categoryScores": {
          "Coding": 56.8,
          "Reasoning": 64.4,
          "Agents": 56.7
        },
        "coverage": 3,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 59.5,
            "percentile": 56.8,
            "peerCount": 22,
            "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
            "measuredAt": "2026-06-30"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 88.9,
            "percentile": 64.4,
            "peerCount": 111,
            "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
            "measuredAt": "2026-06-30"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 79.9,
            "percentile": 56.7,
            "peerCount": 15,
            "sourceUrl": "https://github.com/meituan-longcat/LongCat-2.0",
            "measuredAt": "2026-06-30"
          }
        ]
      },
      {
        "modelId": "gemini-3-6-flash",
        "name": "Gemini 3.6 Flash",
        "providerId": "google",
        "releaseDate": "2026-07-21",
        "rank": 13,
        "score": 54.5,
        "categoryScores": {
          "Coding": 9.4,
          "Reasoning": 75,
          "Agents": 79.2
        },
        "coverage": 4,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 49,
            "percentile": 9.4,
            "peerCount": 16,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
            "measuredAt": "2026-07-21"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 94.1,
            "percentile": 93.2,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 58.9,
            "percentile": 56.8,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 83,
            "percentile": 79.2,
            "peerCount": 12,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-6-flash/",
            "measuredAt": "2026-07-21"
          }
        ]
      },
      {
        "modelId": "dola-seed-2-0-pro",
        "name": "Dola Seed 2.0 Pro",
        "providerId": "bytedance",
        "releaseDate": "2026-02-16",
        "rank": 14,
        "score": 53.9,
        "categoryScores": {
          "Coding": 42.2,
          "Reasoning": 82.8,
          "Agents": 36.7
        },
        "coverage": 7,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 76.5,
            "percentile": 72.9,
            "peerCount": 48,
            "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
            "measuredAt": "2026-02-16"
          },
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 46.9,
            "percentile": 11.4,
            "peerCount": 22,
            "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
            "measuredAt": "2026-02-16"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 92.4,
            "percentile": 82,
            "peerCount": 111,
            "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
            "measuredAt": "2026-02-16"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 90.1,
            "percentile": 98.3,
            "peerCount": 29,
            "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
            "measuredAt": "2026-02-16"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 99,
            "percentile": 93.8,
            "peerCount": 40,
            "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
            "measuredAt": "2026-02-16"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 33.3,
            "percentile": 56.9,
            "peerCount": 29,
            "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
            "measuredAt": "2026-02-16"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 77.3,
            "percentile": 36.7,
            "peerCount": 15,
            "sourceUrl": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/seed2/0214/Seed2.0%20Model%20Card.pdf",
            "measuredAt": "2026-02-16"
          }
        ]
      },
      {
        "modelId": "solar-open-2-250b",
        "name": "Solar Open 2 250B-A15B",
        "providerId": "upstage",
        "releaseDate": "2026-07-24",
        "rank": 15,
        "score": 53.5,
        "categoryScores": {
          "Coding": 69.1,
          "Reasoning": 63.5,
          "Agents": 27.8
        },
        "coverage": 6,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 70.4,
            "percentile": 42.7,
            "peerCount": 48,
            "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
            "measuredAt": "2026-07-24"
          },
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 92.4,
            "percentile": 95.5,
            "peerCount": 11,
            "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
            "measuredAt": "2026-07-24"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 86.3,
            "percentile": 50.9,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
            "measuredAt": "2026-07-24"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 86.2,
            "percentile": 89.7,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
            "measuredAt": "2026-07-24"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 28.8,
            "percentile": 50,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
            "measuredAt": "2026-07-24"
          },
          {
            "axisId": "mcp-atlas",
            "benchmark": "MCP Atlas",
            "category": "Agents",
            "rawScore": 58.2,
            "percentile": 27.8,
            "peerCount": 9,
            "sourceUrl": "https://huggingface.co/upstage/Solar-Open2-250B",
            "measuredAt": "2026-07-24"
          }
        ]
      },
      {
        "modelId": "claude-sonnet-4-5",
        "name": "Claude Sonnet 4.5",
        "providerId": "anthropic",
        "releaseDate": "2025-09-29",
        "rank": 16,
        "score": 45.4,
        "categoryScores": {
          "Coding": 77.1,
          "Reasoning": 28.5,
          "Agents": 30.6
        },
        "coverage": 9,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 77.2,
            "percentile": 77.1,
            "peerCount": 48,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 83.4,
            "percentile": 42.8,
            "peerCount": 111,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 84.2,
            "percentile": 36.3,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 17.7,
            "percentile": 25.9,
            "peerCount": 29,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 33.6,
            "percentile": 17.9,
            "peerCount": 14,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 23.9,
            "percentile": 19.5,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "mcp-atlas",
            "benchmark": "MCP Atlas",
            "category": "Agents",
            "rawScore": 43.8,
            "percentile": 16.7,
            "peerCount": 9,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "tau2-telecom",
            "benchmark": "Tau2-bench Telecom",
            "category": "Agents",
            "rawScore": 98,
            "percentile": 62.5,
            "peerCount": 8,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 61.4,
            "percentile": 12.5,
            "peerCount": 12,
            "sourceUrl": "https://www-cdn.anthropic.com/78073f739564e986ff3e28522761a7a0b4484f84.pdf",
            "measuredAt": "2026-02-17"
          }
        ]
      },
      {
        "modelId": "gpt-5-4-mini",
        "name": "GPT-5.4 mini",
        "providerId": "openai",
        "releaseDate": "2026-03-17",
        "rank": 17,
        "score": 34.8,
        "categoryScores": {
          "Coding": 29.5,
          "Reasoning": 52,
          "Agents": 22.9
        },
        "coverage": 5,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 54.4,
            "percentile": 29.5,
            "peerCount": 22,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
            "measuredAt": "2026-03-17"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 88,
            "percentile": 59.9,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
            "measuredAt": "2026-03-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 51.2,
            "percentile": 44.1,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "toolathlon",
            "benchmark": "Toolathlon",
            "category": "Agents",
            "rawScore": 42.9,
            "percentile": 25,
            "peerCount": 6,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
            "measuredAt": "2026-03-17"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 72.1,
            "percentile": 20.8,
            "peerCount": 12,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
            "measuredAt": "2026-03-17"
          }
        ]
      },
      {
        "modelId": "nvidia-nemotron-3-ultra-550b-a55b-nvfp4",
        "name": "NVIDIA Nemotron 3 Ultra 550B-A55B NVFP4",
        "providerId": "nvidia",
        "releaseDate": "2026-06-04",
        "rank": 18,
        "score": 33.6,
        "categoryScores": {
          "Coding": 38.5,
          "Reasoning": 52.4,
          "Agents": 10
        },
        "coverage": 4,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 69.7,
            "percentile": 38.5,
            "peerCount": 48,
            "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
            "measuredAt": "2026-06-04"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 87.9,
            "percentile": 58.1,
            "peerCount": 111,
            "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
            "measuredAt": "2026-06-04"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 26.1,
            "percentile": 46.6,
            "peerCount": 29,
            "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
            "measuredAt": "2026-06-04"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 41.4,
            "percentile": 10,
            "peerCount": 15,
            "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-ultra-550b-a55b/modelcard",
            "measuredAt": "2026-06-04"
          }
        ]
      },
      {
        "modelId": "gpt-5-4-nano",
        "name": "GPT-5.4 nano",
        "providerId": "openai",
        "releaseDate": "2026-03-17",
        "rank": 19,
        "score": 20.1,
        "categoryScores": {
          "Coding": 15.9,
          "Reasoning": 38,
          "Agents": 6.3
        },
        "coverage": 5,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 52.4,
            "percentile": 15.9,
            "peerCount": 22,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
            "measuredAt": "2026-03-17"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 82.8,
            "percentile": 38.7,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
            "measuredAt": "2026-03-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 44.9,
            "percentile": 37.3,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "toolathlon",
            "benchmark": "Toolathlon",
            "category": "Agents",
            "rawScore": 35.5,
            "percentile": 8.3,
            "peerCount": 6,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
            "measuredAt": "2026-03-17"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 39,
            "percentile": 4.2,
            "peerCount": 12,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
            "measuredAt": "2026-03-17"
          }
        ]
      },
      {
        "modelId": "nvidia-nemotron-3-5-lightning-30b-a3b-nvfp4",
        "name": "NVIDIA Nemotron 3.5 Lightning 30B-A3B NVFP4",
        "providerId": "nvidia",
        "releaseDate": "2026-08-11",
        "rank": 20,
        "score": 15.4,
        "categoryScores": {
          "Coding": 13.5,
          "Reasoning": 29.3,
          "Agents": 3.3
        },
        "coverage": 5,
        "categoryCoverage": 3,
        "missingCategories": [],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 52.8,
            "percentile": 13.5,
            "peerCount": 48,
            "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
            "measuredAt": "2026-08-11"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 75.57,
            "percentile": 25.7,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
            "measuredAt": "2026-08-11"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 81.62,
            "percentile": 50,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
            "measuredAt": "2026-08-11"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 10.47,
            "percentile": 12.1,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
            "measuredAt": "2026-08-11"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 36.81,
            "percentile": 3.3,
            "peerCount": 15,
            "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4",
            "measuredAt": "2026-08-11"
          }
        ]
      }
    ],
    "unranked": [
      {
        "modelId": "gpt-6-sol",
        "name": "GPT-6 Sol",
        "providerId": "openai",
        "releaseDate": "2026-09-22",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 65.6,
          "Reasoning": null,
          "Agents": 16.7
        },
        "coverage": 2,
        "categoryCoverage": 2,
        "missingCategories": [
          "Reasoning"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 68.8,
            "percentile": 65.6,
            "peerCount": 16,
            "sourceUrl": "https://openai.com/index/introducing-gpt-6-sol-and-luna/",
            "measuredAt": "2026-09-22"
          },
          {
            "axisId": "automation-1-0-6",
            "benchmark": "AutomationBench 1.0.6",
            "category": "Agents",
            "rawScore": 33.2,
            "percentile": 16.7,
            "peerCount": 3,
            "sourceUrl": "https://zapier.com/benchmarks",
            "measuredAt": "2026-09-22"
          }
        ]
      },
      {
        "modelId": "gpt-6-luna",
        "name": "GPT-6 Luna",
        "providerId": "openai",
        "releaseDate": "2026-09-22",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 40.6,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Reasoning",
          "Agents"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 66.6,
            "percentile": 40.6,
            "peerCount": 16,
            "sourceUrl": "https://openai.com/index/introducing-gpt-6-sol-and-luna/",
            "measuredAt": "2026-09-22"
          }
        ]
      },
      {
        "modelId": "grok-4-7",
        "name": "Grok 4.7",
        "providerId": "xai",
        "releaseDate": "2026-09-21",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 64.1,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Reasoning",
          "Agents"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 71,
            "percentile": 78.1,
            "peerCount": 16,
            "sourceUrl": "https://x.ai/news/grok-4-7",
            "measuredAt": "2026-09-21"
          },
          {
            "axisId": "terminal-4",
            "benchmark": "Terminal-Bench 4.0",
            "category": "Coding",
            "rawScore": 38,
            "percentile": 50,
            "peerCount": 3,
            "sourceUrl": "https://x.ai/news/grok-4-7",
            "measuredAt": "2026-09-21"
          }
        ]
      },
      {
        "modelId": "glm-5-3-flash",
        "name": "GLM-5.3-Flash",
        "providerId": "zai",
        "releaseDate": "2026-09-16",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 21.9,
          "Reasoning": 59.9,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 63.4,
            "percentile": 21.9,
            "peerCount": 16,
            "sourceUrl": "https://autoclaw.z.ai/blog/model/glm-5.3-flash/",
            "measuredAt": "2026-09-16"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.2,
            "percentile": 68.9,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 55.8,
            "percentile": 50.8,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "deepseek-v4-1-flash",
        "name": "DeepSeek V4.1 Flash",
        "providerId": "deepseek",
        "releaseDate": "2026-09-10",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 96.9,
          "Reasoning": 75.7,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 74.2,
            "percentile": 96.9,
            "peerCount": 16,
            "sourceUrl": "https://api-docs.deepseek.com/updates/",
            "measuredAt": "2026-09-10"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.9,
            "percentile": 75.7,
            "peerCount": 111,
            "sourceUrl": "https://api-docs.deepseek.com/updates/",
            "measuredAt": "2026-09-10"
          }
        ]
      },
      {
        "modelId": "gemini-3-8-flash",
        "name": "Gemini 3.8 Flash",
        "providerId": "google",
        "releaseDate": "2026-09-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 53.7,
          "Reasoning": 84.4,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 73.7,
            "percentile": 90.6,
            "peerCount": 16,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
            "measuredAt": "2026-09-02"
          },
          {
            "axisId": "terminal-4",
            "benchmark": "Terminal-Bench 4.0",
            "category": "Coding",
            "rawScore": 19.1,
            "percentile": 16.7,
            "peerCount": 3,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-8-flash/",
            "measuredAt": "2026-09-02"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 95.3,
            "percentile": 98.6,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-6-astra/",
            "measuredAt": "2026-09-03"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 68.4,
            "percentile": 70.3,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "claude-fable-5-1",
        "name": "Claude Fable 5.1",
        "providerId": "anthropic",
        "releaseDate": "2026-09-01",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 37.5,
          "Reasoning": 97.9,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "terminal-science-0-1",
            "benchmark": "Terminal-Bench Science 0.1",
            "category": "Coding",
            "rawScore": 52.6,
            "percentile": 37.5,
            "peerCount": 4,
            "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
            "measuredAt": "2026-09-01"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 60.9,
            "percentile": 98.3,
            "peerCount": 29,
            "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
            "measuredAt": "2026-09-01"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 90.2,
            "percentile": 97.5,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-8-flash",
        "name": "Qwen3.8-Flash",
        "providerId": "qwen",
        "releaseDate": "2026-08-26",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "solar-pro-4",
        "name": "Solar Pro 4",
        "providerId": "upstage",
        "releaseDate": "2026-08-20",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "glm-5-3",
        "name": "GLM-5.3",
        "providerId": "zai",
        "releaseDate": "2026-08-14",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 46.9,
          "Reasoning": 73.9,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 66.9,
            "percentile": 46.9,
            "peerCount": 16,
            "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
            "measuredAt": "2026-08-14"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.9,
            "percentile": 75.7,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 68.8,
            "percentile": 72,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-8-27b",
        "name": "Qwen3.8 27B",
        "providerId": "qwen",
        "releaseDate": "2026-08-14",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 61.4,
          "Reasoning": 65.3,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 61.7,
            "percentile": 61.4,
            "peerCount": 22,
            "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-27B",
            "measuredAt": "2026-08-14"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 89.2,
            "percentile": 65.3,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-27B",
            "measuredAt": "2026-08-14"
          }
        ]
      },
      {
        "modelId": "deepseek-v4-pro",
        "name": "DeepSeek V4 Pro",
        "providerId": "deepseek",
        "releaseDate": "2026-08-13",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 15.6,
          "Reasoning": 68.3,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 62.7,
            "percentile": 15.6,
            "peerCount": 16,
            "sourceUrl": "https://api-docs.deepseek.com/updates/",
            "measuredAt": "2026-08-13"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.9,
            "percentile": 75.7,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 60,
            "percentile": 89.3,
            "peerCount": 14,
            "sourceUrl": "https://api-docs.deepseek.com/updates/",
            "measuredAt": "2026-08-13"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 45.3,
            "percentile": 39.8,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gemini-3-7-flash",
        "name": "Gemini 3.7 Flash",
        "providerId": "google",
        "releaseDate": "2026-08-13",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 28.1,
          "Reasoning": 86.6,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 65.3,
            "percentile": 28.1,
            "peerCount": 16,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
            "measuredAt": "2026-08-13"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 94.8,
            "percentile": 97.7,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 71.6,
            "percentile": 75.4,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "mai-thinking-1",
        "name": "MAI-Thinking-1",
        "providerId": "microsoft",
        "releaseDate": "2026-08-12",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 52.4,
          "Reasoning": 68.6,
          "Agents": null
        },
        "coverage": 6,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 73.5,
            "percentile": 59.4,
            "peerCount": 48,
            "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
            "measuredAt": "2026-08-12"
          },
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 52.8,
            "percentile": 20.5,
            "peerCount": 22,
            "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
            "measuredAt": "2026-08-12"
          },
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 87.7,
            "percentile": 77.3,
            "peerCount": 11,
            "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
            "measuredAt": "2026-08-12"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 84.2,
            "percentile": 44.6,
            "peerCount": 111,
            "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
            "measuredAt": "2026-08-12"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 85,
            "percentile": 72.4,
            "peerCount": 29,
            "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
            "measuredAt": "2026-08-12"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 97,
            "percentile": 88.8,
            "peerCount": 40,
            "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
            "measuredAt": "2026-08-12"
          }
        ]
      },
      {
        "modelId": "grok-4-6",
        "name": "Grok 4.6",
        "providerId": "xai",
        "releaseDate": "2026-08-12",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 34.4,
          "Reasoning": 79.2,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 65.9,
            "percentile": 34.4,
            "peerCount": 16,
            "sourceUrl": "https://x.ai/news/grok-4-6",
            "measuredAt": "2026-08-12"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 94,
            "percentile": 92.3,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 66,
            "percentile": 66.1,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-8-2-4t-a95b",
        "name": "Qwen3.8 2.4T-A95B",
        "providerId": "qwen",
        "releaseDate": "2026-08-12",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 83.8,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 92.6,
            "percentile": 83.8,
            "peerCount": 111,
            "sourceUrl": "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/qwen3-8-2-4t-a95b",
            "measuredAt": "2026-08-12"
          }
        ]
      },
      {
        "modelId": "qwen3-8-max",
        "name": "Qwen3.8-Max",
        "providerId": "qwen",
        "releaseDate": "2026-08-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 93.2,
          "Reasoning": 82.2,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 67.7,
            "percentile": 93.2,
            "peerCount": 22,
            "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
            "measuredAt": "2026-08-12"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 92.6,
            "percentile": 83.8,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B",
            "measuredAt": "2026-08-12"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 74.7,
            "percentile": 80.5,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "claude-opus-5",
        "name": "Claude Opus 5",
        "providerId": "anthropic",
        "releaseDate": "2026-07-24",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 12.5,
          "Reasoning": 92.3,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "terminal-science-0-1",
            "benchmark": "Terminal-Bench Science 0.1",
            "category": "Coding",
            "rawScore": 29,
            "percentile": 12.5,
            "peerCount": 4,
            "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
            "measuredAt": "2026-09-01"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 93.9,
            "percentile": 91.4,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 56.6,
            "percentile": 94.8,
            "peerCount": 29,
            "sourceUrl": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
            "measuredAt": "2026-09-01"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 85.6,
            "percentile": 90.7,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gemini-3-5-flash-lite",
        "name": "Gemini 3.5 Flash-Lite",
        "providerId": "google",
        "releaseDate": "2026-07-21",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 32,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 83.3,
            "percentile": 41,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 26,
            "percentile": 22.9,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-7-flash",
        "name": "Qwen3.7-Flash",
        "providerId": "qwen",
        "releaseDate": "2026-07-15",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 25.9,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 82.3,
            "percentile": 35.6,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 19.3,
            "percentile": 16.1,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "dola-seed-2-1-turbo",
        "name": "Dola Seed 2.1 Turbo",
        "providerId": "bytedance",
        "releaseDate": "2026-07-13",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "seed-2-1-pro",
        "name": "Seed2.1 Pro",
        "providerId": "bytedance",
        "releaseDate": "2026-07-13",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "gpt-5-6-luna",
        "name": "GPT-5.6 Luna",
        "providerId": "openai",
        "releaseDate": "2026-07-09",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 64.1,
          "Reasoning": 83.1,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 62.7,
            "percentile": 75,
            "peerCount": 22,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 67.2,
            "percentile": 53.1,
            "peerCount": 16,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 92.3,
            "percentile": 80.6,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 82.1,
            "percentile": 85.6,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-5-6-sol",
        "name": "GPT-5.6 Sol",
        "providerId": "openai",
        "releaseDate": "2026-07-09",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 86.5,
          "Reasoning": 96.3,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 64.6,
            "percentile": 88.6,
            "peerCount": 22,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 72.7,
            "percentile": 84.4,
            "peerCount": 16,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 94.6,
            "percentile": 96.8,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 89.1,
            "percentile": 95.8,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-5-6-terra",
        "name": "GPT-5.6 Terra",
        "providerId": "openai",
        "releaseDate": "2026-07-09",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 75.7,
          "Reasoning": 89.7,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 63.4,
            "percentile": 79.5,
            "peerCount": 22,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 69.6,
            "percentile": 71.9,
            "peerCount": 16,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 92.9,
            "percentile": 86.9,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 86,
            "percentile": 92.4,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "grok-4-5",
        "name": "Grok 4.5",
        "providerId": "xai",
        "releaseDate": "2026-07-08",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 71.5,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 93.4,
            "percentile": 88.7,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 57.2,
            "percentile": 54.2,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "claude-sonnet-5",
        "name": "Claude Sonnet 5",
        "providerId": "anthropic",
        "releaseDate": "2026-06-30",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 94.8,
          "Reasoning": 71,
          "Agents": null
        },
        "coverage": 5,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 85.2,
            "percentile": 94.8,
            "peerCount": 48,
            "sourceUrl": "https://www-cdn.anthropic.com/283ef97c476cf442c91d9a37d5b214242a55bb92/Claude%20Sonnet%205%20System%20Card.pdf",
            "measuredAt": "2026-06-30"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.5,
            "percentile": 71.6,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 43.2,
            "percentile": 81,
            "peerCount": 29,
            "sourceUrl": "https://www-cdn.anthropic.com/283ef97c476cf442c91d9a37d5b214242a55bb92/Claude%20Sonnet%205%20System%20Card.pdf",
            "measuredAt": "2026-06-30"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 57.4,
            "percentile": 67.9,
            "peerCount": 14,
            "sourceUrl": "https://www-cdn.anthropic.com/283ef97c476cf442c91d9a37d5b214242a55bb92/Claude%20Sonnet%205%20System%20Card.pdf",
            "measuredAt": "2026-06-30"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 65.6,
            "percentile": 63.6,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "glm-5-2",
        "name": "GLM-5.2",
        "providerId": "zai",
        "releaseDate": "2026-06-24",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 34.5,
          "Reasoning": 69.1,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 62.1,
            "percentile": 65.9,
            "peerCount": 22,
            "sourceUrl": "https://docs.z.ai/guides/llm/glm-5.2",
            "measuredAt": "2026-06-24"
          },
          {
            "axisId": "deepswe-1-1",
            "benchmark": "DeepSWE v1.1",
            "category": "Coding",
            "rawScore": 46.2,
            "percentile": 3.1,
            "peerCount": 16,
            "sourceUrl": "https://huggingface.co/zai-org/GLM-5.3",
            "measuredAt": "2026-08-14"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 91.9,
            "percentile": 79.7,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 59.2,
            "percentile": 58.5,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "kimi-k2-7-code",
        "name": "Kimi K2.7 Code",
        "providerId": "moonshot",
        "releaseDate": "2026-06-12",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 52.4,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 87.9,
            "percentile": 58.1,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 54,
            "percentile": 46.6,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "north-mini-code",
        "name": "North Mini Code",
        "providerId": "cohere",
        "releaseDate": "2026-06-09",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 18.4,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 1,
        "missingCategories": [
          "Reasoning",
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 67.6,
            "percentile": 30.2,
            "peerCount": 48,
            "sourceUrl": "https://cohere.com/blog/north-mini-code",
            "measuredAt": "2026-06-09"
          },
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 40.2,
            "percentile": 2.3,
            "peerCount": 22,
            "sourceUrl": "https://cohere.com/blog/north-mini-code",
            "measuredAt": "2026-06-09"
          },
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 70.3,
            "percentile": 22.7,
            "peerCount": 11,
            "sourceUrl": "https://cohere.com/blog/north-mini-code",
            "measuredAt": "2026-06-09"
          }
        ]
      },
      {
        "modelId": "gemma-4-12b",
        "name": "Gemma 4 12B Unified",
        "providerId": "google",
        "releaseDate": "2026-06-03",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 31.8,
          "Reasoning": 21,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 72,
            "percentile": 31.8,
            "peerCount": 11,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-06-03"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 78.8,
            "percentile": 28.4,
            "peerCount": 111,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-06-03"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 77.2,
            "percentile": 32.8,
            "peerCount": 29,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-06-03"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 5.2,
            "percentile": 1.7,
            "peerCount": 29,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-06-03"
          }
        ]
      },
      {
        "modelId": "minimax-m3",
        "name": "MiniMax M3",
        "providerId": "minimax",
        "releaseDate": "2026-06-01",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 75.7,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.9,
            "percentile": 75.7,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "step-3-7-flash",
        "name": "Step 3.7 Flash",
        "providerId": "stepfun",
        "releaseDate": "2026-05-29",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 55.8,
          "Reasoning": null,
          "Agents": 44.2
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Reasoning"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 76.5,
            "percentile": 72.9,
            "peerCount": 48,
            "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
            "measuredAt": "2026-05-29"
          },
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 56.3,
            "percentile": 38.6,
            "peerCount": 22,
            "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
            "measuredAt": "2026-05-29"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 75.8,
            "percentile": 30,
            "peerCount": 15,
            "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
            "measuredAt": "2026-05-29"
          },
          {
            "axisId": "toolathlon",
            "benchmark": "Toolathlon",
            "category": "Agents",
            "rawScore": 49.5,
            "percentile": 58.3,
            "peerCount": 6,
            "sourceUrl": "https://static.stepfun.com/blog/step-3.7-flash/",
            "measuredAt": "2026-05-29"
          }
        ]
      },
      {
        "modelId": "qwen3-7-plus",
        "name": "Qwen3.7-Plus",
        "providerId": "qwen",
        "releaseDate": "2026-05-26",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 58.1,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 87.9,
            "percentile": 58.1,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "command-a-plus",
        "name": "Command A+",
        "providerId": "cohere",
        "releaseDate": "2026-05-20",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 60,
          "Agents": 31.3
        },
        "coverage": 2,
        "categoryCoverage": 2,
        "missingCategories": [
          "Coding"
        ],
        "components": [
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 90,
            "percentile": 60,
            "peerCount": 40,
            "sourceUrl": "https://cohere.com/blog/command-a-plus",
            "measuredAt": "2026-05-20"
          },
          {
            "axisId": "tau2-telecom",
            "benchmark": "Tau2-bench Telecom",
            "category": "Agents",
            "rawScore": 85,
            "percentile": 31.3,
            "peerCount": 8,
            "sourceUrl": "https://cohere.com/blog/command-a-plus",
            "measuredAt": "2026-05-20"
          }
        ]
      },
      {
        "modelId": "qwen3-7-max",
        "name": "Qwen3.7-Max",
        "providerId": "qwen",
        "releaseDate": "2026-05-20",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "gemini-3-5-flash",
        "name": "Gemini 3.5 Flash",
        "providerId": "google",
        "releaseDate": "2026-05-19",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 73.8,
          "Agents": 74.3
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Coding"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 92.8,
            "percentile": 85.6,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 62.8,
            "percentile": 61.9,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "mcp-atlas",
            "benchmark": "MCP Atlas",
            "category": "Agents",
            "rawScore": 83.6,
            "percentile": 94.4,
            "peerCount": 9,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
            "measuredAt": "2026-05-19"
          },
          {
            "axisId": "osworld-verified",
            "benchmark": "OSWorld-Verified",
            "category": "Agents",
            "rawScore": 78.4,
            "percentile": 54.2,
            "peerCount": 12,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-5-flash/",
            "measuredAt": "2026-05-19"
          }
        ]
      },
      {
        "modelId": "grok-build-0-1",
        "name": "Grok Build 0.1",
        "providerId": "xai",
        "releaseDate": "2026-05-19",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "ernie-5-1",
        "name": "ERNIE 5.1",
        "providerId": "baidu",
        "releaseDate": "2026-05-09",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "mistral-medium-3-5",
        "name": "Mistral Medium 3.5",
        "providerId": "mistral",
        "releaseDate": "2026-04-28",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 80.2,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Reasoning",
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 77.6,
            "percentile": 80.2,
            "peerCount": 48,
            "sourceUrl": "https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5/",
            "measuredAt": "2026-05-22"
          }
        ]
      },
      {
        "modelId": "gpt-5-5-pro",
        "name": "GPT-5.5 Pro",
        "providerId": "openai",
        "releaseDate": "2026-04-23",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 77.5,
          "Agents": 96.7
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Coding"
        ],
        "components": [
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 43.1,
            "percentile": 77.6,
            "peerCount": 29,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
            "measuredAt": "2026-04-23"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 57.2,
            "percentile": 60.7,
            "peerCount": 14,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
            "measuredAt": "2026-04-23"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 87.7,
            "percentile": 94.1,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 90.1,
            "percentile": 96.7,
            "peerCount": 15,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-5/",
            "measuredAt": "2026-04-23"
          }
        ]
      },
      {
        "modelId": "qwen3-6-27b",
        "name": "Qwen3.6 27B",
        "providerId": "qwen",
        "releaseDate": "2026-04-21",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 77.1,
          "Reasoning": 58.4,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 77.2,
            "percentile": 77.1,
            "peerCount": 48,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 87.8,
            "percentile": 55.9,
            "peerCount": 111,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 86.2,
            "percentile": 89.7,
            "peerCount": 29,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 35.1,
            "percentile": 29.7,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "kimi-k2-6",
        "name": "Kimi K2.6",
        "providerId": "moonshot",
        "releaseDate": "2026-04-20",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 63.8,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.8,
            "percentile": 73.4,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 57.2,
            "percentile": 54.2,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-6-flash",
        "name": "Qwen3.6-Flash",
        "providerId": "qwen",
        "releaseDate": "2026-04-16",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 26,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 83.3,
            "percentile": 41,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 17.2,
            "percentile": 11,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "glm-5-1",
        "name": "GLM-5.1",
        "providerId": "zai",
        "releaseDate": "2026-04-07",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 49.5,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 89.9,
            "percentile": 67.6,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 36.8,
            "percentile": 31.4,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gemma-4-26b-a4b",
        "name": "Gemma 4 26B A4B",
        "providerId": "google",
        "releaseDate": "2026-04-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 50,
          "Reasoning": 32.5,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 77.1,
            "percentile": 50,
            "peerCount": 11,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 82.3,
            "percentile": 35.6,
            "peerCount": 111,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 82.6,
            "percentile": 53.4,
            "peerCount": 29,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 8.7,
            "percentile": 8.6,
            "peerCount": 29,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          }
        ]
      },
      {
        "modelId": "gemma-4-31b",
        "name": "Gemma 4 31B",
        "providerId": "google",
        "releaseDate": "2026-04-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 59.1,
          "Reasoning": 52.5,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 80,
            "percentile": 59.1,
            "peerCount": 11,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 84.3,
            "percentile": 45.5,
            "peerCount": 111,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 85.2,
            "percentile": 79.3,
            "peerCount": 29,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 19.5,
            "percentile": 32.8,
            "peerCount": 29,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          }
        ]
      },
      {
        "modelId": "gemma-4-e2b",
        "name": "Gemma 4 E2B",
        "providerId": "google",
        "releaseDate": "2026-04-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 4.5,
          "Reasoning": 3.3,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 44,
            "percentile": 4.5,
            "peerCount": 11,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 43.4,
            "percentile": 1.4,
            "peerCount": 111,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 60,
            "percentile": 5.2,
            "peerCount": 29,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          }
        ]
      },
      {
        "modelId": "gemma-4-e4b",
        "name": "Gemma 4 E4B",
        "providerId": "google",
        "releaseDate": "2026-04-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 13.6,
          "Reasoning": 10.8,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 52,
            "percentile": 13.6,
            "peerCount": 11,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 58.6,
            "percentile": 9.5,
            "peerCount": 111,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 69.4,
            "percentile": 12.1,
            "peerCount": 29,
            "sourceUrl": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "measuredAt": "2026-04-02"
          }
        ]
      },
      {
        "modelId": "qwen3-6-plus",
        "name": "Qwen3.6-Plus",
        "providerId": "qwen",
        "releaseDate": "2026-04-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 47.4,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 88.4,
            "percentile": 61.7,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 38.2,
            "percentile": 33.1,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "minimax-m2-7",
        "name": "MiniMax M2.7",
        "providerId": "minimax",
        "releaseDate": "2026-03-18",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "mistral-small-4",
        "name": "Mistral Small 4",
        "providerId": "mistral",
        "releaseDate": "2026-03-16",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 28.3,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 71.2,
            "percentile": 20.3,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/mistralai/Mistral-Small-4-119B-2603",
            "measuredAt": "2026-03-16"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 78,
            "percentile": 36.2,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/mistralai/Mistral-Small-4-119B-2603",
            "measuredAt": "2026-03-16"
          }
        ]
      },
      {
        "modelId": "nvidia-nemotron-3-super-120b-a12b",
        "name": "NVIDIA Nemotron 3 Super 120B-A12B",
        "providerId": "nvidia",
        "releaseDate": "2026-03-11",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 24,
          "Reasoning": 41,
          "Agents": null
        },
        "coverage": 5,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 60.47,
            "percentile": 24,
            "peerCount": 48,
            "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
            "measuredAt": "2026-03-11"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 83.73,
            "percentile": 60.3,
            "peerCount": 29,
            "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
            "measuredAt": "2026-03-11"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 90.21,
            "percentile": 63.7,
            "peerCount": 40,
            "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
            "measuredAt": "2026-03-11"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 18.26,
            "percentile": 29.3,
            "peerCount": 29,
            "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
            "measuredAt": "2026-03-11"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 22.82,
            "percentile": 10.7,
            "peerCount": 14,
            "sourceUrl": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard",
            "measuredAt": "2026-03-11"
          }
        ]
      },
      {
        "modelId": "grok-4-20-reasoning",
        "name": "Grok 4.20 Reasoning",
        "providerId": "xai",
        "releaseDate": "2026-03-09",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 51.8,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 89.3,
            "percentile": 66.2,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 44.9,
            "percentile": 37.3,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "grok-4-20-multi-agent",
        "name": "Grok 4.20 Multi-Agent",
        "providerId": "xai",
        "releaseDate": "2026-03-09",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "grok-4-20-non-reasoning",
        "name": "Grok 4.20 Non-Reasoning",
        "providerId": "xai",
        "releaseDate": "2026-03-09",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "gpt-5-4-pro",
        "name": "GPT-5.4 Pro",
        "providerId": "openai",
        "releaseDate": "2026-03-05",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 84.9,
          "Agents": 90
        },
        "coverage": 5,
        "categoryCoverage": 2,
        "missingCategories": [
          "Coding"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 94.4,
            "percentile": 95.9,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 42.7,
            "percentile": 74.1,
            "peerCount": 29,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 58.7,
            "percentile": 82.1,
            "peerCount": 14,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 82.5,
            "percentile": 87.3,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 89.3,
            "percentile": 90,
            "peerCount": 15,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4/",
            "measuredAt": "2026-03-05"
          }
        ]
      },
      {
        "modelId": "gemini-3-1-flash-lite",
        "name": "Gemini 3.1 Flash-Lite",
        "providerId": "google",
        "releaseDate": "2026-03-03",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 33.5,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 86.9,
            "percentile": 53.6,
            "peerCount": 111,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
            "measuredAt": "2026-03-03"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 16,
            "percentile": 22.4,
            "peerCount": 29,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-1-flash-lite/",
            "measuredAt": "2026-03-03"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 27.7,
            "percentile": 24.6,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-5-flash",
        "name": "Qwen3.5-Flash",
        "providerId": "qwen",
        "releaseDate": "2026-02-23",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 21.6,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 82.3,
            "percentile": 35.6,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 9.5,
            "percentile": 7.6,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gemini-3-1-pro-preview",
        "name": "Gemini 3.1 Pro Preview",
        "providerId": "google",
        "releaseDate": "2026-02-19",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 25,
          "Reasoning": 77.6,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 54.2,
            "percentile": 25,
            "peerCount": 22,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 94.3,
            "percentile": 95,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-5-6/",
            "measuredAt": "2026-07-09"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 59.6,
            "percentile": 60.2,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-5-397b-a17b",
        "name": "Qwen3.5 397B-A17B",
        "providerId": "qwen",
        "releaseDate": "2026-02-15",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 69.8,
          "Reasoning": 60.9,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 76.4,
            "percentile": 69.8,
            "peerCount": 48,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.5",
            "measuredAt": "2026-02-15"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 88.4,
            "percentile": 61.7,
            "peerCount": 111,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 87.8,
            "percentile": 94.8,
            "peerCount": 29,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.5",
            "measuredAt": "2026-02-15"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 29.5,
            "percentile": 26.3,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-5-plus",
        "name": "Qwen3.5-Plus",
        "providerId": "qwen",
        "releaseDate": "2026-02-15",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 46.4,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 84.8,
            "percentile": 46.4,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "minimax-m2-5",
        "name": "MiniMax M2.5",
        "providerId": "minimax",
        "releaseDate": "2026-02-12",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "ernie-5-0",
        "name": "ERNIE 5.0",
        "providerId": "baidu",
        "releaseDate": "2026-02-06",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 40.9,
          "Reasoning": 54.8,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 76.21,
            "percentile": 40.9,
            "peerCount": 11,
            "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
            "measuredAt": "2026-02-06"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 86.36,
            "percentile": 51.8,
            "peerCount": 111,
            "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
            "measuredAt": "2026-02-06"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 83.8,
            "percentile": 63.8,
            "peerCount": 29,
            "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
            "measuredAt": "2026-02-06"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 89.06,
            "percentile": 48.8,
            "peerCount": 40,
            "sourceUrl": "https://ernie.baidu.com/blog/posts/ernie5.0/",
            "measuredAt": "2026-02-06"
          }
        ]
      },
      {
        "modelId": "qwen3-coder-next",
        "name": "Qwen3-Coder-Next",
        "providerId": "qwen",
        "releaseDate": "2026-02-03",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 25.8,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Reasoning",
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 70.6,
            "percentile": 44.8,
            "peerCount": 48,
            "sourceUrl": "https://huggingface.co/Qwen/Qwen3-Coder-Next",
            "measuredAt": "2026-02-03"
          },
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 44.3,
            "percentile": 6.8,
            "peerCount": 22,
            "sourceUrl": "https://huggingface.co/Qwen/Qwen3-Coder-Next",
            "measuredAt": "2026-02-03"
          }
        ]
      },
      {
        "modelId": "glm-4-7",
        "name": "GLM-4.7",
        "providerId": "zai",
        "releaseDate": "2025-12-22",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 41,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 83.3,
            "percentile": 41,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gemini-3-flash-preview",
        "name": "Gemini 3 Flash Preview",
        "providerId": "google",
        "releaseDate": "2025-12-17",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 85.4,
          "Reasoning": 64.6,
          "Agents": null
        },
        "coverage": 5,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 78,
            "percentile": 85.4,
            "peerCount": 48,
            "sourceUrl": "https://deepmind.google/technologies/gemini/flash/",
            "measuredAt": "2025-12-17"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 90.4,
            "percentile": 70.3,
            "peerCount": 111,
            "sourceUrl": "https://blog.google/products-and-platforms/products/gemini/gemini-3-flash/",
            "measuredAt": "2025-12-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 95.2,
            "percentile": 83.8,
            "peerCount": 40,
            "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-flash/",
            "measuredAt": "2025-12-17"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 33.7,
            "percentile": 60.3,
            "peerCount": 29,
            "sourceUrl": "https://blog.google/products-and-platforms/products/gemini/gemini-3-flash/",
            "measuredAt": "2025-12-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 51.2,
            "percentile": 44.1,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "nvidia-nemotron-3-nano-30b-a3b-bf16",
        "name": "NVIDIA Nemotron 3 Nano 30B-A3B BF16",
        "providerId": "nvidia",
        "releaseDate": "2025-12-15",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 27.5,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 78.3,
            "percentile": 39.7,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
            "measuredAt": "2025-12-15"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 89.1,
            "percentile": 51.2,
            "peerCount": 40,
            "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
            "measuredAt": "2025-12-15"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 10.6,
            "percentile": 15.5,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
            "measuredAt": "2025-12-15"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 15.5,
            "percentile": 3.6,
            "peerCount": 14,
            "sourceUrl": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
            "measuredAt": "2025-12-15"
          }
        ]
      },
      {
        "modelId": "gpt-5-2-pro",
        "name": "GPT-5.2 Pro",
        "providerId": "openai",
        "releaseDate": "2025-12-11",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 72,
          "Agents": 43.3
        },
        "coverage": 6,
        "categoryCoverage": 2,
        "missingCategories": [
          "Coding"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 93.2,
            "percentile": 87.8,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
            "measuredAt": "2025-12-11"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 100,
            "percentile": 97.5,
            "peerCount": 40,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
            "measuredAt": "2025-12-11"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 36.6,
            "percentile": 63.8,
            "peerCount": 29,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
            "measuredAt": "2025-12-11"
          },
          {
            "axisId": "hle-with-tools",
            "benchmark": "Humanity's Last Exam (with tools)",
            "category": "Reasoning",
            "rawScore": 50,
            "percentile": 32.1,
            "peerCount": 14,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
            "measuredAt": "2025-12-11"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 74,
            "percentile": 78.8,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 77.9,
            "percentile": 43.3,
            "peerCount": 15,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
            "measuredAt": "2025-12-11"
          }
        ]
      },
      {
        "modelId": "gpt-5-2",
        "name": "GPT-5.2",
        "providerId": "openai",
        "releaseDate": "2025-12-11",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 62.4,
          "Reasoning": 82.7,
          "Agents": null
        },
        "coverage": 5,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 80,
            "percentile": 90.6,
            "peerCount": 48,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
            "measuredAt": "2025-12-11"
          },
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 55.6,
            "percentile": 34.1,
            "peerCount": 22,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
            "measuredAt": "2025-12-11"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 92.4,
            "percentile": 82,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
            "measuredAt": "2025-12-11"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 100,
            "percentile": 97.5,
            "peerCount": 40,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-2/",
            "measuredAt": "2025-12-11"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 67.4,
            "percentile": 68.6,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "amazon-nova-2-lite",
        "name": "Amazon Nova 2 Lite",
        "providerId": "amazon",
        "releaseDate": "2025-12-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 47.7,
          "Agents": 12.2
        },
        "coverage": 5,
        "categoryCoverage": 2,
        "missingCategories": [
          "Coding"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 79.6,
            "percentile": 30.2,
            "peerCount": 111,
            "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
            "measuredAt": "2025-12-02"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 80.9,
            "percentile": 46.6,
            "peerCount": 29,
            "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
            "measuredAt": "2025-12-02"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 91,
            "percentile": 66.3,
            "peerCount": 40,
            "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
            "measuredAt": "2025-12-02"
          },
          {
            "axisId": "mcp-atlas",
            "benchmark": "MCP Atlas",
            "category": "Agents",
            "rawScore": 24.6,
            "percentile": 5.6,
            "peerCount": 9,
            "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
            "measuredAt": "2025-12-02"
          },
          {
            "axisId": "tau2-telecom",
            "benchmark": "Tau2-bench Telecom",
            "category": "Agents",
            "rawScore": 76,
            "percentile": 18.8,
            "peerCount": 8,
            "sourceUrl": "https://cdn.amazon.science/c5/3d/84514a224666b5be6de4b43ef4aa/nova-2-0-technical-report2.pdf",
            "measuredAt": "2025-12-02"
          }
        ]
      },
      {
        "modelId": "ministral-3-14b",
        "name": "Ministral 3 14B",
        "providerId": "mistral",
        "releaseDate": "2025-12-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 30.2,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 71.2,
            "percentile": 20.3,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-14B-Instruct-2512-BF16",
            "measuredAt": "2025-12-02"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 85,
            "percentile": 40,
            "peerCount": 40,
            "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-14B-Instruct-2512-BF16",
            "measuredAt": "2025-12-02"
          }
        ]
      },
      {
        "modelId": "ministral-3-3b",
        "name": "Ministral 3 3B",
        "providerId": "mistral",
        "releaseDate": "2025-12-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 16.1,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 53.4,
            "percentile": 5.9,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-3B-Instruct-2512",
            "measuredAt": "2025-12-02"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 72.1,
            "percentile": 26.3,
            "peerCount": 40,
            "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-3B-Instruct-2512",
            "measuredAt": "2025-12-02"
          }
        ]
      },
      {
        "modelId": "ministral-3-8b",
        "name": "Ministral 3 8B",
        "providerId": "mistral",
        "releaseDate": "2025-12-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 22.3,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 66.8,
            "percentile": 15.8,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-8B-Instruct-2512-BF16",
            "measuredAt": "2025-12-02"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 78.7,
            "percentile": 28.7,
            "peerCount": 40,
            "sourceUrl": "https://huggingface.co/mistralai/Ministral-3-8B-Instruct-2512-BF16",
            "measuredAt": "2025-12-02"
          }
        ]
      },
      {
        "modelId": "mistral-large-3",
        "name": "Mistral Large 3",
        "providerId": "mistral",
        "releaseDate": "2025-12-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "deepseek-v3-2",
        "name": "DeepSeek V3.2",
        "providerId": "deepseek",
        "releaseDate": "2025-12-01",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 53.1,
          "Reasoning": 56.7,
          "Agents": null
        },
        "coverage": 5,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 73.1,
            "percentile": 53.1,
            "peerCount": 48,
            "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
            "measuredAt": "2025-12-01"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 82.4,
            "percentile": 37.4,
            "peerCount": 111,
            "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
            "measuredAt": "2025-12-01"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 85,
            "percentile": 72.4,
            "peerCount": 29,
            "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
            "measuredAt": "2025-12-01"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 93.1,
            "percentile": 73.8,
            "peerCount": 40,
            "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
            "measuredAt": "2025-12-01"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 25.1,
            "percentile": 43.1,
            "peerCount": 29,
            "sourceUrl": "https://modelscope.cn/models/deepseek-ai/DeepSeek-V3.2/resolve/master/assets/paper.pdf",
            "measuredAt": "2025-12-01"
          }
        ]
      },
      {
        "modelId": "gpt-5-1",
        "name": "GPT-5.1",
        "providerId": "openai",
        "releaseDate": "2025-11-13",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 66.7,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 87.6,
            "percentile": 54.5,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 94.2,
            "percentile": 78.8,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "claude-haiku-4-5",
        "name": "Claude Haiku 4.5",
        "providerId": "anthropic",
        "releaseDate": "2025-10-15",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 55.2,
          "Reasoning": 20.3,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 73.3,
            "percentile": 55.2,
            "peerCount": 48,
            "sourceUrl": "https://www.anthropic.com/news/claude-haiku-4-5",
            "measuredAt": "2025-10-15"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 71.2,
            "percentile": 20.3,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-5-pro",
        "name": "GPT-5 Pro",
        "providerId": "openai",
        "releaseDate": "2025-10-06",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 56.3,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 88.4,
            "percentile": 61.7,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5/",
            "measuredAt": "2025-08-07"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 55.8,
            "percentile": 50.8,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-max",
        "name": "Qwen3-Max",
        "providerId": "qwen",
        "releaseDate": "2025-09-23",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 35.4,
          "Reasoning": 19.2,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 69.6,
            "percentile": 35.4,
            "peerCount": 48,
            "sourceUrl": "https://qwen.ai/blog?from=research.latest-advancements-list&id=241398b9cd6353de490b0f82806c7848c5d2777d",
            "measuredAt": "2025-09-24"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 72.6,
            "percentile": 23.9,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 18.9,
            "percentile": 14.4,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "grok-4-fast-reasoning",
        "name": "Grok 4 Fast Reasoning",
        "providerId": "xai",
        "releaseDate": "2025-09-19",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 51.9,
          "Agents": 16.7
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Coding"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 85.7,
            "percentile": 48.2,
            "peerCount": 111,
            "sourceUrl": "https://x.ai/news/grok-4-fast",
            "measuredAt": "2025-09-19"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 92,
            "percentile": 71.3,
            "peerCount": 40,
            "sourceUrl": "https://x.ai/news/grok-4-fast",
            "measuredAt": "2025-09-19"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 20,
            "percentile": 36.2,
            "peerCount": 29,
            "sourceUrl": "https://x.ai/news/grok-4-fast",
            "measuredAt": "2025-09-19"
          },
          {
            "axisId": "browsecomp",
            "benchmark": "BrowseComp",
            "category": "Agents",
            "rawScore": 44.9,
            "percentile": 16.7,
            "peerCount": 15,
            "sourceUrl": "https://x.ai/news/grok-4-fast",
            "measuredAt": "2025-09-19"
          }
        ]
      },
      {
        "modelId": "longcat-flash-chat",
        "name": "LongCat Flash Chat",
        "providerId": "meituan",
        "releaseDate": "2025-09-01",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 21.9,
          "Reasoning": 31.8,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 60.4,
            "percentile": 21.9,
            "peerCount": 48,
            "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
            "measuredAt": "2025-09-01"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 73.23,
            "percentile": 24.8,
            "peerCount": 111,
            "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
            "measuredAt": "2025-09-01"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 82.68,
            "percentile": 56.9,
            "peerCount": 29,
            "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
            "measuredAt": "2025-09-01"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 61.25,
            "percentile": 13.8,
            "peerCount": 40,
            "sourceUrl": "https://github.com/meituan-longcat/LongCat-Flash-Chat",
            "measuredAt": "2025-09-01"
          }
        ]
      },
      {
        "modelId": "command-a-translate",
        "name": "Command A Translate",
        "providerId": "cohere",
        "releaseDate": "2025-08-28",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "command-a-reasoning",
        "name": "Command A Reasoning",
        "providerId": "cohere",
        "releaseDate": "2025-08-21",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 11.3,
          "Agents": 6.3
        },
        "coverage": 2,
        "categoryCoverage": 2,
        "missingCategories": [
          "Coding"
        ],
        "components": [
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 57,
            "percentile": 11.3,
            "peerCount": 40,
            "sourceUrl": "https://cohere.com/blog/command-a-plus",
            "measuredAt": "2026-05-20"
          },
          {
            "axisId": "tau2-telecom",
            "benchmark": "Tau2-bench Telecom",
            "category": "Agents",
            "rawScore": 37,
            "percentile": 6.3,
            "peerCount": 8,
            "sourceUrl": "https://cohere.com/blog/command-a-plus",
            "measuredAt": "2026-05-20"
          }
        ]
      },
      {
        "modelId": "gpt-5",
        "name": "GPT-5",
        "providerId": "openai",
        "releaseDate": "2025-08-07",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 65.6,
          "Reasoning": 59.9,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 74.9,
            "percentile": 65.6,
            "peerCount": 48,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5/",
            "measuredAt": "2025-08-07"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 86.2,
            "percentile": 50,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 94.6,
            "percentile": 81.3,
            "peerCount": 40,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5/",
            "measuredAt": "2025-08-07"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 55.4,
            "percentile": 48.3,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-5-mini",
        "name": "GPT-5 mini",
        "providerId": "openai",
        "releaseDate": "2025-08-07",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 46.9,
          "Reasoning": 39.4,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 71,
            "percentile": 46.9,
            "peerCount": 48,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-for-developers/",
            "measuredAt": "2025-08-07"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 81.6,
            "percentile": 32.9,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/",
            "measuredAt": "2026-03-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 87.5,
            "percentile": 43.8,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 46.7,
            "percentile": 41.5,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-5-nano",
        "name": "GPT-5 nano",
        "providerId": "openai",
        "releaseDate": "2025-08-07",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 17.7,
          "Reasoning": 24.8,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 54.7,
            "percentile": 17.7,
            "peerCount": 48,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-for-developers/",
            "measuredAt": "2025-08-07"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 69.4,
            "percentile": 16.7,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 85,
            "percentile": 40,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 20,
            "percentile": 17.8,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "claude-opus-4-1",
        "name": "Claude Opus 4.1",
        "providerId": "anthropic",
        "releaseDate": "2025-08-05",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 63.5,
          "Reasoning": 18.4,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 74.5,
            "percentile": 63.5,
            "peerCount": 48,
            "sourceUrl": "https://www.anthropic.com/news/claude-opus-4-1",
            "measuredAt": "2025-08-05"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 77.3,
            "percentile": 27.5,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 12.6,
            "percentile": 9.3,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-oss-120b",
        "name": "gpt-oss-120b",
        "providerId": "openai",
        "releaseDate": "2025-08-05",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 28.1,
          "Reasoning": 46,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 62.4,
            "percentile": 28.1,
            "peerCount": 48,
            "sourceUrl": "https://cdn.openai.com/pdf/419b6906-9da6-406c-a19d-1bb078ac7637/oai_gpt-oss_model_card.pdf",
            "measuredAt": "2025-08-05"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 80.1,
            "percentile": 32,
            "peerCount": 111,
            "sourceUrl": "https://cdn.openai.com/pdf/419b6906-9da6-406c-a19d-1bb078ac7637/oai_gpt-oss_model_card.pdf",
            "measuredAt": "2025-08-05"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 90,
            "percentile": 60,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-oss-20b",
        "name": "gpt-oss-20b",
        "providerId": "openai",
        "releaseDate": "2025-08-05",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 26,
          "Reasoning": 38.8,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 60.7,
            "percentile": 26,
            "peerCount": 48,
            "sourceUrl": "https://cdn.openai.com/pdf/419b6906-9da6-406c-a19d-1bb078ac7637/oai_gpt-oss_model_card.pdf",
            "measuredAt": "2025-08-05"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 71.5,
            "percentile": 22.5,
            "peerCount": 111,
            "sourceUrl": "https://cdn.openai.com/pdf/419b6906-9da6-406c-a19d-1bb078ac7637/oai_gpt-oss_model_card.pdf",
            "measuredAt": "2025-08-05"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 89.2,
            "percentile": 55,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "command-a-vision",
        "name": "Command A Vision",
        "providerId": "cohere",
        "releaseDate": "2025-07-31",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "gemini-2-5-flash-lite",
        "name": "Gemini 2.5 Flash-Lite",
        "providerId": "google",
        "releaseDate": "2025-07-22",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 1,
          "Reasoning": 13,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 27.6,
            "percentile": 1,
            "peerCount": 48,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Lite-Model-Card.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 66.7,
            "percentile": 14.9,
            "peerCount": 111,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Lite-Model-Card.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 63.1,
            "percentile": 18.8,
            "peerCount": 40,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Lite-Model-Card.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 6.9,
            "percentile": 5.2,
            "peerCount": 29,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/Model-Cards/Gemini-2-5-Flash-Lite-Model-Card.pdf",
            "measuredAt": "2025-06-17"
          }
        ]
      },
      {
        "modelId": "qwen3-coder-plus",
        "name": "Qwen3-Coder-Plus",
        "providerId": "qwen",
        "releaseDate": "2025-07-22",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 35.4,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Reasoning",
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 69.6,
            "percentile": 35.4,
            "peerCount": 48,
            "sourceUrl": "https://www.linkedin.com/posts/qwen_were-excited-to-announce-the-upgrade-of-activity-7376348152515260416-b4oF",
            "measuredAt": "2025-09-23"
          }
        ]
      },
      {
        "modelId": "gemini-2-5-pro",
        "name": "Gemini 2.5 Pro",
        "providerId": "google",
        "releaseDate": "2025-06-17",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 19.8,
          "Reasoning": 40,
          "Agents": null
        },
        "coverage": 5,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 59.6,
            "percentile": 19.8,
            "peerCount": 48,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 86.4,
            "percentile": 52.7,
            "peerCount": 111,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 88,
            "percentile": 46.3,
            "peerCount": 40,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 21.6,
            "percentile": 39.7,
            "peerCount": 29,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 24.6,
            "percentile": 21.2,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gemini-2-5-flash",
        "name": "Gemini 2.5 Flash",
        "providerId": "google",
        "releaseDate": "2025-06-17",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 7.3,
          "Reasoning": 27.2,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 48.9,
            "percentile": 7.3,
            "peerCount": 48,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 82.8,
            "percentile": 38.7,
            "peerCount": 111,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 72,
            "percentile": 23.8,
            "peerCount": 40,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
            "measuredAt": "2025-06-17"
          },
          {
            "axisId": "hle-no-tools",
            "benchmark": "Humanity's Last Exam (no tools)",
            "category": "Reasoning",
            "rawScore": 11,
            "percentile": 19,
            "peerCount": 29,
            "sourceUrl": "https://storage.googleapis.com/deepmind-media/gemini/gemini_v2_5_report.pdf",
            "measuredAt": "2025-06-17"
          }
        ]
      },
      {
        "modelId": "o3-pro",
        "name": "o3-pro",
        "providerId": "openai",
        "releaseDate": "2025-06-10",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "claude-opus-4",
        "name": "Claude Opus 4",
        "providerId": "anthropic",
        "releaseDate": "2025-05-22",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 49,
          "Reasoning": 26.6,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 72.5,
            "percentile": 49,
            "peerCount": 48,
            "sourceUrl": "https://www.anthropic.com/news/claude-4",
            "measuredAt": "2025-05-22"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 76.3,
            "percentile": 26.6,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "claude-sonnet-4",
        "name": "Claude Sonnet 4",
        "providerId": "anthropic",
        "releaseDate": "2025-05-22",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 51,
          "Reasoning": 29.3,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 72.7,
            "percentile": 51,
            "peerCount": 48,
            "sourceUrl": "https://www.anthropic.com/news/claude-4",
            "measuredAt": "2025-05-22"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 79.2,
            "percentile": 29.3,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "phi-4-reasoning",
        "name": "Phi-4 Reasoning",
        "providerId": "microsoft",
        "releaseDate": "2025-04-30",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 17.8,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 65.8,
            "percentile": 13.1,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/microsoft/Phi-4-reasoning",
            "measuredAt": "2025-04-30"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 74.3,
            "percentile": 24.1,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/microsoft/Phi-4-reasoning",
            "measuredAt": "2025-04-30"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 62.9,
            "percentile": 16.3,
            "peerCount": 40,
            "sourceUrl": "https://huggingface.co/microsoft/Phi-4-reasoning",
            "measuredAt": "2025-04-30"
          }
        ]
      },
      {
        "modelId": "amazon-nova-premier",
        "name": "Amazon Nova Premier",
        "providerId": "amazon",
        "releaseDate": "2025-04-30",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 5.2,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Reasoning",
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 42.4,
            "percentile": 5.2,
            "peerCount": 48,
            "sourceUrl": "https://aws.amazon.com/about-aws/whats-new/2025/04/amazon-nova-premier-complex-tasks-model-distillation/",
            "measuredAt": "2025-04-30"
          }
        ]
      },
      {
        "modelId": "phi-4-mini-reasoning",
        "name": "Phi-4 Mini Reasoning",
        "providerId": "microsoft",
        "releaseDate": "2025-04-30",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 5,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 52,
            "percentile": 5,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/microsoft/Phi-4-mini-reasoning",
            "measuredAt": "2025-04-30"
          }
        ]
      },
      {
        "modelId": "qwen3-235b-a22b",
        "name": "Qwen3 235B-A22B",
        "providerId": "qwen",
        "releaseDate": "2025-04-29",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 24.9,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 70.7,
            "percentile": 18.5,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 80.8,
            "percentile": 31.3,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "o3",
        "name": "o3",
        "providerId": "openai",
        "releaseDate": "2025-04-16",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 32.3,
          "Reasoning": 38.9,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 69.1,
            "percentile": 32.3,
            "peerCount": 48,
            "sourceUrl": "https://openai.com/index/introducing-gpt-5-for-developers/",
            "measuredAt": "2025-08-07"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 81.8,
            "percentile": 33.8,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 89.2,
            "percentile": 55,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 33.3,
            "percentile": 28,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-4-1",
        "name": "GPT-4.1",
        "providerId": "openai",
        "releaseDate": "2025-04-14",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 15.6,
          "Reasoning": 9.1,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 54.6,
            "percentile": 15.6,
            "peerCount": 48,
            "sourceUrl": "https://openai.com/index/gpt-4-1/",
            "measuredAt": "2025-04-14"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 66.3,
            "percentile": 14,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-4-1/",
            "measuredAt": "2025-04-14"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 6,
            "percentile": 4.2,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-4-1-mini",
        "name": "GPT-4.1 mini",
        "providerId": "openai",
        "releaseDate": "2025-04-14",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 9.1,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 65,
            "percentile": 12.2,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-4-1/",
            "measuredAt": "2025-04-14"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 6.7,
            "percentile": 5.9,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-4-1-nano",
        "name": "GPT-4.1 nano",
        "providerId": "openai",
        "releaseDate": "2025-04-14",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 3.2,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 50.3,
            "percentile": 3.2,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-4-1/",
            "measuredAt": "2025-04-14"
          }
        ]
      },
      {
        "modelId": "glm-4-32b-0414-128k",
        "name": "GLM-4 32B 0414 128K",
        "providerId": "zai",
        "releaseDate": "2025-04-14",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "llama-4-maverick",
        "name": "Llama 4 Maverick",
        "providerId": "meta",
        "releaseDate": "2025-04-05",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 30.4,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 69.8,
            "percentile": 17.6,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
            "measuredAt": "2025-04-05"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 80.5,
            "percentile": 43.1,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
            "measuredAt": "2025-04-05"
          }
        ]
      },
      {
        "modelId": "llama-4-scout",
        "name": "Llama 4 Scout",
        "providerId": "meta",
        "releaseDate": "2025-04-05",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 16.4,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 57.2,
            "percentile": 8.6,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
            "measuredAt": "2025-04-05"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 74.3,
            "percentile": 24.1,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
            "measuredAt": "2025-04-05"
          }
        ]
      },
      {
        "modelId": "command-a",
        "name": "Command A",
        "providerId": "cohere",
        "releaseDate": "2025-03-13",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 15.5,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 69.6,
            "percentile": 15.5,
            "peerCount": 29,
            "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
            "measuredAt": "2025-03-13"
          }
        ]
      },
      {
        "modelId": "phi-4-multimodal-instruct",
        "name": "Phi-4 Multimodal Instruct",
        "providerId": "microsoft",
        "releaseDate": "2025-02-26",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "claude-sonnet-3-7",
        "name": "Claude 3.7 Sonnet",
        "providerId": "anthropic",
        "releaseDate": "2025-02-24",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 40.6,
          "Reasoning": 20,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 70.3,
            "percentile": 40.6,
            "peerCount": 48,
            "sourceUrl": "https://www.anthropic.com/news/claude-3-7-sonnet",
            "measuredAt": "2025-02-24"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 79.7,
            "percentile": 31.1,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 49.2,
            "percentile": 8.8,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "sonar-deep-research",
        "name": "Sonar Deep Research",
        "providerId": "perplexity",
        "releaseDate": "2025-02-14",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "sonar",
        "name": "Sonar",
        "providerId": "perplexity",
        "releaseDate": "2025-01-21",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "sonar-pro",
        "name": "Sonar Pro",
        "providerId": "perplexity",
        "releaseDate": "2025-01-21",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "deepseek-r1",
        "name": "DeepSeek R1",
        "providerId": "deepseek",
        "releaseDate": "2025-01-20",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 11.5,
          "Reasoning": 37,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 49.2,
            "percentile": 11.5,
            "peerCount": 48,
            "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-R1/blob/main/README.md",
            "measuredAt": "2025-01-20"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 71.5,
            "percentile": 22.5,
            "peerCount": 111,
            "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-R1/blob/main/README.md",
            "measuredAt": "2025-01-20"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 84,
            "percentile": 67.2,
            "peerCount": 29,
            "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-R1/blob/main/README.md",
            "measuredAt": "2025-01-20"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 70,
            "percentile": 21.3,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "codestral",
        "name": "Codestral",
        "providerId": "mistral",
        "releaseDate": "2025-01-13",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "deepseek-v3",
        "name": "DeepSeek V3",
        "providerId": "deepseek",
        "releaseDate": "2024-12-26",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 3.1,
          "Reasoning": 15.3,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 42,
            "percentile": 3.1,
            "peerCount": 48,
            "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-V3",
            "measuredAt": "2024-12-26"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 59.1,
            "percentile": 10.4,
            "peerCount": 111,
            "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-V3",
            "measuredAt": "2024-12-26"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 75.9,
            "percentile": 29.3,
            "peerCount": 29,
            "sourceUrl": "https://github.com/deepseek-ai/DeepSeek-V3",
            "measuredAt": "2024-12-26"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 25,
            "percentile": 6.3,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "command-r7b",
        "name": "Command R7B",
        "providerId": "cohere",
        "releaseDate": "2024-12-16",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 1.7,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 42.4,
            "percentile": 1.7,
            "peerCount": 29,
            "sourceUrl": "https://cohere.com/research/papers/command-a-technical-report.pdf",
            "measuredAt": "2025-03-13"
          }
        ]
      },
      {
        "modelId": "phi-4",
        "name": "Phi-4",
        "providerId": "microsoft",
        "releaseDate": "2024-12-12",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 13.4,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 56.1,
            "percentile": 7.7,
            "peerCount": 111,
            "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
            "measuredAt": "2024-12-12"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 70.4,
            "percentile": 19,
            "peerCount": 29,
            "sourceUrl": "https://www.microsoft.com/en-us/research/wp-content/uploads/2024/12/P4TechReport.pdf",
            "measuredAt": "2024-12-12"
          }
        ]
      },
      {
        "modelId": "llama-3-3-70b-instruct",
        "name": "Llama 3.3 70B Instruct",
        "providerId": "meta",
        "releaseDate": "2024-12-06",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 6.4,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 50.5,
            "percentile": 4.1,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
            "measuredAt": "2025-04-05"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 68.9,
            "percentile": 8.6,
            "peerCount": 29,
            "sourceUrl": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
            "measuredAt": "2025-04-05"
          }
        ]
      },
      {
        "modelId": "amazon-nova-lite",
        "name": "Amazon Nova Lite",
        "providerId": "amazon",
        "releaseDate": "2024-12-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "amazon-nova-pro",
        "name": "Amazon Nova Pro",
        "providerId": "amazon",
        "releaseDate": "2024-12-02",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "claude-sonnet-3-5",
        "name": "Claude 3.5 Sonnet",
        "providerId": "anthropic",
        "releaseDate": "2024-10-22",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 9.4,
          "Reasoning": 4.1,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 49,
            "percentile": 9.4,
            "peerCount": 48,
            "sourceUrl": "https://www.anthropic.com/news/claude-3-5-sonnet",
            "measuredAt": "2024-10-22"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 55.3,
            "percentile": 6.8,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 3.3,
            "percentile": 1.3,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "command-r",
        "name": "Command R",
        "providerId": "cohere",
        "releaseDate": "2024-08-30",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "command-r-plus",
        "name": "Command R+",
        "providerId": "cohere",
        "releaseDate": "2024-08-30",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "gpt-4o-mini",
        "name": "GPT-4o mini",
        "providerId": "openai",
        "releaseDate": "2024-07-18",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 1.5,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 40.2,
            "percentile": 0.5,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-4-1/",
            "measuredAt": "2025-04-14"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 0.7,
            "percentile": 2.5,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "gpt-4o",
        "name": "GPT-4o",
        "providerId": "openai",
        "releaseDate": "2024-05-13",
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 2.3,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 46,
            "percentile": 2.3,
            "peerCount": 111,
            "sourceUrl": "https://openai.com/index/gpt-4-1/",
            "measuredAt": "2025-04-14"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 11.7,
            "percentile": 3.8,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 0.4,
            "percentile": 0.8,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-6-35b-a3b",
        "name": "Qwen3.6 35B-A3B",
        "providerId": "qwen",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 57.3,
          "Reasoning": 47,
          "Agents": null
        },
        "coverage": 4,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 73.4,
            "percentile": 57.3,
            "peerCount": 48,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 86,
            "percentile": 49.1,
            "peerCount": 111,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 85.2,
            "percentile": 79.3,
            "peerCount": 29,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 17.5,
            "percentile": 12.7,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "glm-5",
        "name": "GLM-5",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 82.3,
          "Reasoning": 71.1,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 77.8,
            "percentile": 82.3,
            "peerCount": 48,
            "sourceUrl": "https://docs.z.ai/guides/llm/glm-5",
            "measuredAt": "2026-02-11"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 87.8,
            "percentile": 55.9,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 96.7,
            "percentile": 86.3,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-5-27b",
        "name": "Qwen3.5 27B",
        "providerId": "qwen",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 67.7,
          "Reasoning": 65.9,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-verified",
            "benchmark": "SWE-bench Verified",
            "category": "Coding",
            "rawScore": 75,
            "percentile": 67.7,
            "peerCount": 48,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 85.5,
            "percentile": 47.3,
            "peerCount": 111,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          },
          {
            "axisId": "mmlu-pro",
            "benchmark": "MMLU-Pro",
            "category": "Reasoning",
            "rawScore": 86.1,
            "percentile": 84.5,
            "peerCount": 29,
            "sourceUrl": "https://qwen.ai/blog?id=qwen3.6-27b",
            "measuredAt": "2026-04-21"
          }
        ]
      },
      {
        "modelId": "qwen3-8-flash-next",
        "name": "Qwen3.8-Flash-Next",
        "providerId": "qwen",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": 78.5,
          "Reasoning": 78.8,
          "Agents": null
        },
        "coverage": 3,
        "categoryCoverage": 2,
        "missingCategories": [
          "Agents"
        ],
        "components": [
          {
            "axisId": "swe-pro",
            "benchmark": "SWE-bench Pro",
            "category": "Coding",
            "rawScore": 62.5,
            "percentile": 70.5,
            "peerCount": 22,
            "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
            "measuredAt": "2026-09-16"
          },
          {
            "axisId": "livecodebench-6",
            "benchmark": "LiveCodeBench v6",
            "category": "Coding",
            "rawScore": 91.9,
            "percentile": 86.4,
            "peerCount": 11,
            "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
            "measuredAt": "2026-09-16"
          },
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 91.7,
            "percentile": 78.8,
            "peerCount": 111,
            "sourceUrl": "https://huggingface.co/Qwen/Qwen3.8-Flash-Next",
            "measuredAt": "2026-09-16"
          }
        ]
      },
      {
        "modelId": "grok-4-3",
        "name": "Grok 4.3",
        "providerId": "xai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 49.1,
          "Agents": null
        },
        "coverage": 2,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 88.8,
            "percentile": 63.5,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          },
          {
            "axisId": "frontiermath-1-3-epoch",
            "benchmark": "FrontierMath Tiers 1-3 v2 (Epoch AI run)",
            "category": "Reasoning",
            "rawScore": 42.8,
            "percentile": 34.7,
            "peerCount": 59,
            "sourceUrl": "https://epoch.ai/benchmarks/frontiermath",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "glm-4-5",
        "name": "GLM-4.5",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 76.3,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 93.3,
            "percentile": 76.3,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "glm-4-5-air",
        "name": "GLM-4.5-Air",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 33.8,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 83.3,
            "percentile": 33.8,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "glm-4-6",
        "name": "GLM-4.6",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 68.8,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "aime-2025",
            "benchmark": "AIME 2025",
            "category": "Reasoning",
            "rawScore": 91.7,
            "percentile": 68.8,
            "peerCount": 40,
            "sourceUrl": "https://matharena.ai/?comp=aime--aime_2025",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "glm-4-7-flash",
        "name": "GLM-4.7-Flash",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 11.3,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 60.5,
            "percentile": 11.3,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "qwen3-5-35b-a3b",
        "name": "Qwen3.5 35B-A3B",
        "providerId": "qwen",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": 43.7,
          "Agents": null
        },
        "coverage": 1,
        "categoryCoverage": 1,
        "missingCategories": [
          "Coding",
          "Agents"
        ],
        "components": [
          {
            "axisId": "gpqa",
            "benchmark": "GPQA Diamond",
            "category": "Reasoning",
            "rawScore": 83.5,
            "percentile": 43.7,
            "peerCount": 111,
            "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
            "measuredAt": "2026-09-17"
          }
        ]
      },
      {
        "modelId": "glm-4-5-airx",
        "name": "GLM-4.5-AirX",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "glm-4-5-flash",
        "name": "GLM-4.5-Flash",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "glm-4-5-x",
        "name": "GLM-4.5-X",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "glm-4-7-flashx",
        "name": "GLM-4.7-FlashX",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "glm-5-turbo",
        "name": "GLM-5-Turbo",
        "providerId": "zai",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "mimo-v2-5",
        "name": "MiMo-V2.5",
        "providerId": "xiaomi",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "mimo-v2-5-pro",
        "name": "MiMo-V2.5-Pro",
        "providerId": "xiaomi",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "north-small-translate",
        "name": "North Small Translate",
        "providerId": "cohere",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "qwen3-5-122b-a10b",
        "name": "Qwen3.5 122B-A10B",
        "providerId": "qwen",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      },
      {
        "modelId": "sonar-reasoning-pro",
        "name": "Sonar Reasoning Pro",
        "providerId": "perplexity",
        "releaseDate": null,
        "rank": null,
        "score": null,
        "categoryScores": {
          "Coding": null,
          "Reasoning": null,
          "Agents": null
        },
        "coverage": 0,
        "categoryCoverage": 0,
        "missingCategories": [
          "Coding",
          "Reasoning",
          "Agents"
        ],
        "components": []
      }
    ]
  },
  "meta": {
    "source": "https://llmmetric.com",
    "attribution": "Please cite llmmetric.com. Upstream records remain subject to their source terms.",
    "generatedAt": "2026-09-22T22:54:59.398Z"
  }
}