{
  "updated": "2026-09-23",
  "version": "1.0",
  "sources": [
    {
      "id": "databench",
      "name": "Hex DataBench v1.1",
      "url": "https://hex.tech/databench/",
      "methodologyUrl": "https://hex.tech/blog/databench-agentic-analytics-benchmark/",
      "date": "2026-09-22",
      "dateLabel": "Leaderboard updated",
      "description": "Analytics agents working in a synthetic business warehouse. The published blended score reflects task rubrics, with API cost and median duration measured in the Hex harness.",
      "limitations": "The v1 design contains 100 tasks. An LLM judge evaluates the work. Version 1.1 changed judge calibration on September 17; older v1 scores are not mixed in. No confidence intervals are published. These costs and times describe this harness, not every agent."
    },
    {
      "id": "livebench",
      "name": "LiveBench",
      "url": "https://livebench.ai/",
      "methodologyUrl": "https://github.com/LiveBench/new-livebench",
      "date": null,
      "dateLabel": "Question-set date shown per study",
      "description": "Checkable benchmark tasks grouped into reasoning, coding, agentic coding, mathematics, data, language and instruction following. Category means and task-weighted Agent Fit are calculated from the published task scores.",
      "limitations": "Each sweep uses one question set and one model identity. The question-set date is not a model launch or evaluation date. Unknown default efforts are excluded. Older sweeps do not establish the behavior of newer model versions. No per-mode cost or confidence interval is inferred."
    },
    {
      "id": "ockbench",
      "name": "OckBench",
      "url": "https://ockbench.github.io/",
      "methodologyUrl": "https://github.com/OckBench/OckBench",
      "date": null,
      "dateLabel": "Retrieved September 23, 2026",
      "description": "200 selected problems: 100 mathematics, 60 coding and 40 science. Accuracy and average output tokens expose quality versus token use; domain results show where effort helps.",
      "limitations": "Mathematics uses an LLM judge; code runs against tests and science checks answers. Sampling follows model-specific configurations. Opus 5 medium includes nine persistent zero-output failures, counted as incorrect but omitted from average output tokens. Token counts are not dollars or latency."
    },
    {
      "id": "stet",
      "name": "Stet coding case study",
      "url": "https://www.stet.sh/blog/gpt-55-codex-graphql-reasoning-curve",
      "methodologyUrl": "https://www.stet.sh/blog/gpt-55-codex-graphql-reasoning-curve",
      "date": "2026-05-07",
      "dateLabel": "Published",
      "description": "GPT-5.5 in Codex on 26 matched GraphQL-go-tools repository tasks. Test outcomes, equivalence to human changes and AI review outcomes answer different coding-quality questions.",
      "limitations": "Exploratory evidence: one repository, one seed per task, stitched runs and uncalibrated GPT-5.4 judging. The publisher calls this inspect-grade, not decision-grade evidence. Xhigh costs retain pre-regrade records. Do not generalize small gaps."
    }
  ],
  "families": [
    {
      "slug": "claude-fable-5",
      "name": "Claude Fable 5",
      "provider": "Anthropic",
      "profileSlug": "claude-fable-5",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "Fable 5 \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 15.0
              },
              "cost": 1.136589,
              "latency": 79.822,
              "reportedTokens": 4707.063,
              "order": 2
            },
            {
              "id": "Fable 5 \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 14.6667
              },
              "cost": 1.504295,
              "latency": 124.995,
              "reportedTokens": 7237.613,
              "order": 3
            },
            {
              "id": "Fable 5 \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 16.0
              },
              "cost": 1.880558,
              "latency": 156.708,
              "reportedTokens": 10213.29,
              "order": 4
            },
            {
              "id": "Fable 5 \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 18.0
              },
              "cost": 3.486334,
              "latency": 379.986,
              "reportedTokens": 27499.133,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "provider": "Anthropic",
      "profileSlug": "claude-fable-5-1",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "Fable 5.1 \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 16.3333
              },
              "cost": 0.924505,
              "latency": 101.654,
              "reportedTokens": 6069.003,
              "order": 2
            },
            {
              "id": "Fable 5.1 \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 21.0
              },
              "cost": 1.182898,
              "latency": 126.329,
              "reportedTokens": 8117.263,
              "order": 3
            },
            {
              "id": "Fable 5.1 \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 20.3333
              },
              "cost": 1.571076,
              "latency": 192.618,
              "reportedTokens": 11420.173,
              "order": 4
            },
            {
              "id": "Fable 5.1 \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 21.6667
              },
              "cost": 2.495851,
              "latency": 296.722,
              "reportedTokens": 21082.637,
              "order": 5
            },
            {
              "id": "Fable 5.1 \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 27.0
              },
              "cost": 3.358707,
              "latency": 418.999,
              "reportedTokens": 31331.52,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "claude-opus-4-5",
      "name": "Claude Opus 4.5",
      "provider": "Anthropic",
      "profileSlug": "claude-opus-4-5",
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 64K thinking budget",
          "scope": "64K thinking budget",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "claude-opus-4-5-20251101-thinking-64k-low-effort",
              "effort": "low",
              "metrics": {
                "Reasoning": 70.4423,
                "Coding": 77.48,
                "Agentic Coding": 50.0,
                "Mathematics": 76.6527,
                "Data Analysis": 53.0227,
                "Language": 80.751,
                "IF": 50.8125,
                "overall": 65.5945,
                "balanced": 59.5372,
                "coding": 61.3916,
                "research": 62.8367,
                "writing": 65.7324
              },
              "order": 2
            },
            {
              "id": "claude-opus-4-5-20251101-thinking-64k-medium-effort",
              "effort": "medium",
              "metrics": {
                "Reasoning": 82.2067,
                "Coding": 77.48,
                "Agentic Coding": 58.3333,
                "Mathematics": 85.2405,
                "Data Analysis": 72.3487,
                "Language": 80.3503,
                "IF": 60.5168,
                "overall": 73.7823,
                "balanced": 70.0582,
                "coding": 67.8767,
                "research": 74.6329,
                "writing": 71.7037
              },
              "order": 3
            },
            {
              "id": "claude-opus-4-5-20251101-thinking-64k-high-effort",
              "effort": "high",
              "metrics": {
                "Reasoning": 80.0865,
                "Coding": 79.654,
                "Agentic Coding": 63.3333,
                "Mathematics": 90.389,
                "Data Analysis": 74.4417,
                "Language": 81.261,
                "IF": 62.546,
                "overall": 75.9588,
                "balanced": 71.4608,
                "coding": 70.6638,
                "research": 75.0611,
                "writing": 72.6631
              },
              "order": 4
            }
          ]
        },
        {
          "id": "livebench-2026_01_08-no-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Without extended thinking",
          "scope": "Without extended thinking",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "claude-opus-4-5-20251101-low-effort",
              "effort": "low",
              "metrics": {
                "Reasoning": 47.8125,
                "Coding": 78.1845,
                "Agentic Coding": 50.0,
                "Mathematics": 64.1033,
                "Data Analysis": 44.1667,
                "Language": 77.1787,
                "IF": 28.9543,
                "overall": 55.7714,
                "balanced": 46.0258,
                "coding": 56.0226,
                "research": 47.352,
                "writing": 51.0728
              },
              "order": 2
            },
            {
              "id": "claude-opus-4-5-20251101-medium-effort",
              "effort": "medium",
              "metrics": {
                "Reasoning": 53.2115,
                "Coding": 78.506,
                "Agentic Coding": 63.3333,
                "Mathematics": 66.3178,
                "Data Analysis": 45.5413,
                "Language": 78.6567,
                "IF": 28.1085,
                "overall": 59.0964,
                "balanced": 50.339,
                "coding": 62.8444,
                "research": 49.7066,
                "writing": 52.0932
              },
              "order": 3
            },
            {
              "id": "claude-opus-4-5-20251101-high-effort",
              "effort": "high",
              "metrics": {
                "Reasoning": 54.9038,
                "Coding": 77.8015,
                "Agentic Coding": 61.6667,
                "Mathematics": 66.445,
                "Data Analysis": 45.6123,
                "Language": 77.0917,
                "IF": 26.5917,
                "overall": 58.5875,
                "balanced": 50.0744,
                "coding": 61.9852,
                "research": 49.7821,
                "writing": 51.0385
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "claude-opus-4-7",
      "name": "Claude Opus 4.7",
      "provider": "Anthropic",
      "profileSlug": "claude-opus-4-7",
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Thinking",
          "scope": "Thinking",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "claude-opus-4-7-low-effort",
              "effort": "low",
              "metrics": {
                "Reasoning": 74.8365,
                "Coding": 76.4545,
                "Agentic Coding": 60.0,
                "Mathematics": 76.3405,
                "Data Analysis": 75.5977,
                "Language": 74.5567,
                "IF": 46.1123,
                "overall": 69.1283,
                "balanced": 64.9641,
                "coding": 65.773,
                "research": 69.278,
                "writing": 61.7987
              },
              "order": 2
            },
            {
              "id": "claude-opus-4-7-medium-effort",
              "effort": "medium",
              "metrics": {
                "Reasoning": 80.0337,
                "Coding": 79.9755,
                "Agentic Coding": 56.6667,
                "Mathematics": 86.3792,
                "Data Analysis": 75.75,
                "Language": 76.5297,
                "IF": 50.8667,
                "overall": 72.3145,
                "balanced": 67.4202,
                "coding": 66.5844,
                "research": 72.3896,
                "writing": 65.507
              },
              "order": 3
            },
            {
              "id": "claude-opus-4-7-high-effort",
              "effort": "high",
              "metrics": {
                "Reasoning": 83.827,
                "Coding": 83.175,
                "Agentic Coding": 61.6667,
                "Mathematics": 89.293,
                "Data Analysis": 77.269,
                "Language": 74.4783,
                "IF": 54.5,
                "overall": 74.887,
                "balanced": 71.0143,
                "coding": 70.7266,
                "research": 74.5919,
                "writing": 66.8904
              },
              "order": 4
            },
            {
              "id": "claude-opus-4-7-xhigh-effort",
              "effort": "xhigh",
              "metrics": {
                "Reasoning": 87.6923,
                "Coding": 82.088,
                "Agentic Coding": 60.0,
                "Mathematics": 93.1038,
                "Data Analysis": 78.2637,
                "Language": 77.914,
                "IF": 59.3417,
                "overall": 76.9148,
                "balanced": 73.0915,
                "coding": 70.7144,
                "research": 77.7268,
                "writing": 71.0232
              },
              "order": 5
            }
          ]
        }
      ]
    },
    {
      "slug": "claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "provider": "Anthropic",
      "profileSlug": "claude-opus-4-8",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "Opus 4.8 \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 12.0
              },
              "cost": 0.60302,
              "latency": 96.2,
              "reportedTokens": 6889.67,
              "order": 2
            },
            {
              "id": "Opus 4.8 \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 17.0
              },
              "cost": 0.758971,
              "latency": 133.357,
              "reportedTokens": 10074.123,
              "order": 3
            },
            {
              "id": "Opus 4.8 \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 14.0
              },
              "cost": 0.857424,
              "latency": 186.73,
              "reportedTokens": 12218.97,
              "order": 4
            },
            {
              "id": "Opus 4.8 \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 18.6667
              },
              "cost": 1.200635,
              "latency": 251.027,
              "reportedTokens": 20038.797,
              "order": 5
            },
            {
              "id": "Opus 4.8 \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 15.0
              },
              "cost": 1.783553,
              "latency": 433.948,
              "reportedTokens": 34704.827,
              "order": 6
            }
          ]
        },
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Thinking",
          "scope": "Thinking",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "claude-opus-4-8-low-effort",
              "effort": "low",
              "metrics": {
                "Reasoning": 83.5145,
                "Coding": 76.4545,
                "Agentic Coding": 38.3333,
                "Mathematics": 80.4873,
                "Data Analysis": 76.4037,
                "Language": 77.3477,
                "IF": 59.1168,
                "overall": 70.2368,
                "balanced": 66.6062,
                "coding": 58.6252,
                "research": 75.5767,
                "writing": 70.0688
              },
              "order": 2
            },
            {
              "id": "claude-opus-4-8-medium-effort",
              "effort": "medium",
              "metrics": {
                "Reasoning": 84.4905,
                "Coding": 78.8885,
                "Agentic Coding": 58.3333,
                "Mathematics": 81.6287,
                "Data Analysis": 77.2107,
                "Language": 80.3927,
                "IF": 58.125,
                "overall": 74.1528,
                "balanced": 71.0155,
                "coding": 68.4026,
                "research": 76.4188,
                "writing": 70.9869
              },
              "order": 3
            },
            {
              "id": "claude-opus-4-8-high-effort",
              "effort": "high",
              "metrics": {
                "Reasoning": 89.6923,
                "Coding": 76.393,
                "Agentic Coding": 60.0,
                "Mathematics": 82.0002,
                "Data Analysis": 78.0437,
                "Language": 81.644,
                "IF": 63.3625,
                "overall": 75.8765,
                "balanced": 74.0941,
                "coding": 69.708,
                "research": 79.7245,
                "writing": 74.6246
              },
              "order": 4
            },
            {
              "id": "claude-opus-4-8-xhigh-effort",
              "effort": "xhigh",
              "metrics": {
                "Reasoning": 89.7115,
                "Coding": 79.2715,
                "Agentic Coding": 60.0,
                "Mathematics": 84.3165,
                "Data Analysis": 78.344,
                "Language": 81.4173,
                "IF": 67.4458,
                "overall": 77.2152,
                "balanced": 75.4536,
                "coding": 70.9827,
                "research": 80.604,
                "writing": 76.3742
              },
              "order": 5
            }
          ]
        },
        {
          "id": "ockbench-200",
          "source": "ockbench",
          "name": "Math, code & science \u00b7 OckBench",
          "scope": "200 selected problems",
          "dataUrl": "https://ockbench.github.io/static/data/top200_model_performance.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall accuracy",
              "description": "Correct answers out of 200 problems. Domain proportions are 100 math, 60 coding and 40 science."
            },
            {
              "key": "Math",
              "label": "Mathematics",
              "description": "Accuracy on 100 selected mathematics problems, graded with an LLM judge."
            },
            {
              "key": "Coding",
              "label": "Coding tests",
              "description": "Accuracy on 60 selected programming problems checked with executable tests."
            },
            {
              "key": "Science",
              "label": "Science",
              "description": "Accuracy on 40 selected science problems with checked multiple-choice answers."
            }
          ],
          "rows": [
            {
              "id": "opus-4.8-low",
              "effort": "low",
              "metrics": {
                "overall": 73.5,
                "Math": 80.0,
                "Coding": 70.0,
                "Science": 62.5
              },
              "outputTokens": 3652.0,
              "domainTokens": {
                "Math": 6214.0,
                "Coding": 398.0,
                "Science": 2128.0
              },
              "sampleSize": 200,
              "order": 2
            },
            {
              "id": "opus-4.8-medium",
              "effort": "medium",
              "metrics": {
                "overall": 86.0,
                "Math": 88.0,
                "Coding": 86.7,
                "Science": 80.0
              },
              "outputTokens": 7712.0,
              "domainTokens": {
                "Math": 11207.0,
                "Coding": 3000.0,
                "Science": 6042.0
              },
              "sampleSize": 200,
              "order": 3
            },
            {
              "id": "opus-4.8-high",
              "effort": "high",
              "metrics": {
                "overall": 86.0,
                "Math": 91.0,
                "Coding": 81.7,
                "Science": 80.0
              },
              "outputTokens": 8870.0,
              "domainTokens": {
                "Math": 13934.0,
                "Coding": 1591.0,
                "Science": 7131.0
              },
              "sampleSize": 200,
              "order": 4
            },
            {
              "id": "opus-4.8-xhigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 91.5,
                "Math": 90.0,
                "Coding": 93.3,
                "Science": 92.5
              },
              "outputTokens": 16876.0,
              "domainTokens": {
                "Math": 25038.0,
                "Coding": 4002.0,
                "Science": 15785.0
              },
              "sampleSize": 200,
              "order": 5
            },
            {
              "id": "opus-4.8-max",
              "effort": "max",
              "metrics": {
                "overall": 90.0,
                "Math": 84.0,
                "Coding": 100.0,
                "Science": 90.0
              },
              "outputTokens": 36544.0,
              "domainTokens": {
                "Math": 55252.0,
                "Coding": 6906.0,
                "Science": 34230.0
              },
              "sampleSize": 200,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "claude-opus-5",
      "name": "Claude Opus 5",
      "provider": "Anthropic",
      "profileSlug": "claude-opus-5",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "Opus 5 \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 15.6667
              },
              "cost": 0.744621,
              "latency": 98.161,
              "reportedTokens": 6922.533,
              "order": 2
            },
            {
              "id": "Opus 5 \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 17.0
              },
              "cost": 1.101234,
              "latency": 149.473,
              "reportedTokens": 11426.72,
              "order": 3
            },
            {
              "id": "Opus 5 \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 18.0
              },
              "cost": 1.611204,
              "latency": 246.499,
              "reportedTokens": 18898.455,
              "order": 4
            },
            {
              "id": "Opus 5 \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 17.3333
              },
              "cost": 1.887681,
              "latency": 337.463,
              "reportedTokens": 24172.323,
              "order": 5
            },
            {
              "id": "Opus 5 \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 19.6667
              },
              "cost": 2.109061,
              "latency": 392.729,
              "reportedTokens": 29090.98,
              "order": 6
            }
          ]
        },
        {
          "id": "ockbench-200",
          "source": "ockbench",
          "name": "Math, code & science \u00b7 OckBench",
          "scope": "200 selected problems",
          "dataUrl": "https://ockbench.github.io/static/data/top200_model_performance.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall accuracy",
              "description": "Correct answers out of 200 problems. Domain proportions are 100 math, 60 coding and 40 science."
            },
            {
              "key": "Math",
              "label": "Mathematics",
              "description": "Accuracy on 100 selected mathematics problems, graded with an LLM judge."
            },
            {
              "key": "Coding",
              "label": "Coding tests",
              "description": "Accuracy on 60 selected programming problems checked with executable tests."
            },
            {
              "key": "Science",
              "label": "Science",
              "description": "Accuracy on 40 selected science problems with checked multiple-choice answers."
            }
          ],
          "rows": [
            {
              "id": "opus-5-low",
              "effort": "low",
              "metrics": {
                "overall": 85.0,
                "Math": 82.0,
                "Coding": 95.0,
                "Science": 77.5
              },
              "outputTokens": 1623.0,
              "domainTokens": {
                "Math": 2763.0,
                "Coding": 550.0,
                "Science": 384.0
              },
              "sampleSize": 200,
              "order": 2
            },
            {
              "id": "opus-5-medium",
              "effort": "medium",
              "metrics": {
                "overall": 85.0,
                "Math": 85.0,
                "Coding": 86.7,
                "Science": 82.5
              },
              "outputTokens": 4215.0,
              "domainTokens": {
                "Math": 6715.0,
                "Coding": 968.0,
                "Science": 2133.0
              },
              "sampleSize": 200,
              "order": 3
            },
            {
              "id": "opus-5-high",
              "effort": "high",
              "metrics": {
                "overall": 94.0,
                "Math": 93.0,
                "Coding": 100.0,
                "Science": 87.5
              },
              "outputTokens": 6745.0,
              "domainTokens": {
                "Math": 11442.0,
                "Coding": 968.0,
                "Science": 3666.0
              },
              "sampleSize": 200,
              "order": 4
            },
            {
              "id": "opus-5-xhigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 95.0,
                "Math": 94.0,
                "Coding": 100.0,
                "Science": 90.0
              },
              "outputTokens": 9740.0,
              "domainTokens": {
                "Math": 15652.0,
                "Coding": 1929.0,
                "Science": 6679.0
              },
              "sampleSize": 200,
              "order": 5
            },
            {
              "id": "opus-5-max",
              "effort": "max",
              "metrics": {
                "overall": 94.5,
                "Math": 94.0,
                "Coding": 100.0,
                "Science": 87.5
              },
              "outputTokens": 13130.0,
              "domainTokens": {
                "Math": 21838.0,
                "Coding": 1845.0,
                "Science": 8289.0
              },
              "sampleSize": 200,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "claude-opus-5-5",
      "name": "Claude Opus 5.5",
      "provider": "Anthropic",
      "profileSlug": "claude-opus-5-5",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "Opus 5.5 \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 54.0
              },
              "cost": 0.61010294975,
              "latency": 110.751,
              "reportedTokens": 13010.66,
              "order": 2
            },
            {
              "id": "Opus 5.5 \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 62.5
              },
              "cost": 0.97343032025,
              "latency": 169.561,
              "reportedTokens": 22482.31,
              "order": 3
            },
            {
              "id": "Opus 5.5 \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 65.0
              },
              "cost": 1.248579639,
              "latency": 220.141,
              "reportedTokens": 30534.055,
              "order": 4
            },
            {
              "id": "Opus 5.5 \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 67.0
              },
              "cost": 2.1610205525,
              "latency": 387.65,
              "reportedTokens": 62781.995,
              "order": 5
            },
            {
              "id": "Opus 5.5 \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 70.5
              },
              "cost": 3.5739114765,
              "latency": 677.163,
              "reportedTokens": 115101.01,
              "order": 6
            }
          ]
        },
        {
          "id": "livebench-2026_06_25-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-06-25 \u00b7 Thinking",
          "scope": "Thinking",
          "questionSet": "2026-06-25",
          "dataUrl": "https://livebench.ai/table_2026_06_25.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "claude-opus-5-5-xhigh-effort",
              "effort": "xhigh",
              "metrics": {
                "Reasoning": 90.6538,
                "Coding": 89.253,
                "Agentic Coding": 65.3533,
                "Mathematics": 96.803,
                "Data Analysis": 79.803,
                "Language": 85.4987,
                "IF": 67.0498,
                "overall": 82.0592,
                "balanced": 77.925,
                "coding": 76.4879,
                "research": 81.9045,
                "writing": 77.9699
              },
              "order": 5
            },
            {
              "id": "claude-opus-5-5-max-effort",
              "effort": "max",
              "metrics": {
                "Reasoning": 92.1538,
                "Coding": 89.253,
                "Agentic Coding": 71.717,
                "Mathematics": 97.0775,
                "Data Analysis": 80.307,
                "Language": 86.27,
                "IF": 65.7378,
                "overall": 83.2166,
                "balanced": 79.3953,
                "coding": 79.4454,
                "research": 82.434,
                "writing": 77.913
              },
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "claude-sonnet-4-6",
      "name": "Claude Sonnet 4.6",
      "provider": "Anthropic",
      "profileSlug": "claude-sonnet-4-6",
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Adaptive thinking",
          "scope": "Adaptive thinking",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "claude-sonnet-4-6-thinking-auto-low-effort",
              "effort": "low",
              "metrics": {
                "Reasoning": 77.4423,
                "Coding": 74.2805,
                "Agentic Coding": 63.3333,
                "Mathematics": 80.382,
                "Data Analysis": 74.6113,
                "Language": 71.3867,
                "IF": 51.6332,
                "overall": 70.4385,
                "balanced": 67.4274,
                "coding": 67.5638,
                "research": 70.5228,
                "writing": 63.406
              },
              "order": 2
            },
            {
              "id": "claude-sonnet-4-6-thinking-auto-medium-effort",
              "effort": "medium",
              "metrics": {
                "Reasoning": 84.7692,
                "Coding": 79.2715,
                "Agentic Coding": 60.0,
                "Mathematics": 86.994,
                "Data Analysis": 77.946,
                "Language": 76.1027,
                "IF": 63.2208,
                "overall": 75.472,
                "balanced": 72.855,
                "coding": 69.8189,
                "research": 77.1126,
                "writing": 71.6058
              },
              "order": 3
            },
            {
              "id": "claude-sonnet-4-6-thinking-auto-high-effort",
              "effort": "high",
              "metrics": {
                "Reasoning": 86.375,
                "Coding": 79.9755,
                "Agentic Coding": 56.6667,
                "Mathematics": 86.5337,
                "Data Analysis": 76.0567,
                "Language": 77.6933,
                "IF": 63.9165,
                "overall": 75.3168,
                "balanced": 72.631,
                "coding": 68.8405,
                "research": 77.4856,
                "writing": 72.796
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "provider": "Anthropic",
      "profileSlug": "claude-sonnet-5",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "Sonnet 5 \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 11.6667
              },
              "cost": 0.263527,
              "latency": 75.366,
              "reportedTokens": 5865.98,
              "order": 2
            },
            {
              "id": "Sonnet 5 \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 13.3333
              },
              "cost": 0.362365,
              "latency": 113.951,
              "reportedTokens": 9081.475,
              "order": 3
            },
            {
              "id": "Sonnet 5 \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 13.6667
              },
              "cost": 0.492168,
              "latency": 170.359,
              "reportedTokens": 14321.345,
              "order": 4
            },
            {
              "id": "Sonnet 5 \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 14.3333
              },
              "cost": 0.619515,
              "latency": 213.636,
              "reportedTokens": 20164.627,
              "order": 5
            },
            {
              "id": "Sonnet 5 \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 12.3333
              },
              "cost": 1.067905,
              "latency": 450.945,
              "reportedTokens": 43089.353,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "deepseek-v4-flash",
      "name": "DeepSeek V4 Flash",
      "provider": "DeepSeek",
      "profileSlug": "deepseek-v4-flash",
      "studies": [
        {
          "id": "ockbench-200",
          "source": "ockbench",
          "name": "Math, code & science \u00b7 OckBench",
          "scope": "200 selected problems",
          "dataUrl": "https://ockbench.github.io/static/data/top200_model_performance.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall accuracy",
              "description": "Correct answers out of 200 problems. Domain proportions are 100 math, 60 coding and 40 science."
            },
            {
              "key": "Math",
              "label": "Mathematics",
              "description": "Accuracy on 100 selected mathematics problems, graded with an LLM judge."
            },
            {
              "key": "Coding",
              "label": "Coding tests",
              "description": "Accuracy on 60 selected programming problems checked with executable tests."
            },
            {
              "key": "Science",
              "label": "Science",
              "description": "Accuracy on 40 selected science problems with checked multiple-choice answers."
            }
          ],
          "rows": [
            {
              "id": "deepseek-v4-flash-none",
              "effort": "none",
              "metrics": {
                "overall": 20.5,
                "Math": 15.0,
                "Coding": 15.0,
                "Science": 42.5
              },
              "outputTokens": 3387.0,
              "domainTokens": {
                "Math": 2194.0,
                "Coding": 656.0,
                "Science": 10259.0
              },
              "sampleSize": 200,
              "order": 0
            },
            {
              "id": "deepseek-v4-flash-high",
              "effort": "high",
              "metrics": {
                "overall": 78.0,
                "Math": 79.0,
                "Coding": 78.3,
                "Science": 75.0
              },
              "outputTokens": 40735.0,
              "domainTokens": {
                "Math": 65219.0,
                "Coding": 10665.0,
                "Science": 30752.0
              },
              "sampleSize": 200,
              "order": 4
            },
            {
              "id": "deepseek-v4-flash-max",
              "effort": "max",
              "metrics": {
                "overall": 83.0,
                "Math": 85.0,
                "Coding": 90.0,
                "Science": 67.5
              },
              "outputTokens": 80870.0,
              "domainTokens": {
                "Math": 109178.0,
                "Coding": 36951.0,
                "Science": 82348.0
              },
              "sampleSize": 200,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "deepseek-v4-pro",
      "name": "DeepSeek V4 Pro",
      "provider": "DeepSeek",
      "profileSlug": "deepseek-v4-pro",
      "studies": [
        {
          "id": "ockbench-200",
          "source": "ockbench",
          "name": "Math, code & science \u00b7 OckBench",
          "scope": "200 selected problems",
          "dataUrl": "https://ockbench.github.io/static/data/top200_model_performance.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall accuracy",
              "description": "Correct answers out of 200 problems. Domain proportions are 100 math, 60 coding and 40 science."
            },
            {
              "key": "Math",
              "label": "Mathematics",
              "description": "Accuracy on 100 selected mathematics problems, graded with an LLM judge."
            },
            {
              "key": "Coding",
              "label": "Coding tests",
              "description": "Accuracy on 60 selected programming problems checked with executable tests."
            },
            {
              "key": "Science",
              "label": "Science",
              "description": "Accuracy on 40 selected science problems with checked multiple-choice answers."
            }
          ],
          "rows": [
            {
              "id": "deepseek-v4-pro-none",
              "effort": "none",
              "metrics": {
                "overall": 25.0,
                "Math": 16.0,
                "Coding": 26.7,
                "Science": 45.0
              },
              "outputTokens": 1460.0,
              "domainTokens": {
                "Math": 1343.0,
                "Coding": 2335.0,
                "Science": 442.0
              },
              "sampleSize": 200,
              "order": 0
            },
            {
              "id": "deepseek-v4-pro-high",
              "effort": "high",
              "metrics": {
                "overall": 84.0,
                "Math": 85.0,
                "Coding": 85.0,
                "Science": 80.0
              },
              "outputTokens": 43652.0,
              "domainTokens": {
                "Math": 57981.0,
                "Coding": 22497.0,
                "Science": 39561.0
              },
              "sampleSize": 200,
              "order": 4
            },
            {
              "id": "deepseek-v4-pro-max",
              "effort": "max",
              "metrics": {
                "overall": 84.0,
                "Math": 85.0,
                "Coding": 88.3,
                "Science": 75.0
              },
              "outputTokens": 81408.0,
              "domainTokens": {
                "Math": 98646.0,
                "Coding": 53951.0,
                "Science": 79503.0
              },
              "sampleSize": 200,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "deepseek-v4-1-flash",
      "name": "DeepSeek V4.1 Flash",
      "provider": "DeepSeek",
      "profileSlug": "deepseek-v4-1-flash",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "DeepSeek V4.1 Flash \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 13.6667
              },
              "cost": 0.103412,
              "latency": 117.736,
              "reportedTokens": 23309.39,
              "order": 2
            },
            {
              "id": "DeepSeek V4.1 Flash \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 14.0
              },
              "cost": 0.139078,
              "latency": 168.424,
              "reportedTokens": 31679.323,
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "gemini-3-flash-preview",
      "name": "Gemini 3 Flash Preview",
      "provider": "Google",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Preview",
          "scope": "Preview",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gemini-3-flash-preview-minimal",
              "effort": "minimal",
              "metrics": {
                "Reasoning": 49.1683,
                "Coding": 78.567,
                "Agentic Coding": 43.3333,
                "Mathematics": 68.0987,
                "Data Analysis": 48.3057,
                "Language": 78.6463,
                "IF": 28.3208,
                "overall": 56.3486,
                "balanced": 45.5999,
                "coding": 53.2774,
                "research": 49.1617,
                "writing": 51.5781
              },
              "order": 1
            },
            {
              "id": "gemini-3-flash-preview-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 74.548,
                "Coding": 73.8975,
                "Agentic Coding": 40.0,
                "Mathematics": 84.1743,
                "Data Analysis": 74.7727,
                "Language": 84.5647,
                "IF": 74.8625,
                "overall": 72.4028,
                "balanced": 67.6857,
                "coding": 58.8377,
                "research": 76.1808,
                "writing": 78.6962
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "gemini-3-pro-preview",
      "name": "Gemini 3 Pro Preview",
      "provider": "Google",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 November 2025 preview",
          "scope": "November 2025 preview",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gemini-3-pro-preview-11-2025-low",
              "effort": "low",
              "metrics": {
                "Reasoning": 70.6155,
                "Coding": 70.6365,
                "Agentic Coding": 55.0,
                "Mathematics": 77.734,
                "Data Analysis": 66.9127,
                "Language": 79.4917,
                "IF": 26.8792,
                "overall": 63.8957,
                "balanced": 56.005,
                "coding": 59.2212,
                "research": 62.0888,
                "writing": 54.4847
              },
              "order": 2
            },
            {
              "id": "gemini-3-pro-preview-11-2025-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 77.4183,
                "Coding": 74.602,
                "Agentic Coding": 55.0,
                "Mathematics": 81.837,
                "Data Analysis": 74.3897,
                "Language": 84.621,
                "IF": 65.846,
                "overall": 73.3877,
                "balanced": 69.3056,
                "coding": 65.3279,
                "research": 75.2756,
                "writing": 75.0918
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "gemini-3-1-pro-preview",
      "name": "Gemini 3.1 Pro Preview",
      "provider": "Google",
      "profileSlug": "gemini-3-1-pro-preview",
      "studies": [
        {
          "id": "ockbench-200",
          "source": "ockbench",
          "name": "Math, code & science \u00b7 OckBench",
          "scope": "200 selected problems",
          "dataUrl": "https://ockbench.github.io/static/data/top200_model_performance.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall accuracy",
              "description": "Correct answers out of 200 problems. Domain proportions are 100 math, 60 coding and 40 science."
            },
            {
              "key": "Math",
              "label": "Mathematics",
              "description": "Accuracy on 100 selected mathematics problems, graded with an LLM judge."
            },
            {
              "key": "Coding",
              "label": "Coding tests",
              "description": "Accuracy on 60 selected programming problems checked with executable tests."
            },
            {
              "key": "Science",
              "label": "Science",
              "description": "Accuracy on 40 selected science problems with checked multiple-choice answers."
            }
          ],
          "rows": [
            {
              "id": "gemini-3.1-pro-preview-low",
              "effort": "low",
              "metrics": {
                "overall": 79.5,
                "Math": 73.0,
                "Coding": 91.7,
                "Science": 77.5
              },
              "outputTokens": 3331.0,
              "domainTokens": {
                "Math": 4208.0,
                "Coding": 2652.0,
                "Science": 2155.0
              },
              "sampleSize": 200,
              "order": 2
            },
            {
              "id": "gemini-3.1-pro-preview-medium",
              "effort": "medium",
              "metrics": {
                "overall": 87.5,
                "Math": 87.0,
                "Coding": 90.0,
                "Science": 85.0
              },
              "outputTokens": 7116.0,
              "domainTokens": {
                "Math": 9805.0,
                "Coding": 4734.0,
                "Science": 3910.0
              },
              "sampleSize": 200,
              "order": 3
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5",
      "name": "GPT-5",
      "provider": "OpenAI",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2025_05_30-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2025-05-30 \u00b7 Explicit effort rows",
          "scope": "Explicit effort rows",
          "questionSet": "2025-05-30",
          "dataUrl": "https://livebench.ai/table_2025_05_30.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5-minimal",
              "effort": "minimal",
              "metrics": {
                "Reasoning": 54.861,
                "Coding": 72.5505,
                "Agentic Coding": 26.6667,
                "Mathematics": 58.9807,
                "Data Analysis": 64.433,
                "Language": 51.016,
                "IF": 76.8625,
                "overall": 57.91,
                "balanced": 57.9273,
                "coding": 49.6805,
                "research": 61.5562,
                "writing": 63.2237
              },
              "order": 1
            },
            {
              "id": "gpt-5-low",
              "effort": "low",
              "metrics": {
                "Reasoning": 90.4723,
                "Coding": 74.2805,
                "Agentic Coding": 35.0,
                "Mathematics": 85.3307,
                "Data Analysis": 69.721,
                "Language": 78.7347,
                "IF": 88.9917,
                "overall": 74.6473,
                "balanced": 74.2758,
                "coding": 60.5042,
                "research": 82.1902,
                "writing": 85.111
              },
              "order": 2
            },
            {
              "id": "gpt-5-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 98.1667,
                "Coding": 77.0975,
                "Agentic Coding": 46.6667,
                "Mathematics": 92.772,
                "Data Analysis": 71.6345,
                "Language": 80.827,
                "IF": 88.1125,
                "overall": 79.3253,
                "balanced": 79.2664,
                "coding": 67.6655,
                "research": 85.5952,
                "writing": 86.7064
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-mini",
      "name": "GPT-5 mini",
      "provider": "OpenAI",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Explicit effort rows",
          "scope": "Explicit effort rows",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5-mini-minimal",
              "effort": "minimal",
              "metrics": {
                "Reasoning": 32.298,
                "Coding": 70.698,
                "Agentic Coding": 21.6667,
                "Mathematics": 47.5305,
                "Data Analysis": 41.5063,
                "Language": 44.7887,
                "IF": 20.8125,
                "overall": 39.9001,
                "balanced": 32.5216,
                "coding": 37.8854,
                "research": 34.637,
                "writing": 32.1258
              },
              "order": 1
            },
            {
              "id": "gpt-5-mini-low",
              "effort": "low",
              "metrics": {
                "Reasoning": 45.899,
                "Coding": 69.5495,
                "Agentic Coding": 36.6667,
                "Mathematics": 63.2392,
                "Data Analysis": 44.9873,
                "Language": 60.4123,
                "IF": 50.7125,
                "overall": 53.0667,
                "balanced": 47.4842,
                "coding": 49.3209,
                "research": 48.7652,
                "writing": 53.8704
              },
              "order": 2
            },
            {
              "id": "gpt-5-mini-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 68.322,
                "Coding": 68.2025,
                "Agentic Coding": 46.6667,
                "Mathematics": 82.2005,
                "Data Analysis": 55.195,
                "Language": 75.5207,
                "IF": 65.271,
                "overall": 65.9112,
                "balanced": 61.2472,
                "coding": 58.2362,
                "research": 64.8535,
                "writing": 69.8285
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-nano",
      "name": "GPT-5 nano",
      "provider": "OpenAI",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Explicit effort rows",
          "scope": "Explicit effort rows",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5-nano-low",
              "effort": "low",
              "metrics": {
                "Reasoning": 27.678,
                "Coding": 52.725,
                "Agentic Coding": 10.0,
                "Mathematics": 48.907,
                "Data Analysis": 36.5663,
                "Language": 35.4013,
                "IF": 29.0957,
                "overall": 34.3391,
                "balanced": 28.3348,
                "coding": 27.3788,
                "research": 31.7866,
                "writing": 31.4053
              },
              "order": 2
            },
            {
              "id": "gpt-5-nano-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 40.2885,
                "Coding": 62.3855,
                "Agentic Coding": 23.3333,
                "Mathematics": 68.4062,
                "Data Analysis": 43.4057,
                "Language": 46.8417,
                "IF": 55.6998,
                "overall": 48.623,
                "balanced": 43.4276,
                "coding": 40.8289,
                "research": 45.2889,
                "writing": 49.8448
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-1",
      "name": "GPT-5.1",
      "provider": "OpenAI",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 2025-11-13 version",
          "scope": "2025-11-13 version",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5.1-2025-11-13-low",
              "effort": "low",
              "metrics": {
                "Reasoning": 59.6442,
                "Coding": 80.297,
                "Agentic Coding": 40.0,
                "Mathematics": 68.3783,
                "Data Analysis": 45.8673,
                "Language": 77.8517,
                "IF": 47.6247,
                "overall": 59.9519,
                "balanced": 52.7093,
                "coding": 55.7982,
                "research": 55.8384,
                "writing": 61.5184
              },
              "order": 2
            },
            {
              "id": "gpt-5.1-2025-11-13-medium",
              "effort": "medium",
              "metrics": {
                "Reasoning": 73.9808,
                "Coding": 75.689,
                "Agentic Coding": 53.3333,
                "Mathematics": 79.0105,
                "Data Analysis": 63.2237,
                "Language": 78.657,
                "IF": 60.3042,
                "overall": 69.1712,
                "balanced": 64.9894,
                "coding": 63.8342,
                "research": 68.7198,
                "writing": 69.6968
              },
              "order": 3
            },
            {
              "id": "gpt-5.1-2025-11-13-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 78.7933,
                "Coding": 72.489,
                "Agentic Coding": 53.3333,
                "Mathematics": 86.8995,
                "Data Analysis": 69.6063,
                "Language": 79.2603,
                "IF": 63.904,
                "overall": 72.0408,
                "balanced": 67.9705,
                "coding": 63.9561,
                "research": 73.1294,
                "writing": 72.2799
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-2",
      "name": "GPT-5.2",
      "provider": "OpenAI",
      "profileSlug": "gpt-5-2",
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 2025-12-11 version",
          "scope": "2025-12-11 version",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5.2-2025-12-11-low",
              "effort": "low",
              "metrics": {
                "Reasoning": 71.2308,
                "Coding": 73.8975,
                "Agentic Coding": 50.0,
                "Mathematics": 84.7142,
                "Data Analysis": 52.7383,
                "Language": 70.1807,
                "IF": 54.5457,
                "overall": 65.3296,
                "balanced": 60.3062,
                "coding": 60.8084,
                "research": 62.1885,
                "writing": 63.3025
              },
              "order": 2
            },
            {
              "id": "gpt-5.2-2025-12-11-medium",
              "effort": "medium",
              "metrics": {
                "Reasoning": 84.173,
                "Coding": 72.1065,
                "Agentic Coding": 51.6667,
                "Mathematics": 92.069,
                "Data Analysis": 70.3693,
                "Language": 74.9433,
                "IF": 57.5252,
                "overall": 71.8362,
                "balanced": 67.7326,
                "coding": 63.2604,
                "research": 73.3179,
                "writing": 68.4896
              },
              "order": 3
            },
            {
              "id": "gpt-5.2-2025-12-11-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 83.2115,
                "Coding": 76.0715,
                "Agentic Coding": 51.6667,
                "Mathematics": 93.166,
                "Data Analysis": 78.1633,
                "Language": 79.809,
                "IF": 61.7707,
                "overall": 74.837,
                "balanced": 70.0711,
                "coding": 64.7302,
                "research": 76.8985,
                "writing": 72.2022
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-3-codex",
      "name": "GPT-5.3 Codex",
      "provider": "OpenAI",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Coding model",
          "scope": "Coding model",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5.3-codex-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 80.1538,
                "Coding": 78.1845,
                "Agentic Coding": 55.0,
                "Mathematics": 87.8358,
                "Data Analysis": 62.688,
                "Language": 80.0867,
                "IF": 65.375,
                "overall": 72.7605,
                "balanced": 68.6115,
                "coding": 66.7659,
                "research": 71.9482,
                "writing": 73.4765
              },
              "order": 4
            },
            {
              "id": "gpt-5.3-codex-xhigh",
              "effort": "xhigh",
              "metrics": {
                "Reasoning": 71.423,
                "Coding": 77.48,
                "Agentic Coding": 66.6667,
                "Mathematics": 85.6995,
                "Data Analysis": 49.679,
                "Language": 79.18,
                "IF": 71.3415,
                "overall": 71.6385,
                "balanced": 67.7955,
                "coding": 71.0916,
                "research": 66.047,
                "writing": 74.4891
              },
              "order": 5
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-4",
      "name": "GPT-5.4",
      "provider": "OpenAI",
      "profileSlug": "gpt-5-4",
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Explicit effort rows",
          "scope": "Explicit effort rows",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5.4-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 85.6538,
                "Coding": 78.1845,
                "Agentic Coding": 46.6667,
                "Mathematics": 89.9813,
                "Data Analysis": 77.0483,
                "Language": 83.009,
                "IF": 64.9543,
                "overall": 75.0711,
                "balanced": 70.6437,
                "coding": 63.7988,
                "research": 78.5355,
                "writing": 75.2811
              },
              "order": 4
            },
            {
              "id": "gpt-5.4-xhigh",
              "effort": "xhigh",
              "metrics": {
                "Reasoning": 88.1155,
                "Coding": 77.5415,
                "Agentic Coding": 70.0,
                "Mathematics": 94.148,
                "Data Analysis": 79.3133,
                "Language": 82.6337,
                "IF": 70.2168,
                "overall": 80.2812,
                "balanced": 77.64,
                "coding": 75.0015,
                "research": 81.0728,
                "writing": 77.8683
              },
              "order": 5
            }
          ]
        },
        {
          "id": "ockbench-200",
          "source": "ockbench",
          "name": "Math, code & science \u00b7 OckBench",
          "scope": "200 selected problems",
          "dataUrl": "https://ockbench.github.io/static/data/top200_model_performance.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall accuracy",
              "description": "Correct answers out of 200 problems. Domain proportions are 100 math, 60 coding and 40 science."
            },
            {
              "key": "Math",
              "label": "Mathematics",
              "description": "Accuracy on 100 selected mathematics problems, graded with an LLM judge."
            },
            {
              "key": "Coding",
              "label": "Coding tests",
              "description": "Accuracy on 60 selected programming problems checked with executable tests."
            },
            {
              "key": "Science",
              "label": "Science",
              "description": "Accuracy on 40 selected science problems with checked multiple-choice answers."
            }
          ],
          "rows": [
            {
              "id": "gpt-5.4",
              "effort": "none",
              "metrics": {
                "overall": 23.5,
                "Math": 17.0,
                "Coding": 20.0,
                "Science": 45.0
              },
              "outputTokens": 689.0,
              "domainTokens": {
                "Math": 1209.0,
                "Coding": 122.0,
                "Science": 241.0
              },
              "sampleSize": 200,
              "order": 0
            },
            {
              "id": "gpt-5.4-low",
              "effort": "low",
              "metrics": {
                "overall": 78.0,
                "Math": 73.0,
                "Coding": 91.7,
                "Science": 70.0
              },
              "outputTokens": 3056.0,
              "domainTokens": {
                "Math": 4646.0,
                "Coding": 1181.0,
                "Science": 1891.0
              },
              "sampleSize": 200,
              "order": 2
            },
            {
              "id": "gpt-5.4-medium",
              "effort": "medium",
              "metrics": {
                "overall": 82.0,
                "Math": 78.0,
                "Coding": 93.3,
                "Science": 75.0
              },
              "outputTokens": 3177.0,
              "domainTokens": {
                "Math": 3235.0,
                "Coding": 2338.0,
                "Science": 4308.0
              },
              "sampleSize": 200,
              "order": 3
            },
            {
              "id": "gpt-5.4-high",
              "effort": "high",
              "metrics": {
                "overall": 80.5,
                "Math": 74.0,
                "Coding": 98.3,
                "Science": 70.0
              },
              "outputTokens": 3267.0,
              "domainTokens": {
                "Math": 659.0,
                "Coding": 3877.0,
                "Science": 7698.0
              },
              "sampleSize": 200,
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-4-mini",
      "name": "GPT-5.4 mini",
      "provider": "OpenAI",
      "profileSlug": "gpt-5-4-mini",
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Explicit effort rows",
          "scope": "Explicit effort rows",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5.4-mini-low",
              "effort": "low",
              "metrics": {
                "Reasoning": 40.343,
                "Coding": 72.906,
                "Agentic Coding": 23.5963,
                "Mathematics": 64.0895,
                "Data Analysis": 51.88,
                "Language": 57.5937,
                "IF": 36.3687,
                "overall": 49.5396,
                "balanced": 40.987,
                "coding": 42.1785,
                "research": 45.5969,
                "writing": 45.4549
              },
              "order": 2
            },
            {
              "id": "gpt-5.4-mini-medium",
              "effort": "medium",
              "metrics": {
                "Reasoning": 62.045,
                "Coding": 71.453,
                "Agentic Coding": 27.0177,
                "Mathematics": 70.4025,
                "Data Analysis": 64.293,
                "Language": 62.3703,
                "IF": 50.755,
                "overall": 58.3338,
                "balanced": 53.495,
                "coding": 47.9761,
                "research": 60.5102,
                "writing": 57.0946
              },
              "order": 3
            },
            {
              "id": "gpt-5.4-mini-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 69.6697,
                "Coding": 70.94,
                "Agentic Coding": 38.9473,
                "Mathematics": 74.0985,
                "Data Analysis": 69.267,
                "Language": 65.7587,
                "IF": 56.2858,
                "overall": 63.5667,
                "balanced": 60.2459,
                "coding": 54.8873,
                "research": 66.2855,
                "writing": 62.0825
              },
              "order": 4
            },
            {
              "id": "gpt-5.4-mini-xhigh",
              "effort": "xhigh",
              "metrics": {
                "Reasoning": 72.4967,
                "Coding": 71.624,
                "Agentic Coding": 47.456,
                "Mathematics": 78.5555,
                "Data Analysis": 70.945,
                "Language": 71.462,
                "IF": 60.2653,
                "overall": 67.5435,
                "balanced": 64.1107,
                "coding": 59.7434,
                "research": 69.4297,
                "writing": 66.5787
              },
              "order": 5
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-4-nano",
      "name": "GPT-5.4 nano",
      "provider": "OpenAI",
      "profileSlug": "gpt-5-4-nano",
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Explicit effort rows",
          "scope": "Explicit effort rows",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5.4-nano-low",
              "effort": "low",
              "metrics": {
                "Reasoning": 42.1283,
                "Coding": 69.4015,
                "Agentic Coding": 37.2807,
                "Mathematics": 65.156,
                "Data Analysis": 43.968,
                "Language": 45.5587,
                "IF": 37.1752,
                "overall": 48.6669,
                "balanced": 42.9238,
                "coding": 47.6335,
                "research": 42.2041,
                "writing": 41.2716
              },
              "order": 2
            },
            {
              "id": "gpt-5.4-nano-medium",
              "effort": "medium",
              "metrics": {
                "Reasoning": 64.4903,
                "Coding": 70.2565,
                "Agentic Coding": 40.7017,
                "Mathematics": 83.0957,
                "Data Analysis": 48.716,
                "Language": 51.0373,
                "IF": 50.945,
                "overall": 58.4632,
                "balanced": 54.5567,
                "coding": 54.1607,
                "research": 55.031,
                "writing": 53.0137
              },
              "order": 3
            },
            {
              "id": "gpt-5.4-nano-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 72.0032,
                "Coding": 68.376,
                "Agentic Coding": 49.1227,
                "Mathematics": 88.5935,
                "Data Analysis": 52.6583,
                "Language": 54.7507,
                "IF": 53.7448,
                "overall": 62.7499,
                "balanced": 59.598,
                "coding": 58.793,
                "research": 59.9602,
                "writing": 56.8859
              },
              "order": 4
            },
            {
              "id": "gpt-5.4-nano-xhigh",
              "effort": "xhigh",
              "metrics": {
                "Reasoning": 81.0513,
                "Coding": 72.137,
                "Agentic Coding": 49.1227,
                "Mathematics": 91.2703,
                "Data Analysis": 67.6427,
                "Language": 62.468,
                "IF": 67.2048,
                "overall": 70.1281,
                "balanced": 68.3012,
                "coding": 62.6245,
                "research": 71.4719,
                "writing": 67.387
              },
              "order": 5
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-5",
      "name": "GPT-5.5",
      "provider": "OpenAI",
      "profileSlug": "gpt-5-5",
      "studies": [
        {
          "id": "livebench-2026_01_08-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2026-01-08 \u00b7 Explicit effort rows",
          "scope": "Explicit effort rows",
          "questionSet": "2026-01-08",
          "dataUrl": "https://livebench.ai/table_2026_01_08.csv",
          "defaultMetric": "balanced",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 7 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "balanced",
              "label": "General agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "coding",
              "label": "Coding agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Agentic Coding",
              "label": "Agentic Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "gpt-5.5-medium",
              "effort": "medium",
              "metrics": {
                "Reasoning": 87.274,
                "Coding": 78.6285,
                "Agentic Coding": 16.6667,
                "Mathematics": 69.7695,
                "Data Analysis": 76.987,
                "Language": 85.5703,
                "IF": 65.7375,
                "overall": 68.6619,
                "balanced": 65.3608,
                "coding": 50.7534,
                "research": 79.6251,
                "writing": 76.9011
              },
              "order": 3
            },
            {
              "id": "gpt-5.5-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 86.2115,
                "Coding": 82.5325,
                "Agentic Coding": 30.0,
                "Mathematics": 96.0692,
                "Data Analysis": 79.4787,
                "Language": 87.4273,
                "IF": 71.9417,
                "overall": 76.2373,
                "balanced": 70.0239,
                "coding": 58.3857,
                "research": 81.5201,
                "writing": 80.2764
              },
              "order": 4
            },
            {
              "id": "gpt-5.5-xhigh",
              "effort": "xhigh",
              "metrics": {
                "Reasoning": 87.7115,
                "Coding": 82.471,
                "Agentic Coding": 56.6667,
                "Mathematics": 96.3203,
                "Data Analysis": 81.0803,
                "Language": 87.659,
                "IF": 73.0418,
                "overall": 80.7072,
                "balanced": 76.3164,
                "coding": 70.7022,
                "research": 82.7803,
                "writing": 81.0891
              },
              "order": 5
            }
          ]
        },
        {
          "id": "ockbench-200",
          "source": "ockbench",
          "name": "Math, code & science \u00b7 OckBench",
          "scope": "200 selected problems",
          "dataUrl": "https://ockbench.github.io/static/data/top200_model_performance.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall accuracy",
              "description": "Correct answers out of 200 problems. Domain proportions are 100 math, 60 coding and 40 science."
            },
            {
              "key": "Math",
              "label": "Mathematics",
              "description": "Accuracy on 100 selected mathematics problems, graded with an LLM judge."
            },
            {
              "key": "Coding",
              "label": "Coding tests",
              "description": "Accuracy on 60 selected programming problems checked with executable tests."
            },
            {
              "key": "Science",
              "label": "Science",
              "description": "Accuracy on 40 selected science problems with checked multiple-choice answers."
            }
          ],
          "rows": [
            {
              "id": "gpt-5.5",
              "effort": "none",
              "metrics": {
                "overall": 26.0,
                "Math": 15.0,
                "Coding": 38.3,
                "Science": 35.0
              },
              "outputTokens": 260.0,
              "domainTokens": {
                "Math": 431.0,
                "Coding": 141.0,
                "Science": 9.0
              },
              "sampleSize": 200,
              "order": 0
            },
            {
              "id": "gpt-5.5-low",
              "effort": "low",
              "metrics": {
                "overall": 75.0,
                "Math": 65.0,
                "Coding": 95.0,
                "Science": 70.0
              },
              "outputTokens": 1603.0,
              "domainTokens": {
                "Math": 2255.0,
                "Coding": 738.0,
                "Science": 1271.0
              },
              "sampleSize": 200,
              "order": 2
            },
            {
              "id": "gpt-5.5-medium",
              "effort": "medium",
              "metrics": {
                "overall": 86.0,
                "Math": 80.0,
                "Coding": 96.7,
                "Science": 85.0
              },
              "outputTokens": 4692.0,
              "domainTokens": {
                "Math": 6723.0,
                "Coding": 1739.0,
                "Science": 4043.0
              },
              "sampleSize": 200,
              "order": 3
            },
            {
              "id": "gpt-5.5-high",
              "effort": "high",
              "metrics": {
                "overall": 88.5,
                "Math": 82.0,
                "Coding": 100.0,
                "Science": 87.5
              },
              "outputTokens": 10018.0,
              "domainTokens": {
                "Math": 14062.0,
                "Coding": 2853.0,
                "Science": 10655.0
              },
              "sampleSize": 200,
              "order": 4
            },
            {
              "id": "gpt-5.5-xhigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 90.0,
                "Math": 85.0,
                "Coding": 100.0,
                "Science": 87.5
              },
              "outputTokens": 13271.0,
              "domainTokens": {
                "Math": 19115.0,
                "Coding": 4501.0,
                "Science": 11961.0
              },
              "sampleSize": 200,
              "order": 5
            }
          ]
        },
        {
          "id": "stet-graphql-26",
          "source": "stet",
          "name": "Repository coding \u00b7 Stet exploratory study",
          "scope": "26 GraphQL-go-tools tasks \u00b7 Codex 0.128.0",
          "defaultMetric": "tests",
          "exploratory": true,
          "metrics": [
            {
              "key": "tests",
              "label": "Tests passed",
              "description": "Share of 26 tasks whose generated patch passed tests."
            },
            {
              "key": "equivalence",
              "label": "Human-PR equivalence",
              "description": "Share judged to match the intent of the original human change. AI-judged, not a human review score."
            },
            {
              "key": "review",
              "label": "AI review passed",
              "description": "Share accepted by an uncalibrated AI code reviewer."
            },
            {
              "key": "all",
              "label": "All three passed",
              "description": "Share passing tests, human-PR equivalence judging and AI review together."
            }
          ],
          "rows": [
            {
              "id": "GPT-5.5 Codex / low",
              "effort": "low",
              "metrics": {
                "tests": 80.7692,
                "equivalence": 15.3846,
                "review": 11.5385,
                "all": 11.5385
              },
              "cost": 2.65,
              "latency": 294.6,
              "sampleSize": 26,
              "order": 2
            },
            {
              "id": "GPT-5.5 Codex / medium",
              "effort": "medium",
              "metrics": {
                "tests": 80.7692,
                "equivalence": 42.3077,
                "review": 19.2308,
                "all": 19.2308
              },
              "cost": 3.13,
              "latency": 371.8,
              "sampleSize": 26,
              "order": 3
            },
            {
              "id": "GPT-5.5 Codex / high",
              "effort": "high",
              "metrics": {
                "tests": 96.1538,
                "equivalence": 69.2308,
                "review": 38.4615,
                "all": 38.4615
              },
              "cost": 4.49,
              "latency": 572.9,
              "sampleSize": 26,
              "order": 4
            },
            {
              "id": "GPT-5.5 Codex / xhigh",
              "effort": "xhigh",
              "metrics": {
                "tests": 92.3077,
                "equivalence": 88.4615,
                "review": 69.2308,
                "all": 65.3846
              },
              "cost": 9.77,
              "latency": 732.7,
              "sampleSize": 26,
              "order": 5
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-6-luna",
      "name": "GPT-5.6 Luna",
      "provider": "OpenAI",
      "profileSlug": "gpt-5-6-luna",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "GPT-5.6 Luna \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 19.0
              },
              "cost": 0.013141,
              "latency": 61.331,
              "reportedTokens": 2349.41,
              "order": 2
            },
            {
              "id": "GPT-5.6 Luna \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 24.0
              },
              "cost": 0.018455,
              "latency": 78.83,
              "reportedTokens": 3838.337,
              "order": 3
            },
            {
              "id": "GPT-5.6 Luna \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 31.3333
              },
              "cost": 0.030664,
              "latency": 118.097,
              "reportedTokens": 7884.22,
              "order": 4
            },
            {
              "id": "GPT-5.6 Luna \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 35.3333
              },
              "cost": 0.039638,
              "latency": 145.33,
              "reportedTokens": 11003.187,
              "order": 5
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-6-sol",
      "name": "GPT-5.6 Sol",
      "provider": "OpenAI",
      "profileSlug": "gpt-5-6-sol",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "GPT-5.6 Sol \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 29.6667
              },
              "cost": 0.369373,
              "latency": 76.207,
              "reportedTokens": 2808.507,
              "order": 2
            },
            {
              "id": "GPT-5.6 Sol \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 39.6667
              },
              "cost": 0.613092,
              "latency": 134.261,
              "reportedTokens": 5560.335,
              "order": 3
            },
            {
              "id": "GPT-5.6 Sol \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 38.0
              },
              "cost": 0.748522,
              "latency": 173.33,
              "reportedTokens": 7967.48,
              "order": 4
            },
            {
              "id": "GPT-5.6 Sol \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 40.3333
              },
              "cost": 0.983077,
              "latency": 210.172,
              "reportedTokens": 11427.47,
              "order": 5
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-5-6-terra",
      "name": "GPT-5.6 Terra",
      "provider": "OpenAI",
      "profileSlug": "gpt-5-6-terra",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "GPT-5.6 Terra \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 27.0
              },
              "cost": 0.143362,
              "latency": 64.548,
              "reportedTokens": 2820.823,
              "order": 2
            },
            {
              "id": "GPT-5.6 Terra \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 29.3333
              },
              "cost": 0.151817,
              "latency": 69.391,
              "reportedTokens": 3165.175,
              "order": 3
            },
            {
              "id": "GPT-5.6 Terra \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 32.6667
              },
              "cost": 0.2037,
              "latency": 95.158,
              "reportedTokens": 5378.26,
              "order": 4
            },
            {
              "id": "GPT-5.6 Terra \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 39.3333
              },
              "cost": 0.2972,
              "latency": 123.851,
              "reportedTokens": 9486.007,
              "order": 5
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-6-astra",
      "name": "GPT-6 Astra",
      "provider": "OpenAI",
      "profileSlug": "gpt-6-astra",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "GPT-6 Astra \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 55.0
              },
              "cost": 0.854904,
              "latency": 88.55,
              "reportedTokens": 2737.39,
              "order": 2
            },
            {
              "id": "GPT-6 Astra \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 52.6667
              },
              "cost": 1.236406,
              "latency": 128.25,
              "reportedTokens": 4855.013,
              "order": 3
            },
            {
              "id": "GPT-6 Astra \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 52.0
              },
              "cost": 1.615272,
              "latency": 186.527,
              "reportedTokens": 7695.993,
              "order": 4
            },
            {
              "id": "GPT-6 Astra \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 52.3333
              },
              "cost": 1.750837,
              "latency": 224.332,
              "reportedTokens": 9293.433,
              "order": 5
            },
            {
              "id": "GPT-6 Astra \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 55.0
              },
              "cost": 2.114002,
              "latency": 313.865,
              "reportedTokens": 14828.06,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-6-luna",
      "name": "GPT-6 Luna",
      "provider": "OpenAI",
      "profileSlug": "gpt-6-luna",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "GPT-6 Luna \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 30.3333
              },
              "cost": 0.0107957625,
              "latency": 63.516,
              "reportedTokens": 2619.027,
              "order": 2
            },
            {
              "id": "GPT-6 Luna \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 42.6667
              },
              "cost": 0.019124674467,
              "latency": 116.566,
              "reportedTokens": 9117.587,
              "order": 3
            },
            {
              "id": "GPT-6 Luna \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 46.0
              },
              "cost": 0.025823215225,
              "latency": 175.446,
              "reportedTokens": 15545.325,
              "order": 4
            },
            {
              "id": "GPT-6 Luna \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 46.0
              },
              "cost": 0.03956597605,
              "latency": 188.733,
              "reportedTokens": 19321.703,
              "order": 5
            },
            {
              "id": "GPT-6 Luna \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 52.3333
              },
              "cost": 0.057961059717,
              "latency": 353.172,
              "reportedTokens": 37928.297,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "gpt-6-sol",
      "name": "GPT-6 Sol",
      "provider": "OpenAI",
      "profileSlug": "gpt-6-sol",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "GPT-6 Sol \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 50.3333
              },
              "cost": 0.174839599833,
              "latency": 97.585,
              "reportedTokens": 2752.157,
              "order": 2
            },
            {
              "id": "GPT-6 Sol \u00b7 Medium",
              "effort": "medium",
              "metrics": {
                "overall": 51.3333
              },
              "cost": 0.270769650167,
              "latency": 146.184,
              "reportedTokens": 4774.527,
              "order": 3
            },
            {
              "id": "GPT-6 Sol \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 54.6667
              },
              "cost": 0.387394600167,
              "latency": 238.047,
              "reportedTokens": 8464.567,
              "order": 4
            },
            {
              "id": "GPT-6 Sol \u00b7 XHigh",
              "effort": "xhigh",
              "metrics": {
                "overall": 61.3333
              },
              "cost": 0.581652172833,
              "latency": 326.66,
              "reportedTokens": 14595.657,
              "order": 5
            },
            {
              "id": "GPT-6 Sol \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 57.0
              },
              "cost": 0.984485202833,
              "latency": 617.514,
              "reportedTokens": 29926.53,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "o3",
      "name": "o3",
      "provider": "OpenAI",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2025_04_25-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2025-04-25 \u00b7 2025-04-16 version",
          "scope": "2025-04-16 version",
          "questionSet": "2025-04-25",
          "dataUrl": "https://livebench.ai/table_2025_04_25.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 6 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "o3-2025-04-16-medium",
              "effort": "medium",
              "metrics": {
                "Reasoning": 91.0,
                "Coding": 77.863,
                "Mathematics": 80.6573,
                "Data Analysis": 68.1925,
                "Language": 73.4813,
                "IF": 84.3208,
                "overall": 79.2525,
                "research": 80.1941,
                "writing": 80.9869
              },
              "order": 3
            },
            {
              "id": "o3-2025-04-16-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 93.3333,
                "Coding": 76.7145,
                "Mathematics": 85.0037,
                "Data Analysis": 67.02,
                "Language": 75.996,
                "IF": 86.1748,
                "overall": 80.707,
                "research": 81.407,
                "writing": 83.177
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "o3-mini",
      "name": "o3-mini",
      "provider": "OpenAI",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2025_04_02-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2025-04-02 \u00b7 2025-01-31 version",
          "scope": "2025-01-31 version",
          "questionSet": "2025-04-02",
          "dataUrl": "https://livebench.ai/table_2025_04_02.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 6 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "o3-mini-2025-01-31-low",
              "effort": "low",
              "metrics": {
                "Reasoning": 55.9443,
                "Coding": 50.762,
                "Mathematics": 61.6747,
                "Data Analysis": 62.04,
                "Language": 48.0707,
                "IF": 80.0625,
                "overall": 59.759,
                "research": 61.4156,
                "writing": 63.648
              },
              "order": 2
            },
            {
              "id": "o3-mini-2025-01-31-medium",
              "effort": "medium",
              "metrics": {
                "Reasoning": 69.0,
                "Coding": 58.4285,
                "Mathematics": 71.6753,
                "Data Analysis": 66.56,
                "Language": 54.1173,
                "IF": 83.1582,
                "overall": 67.1566,
                "research": 68.8672,
                "writing": 69.4181
              },
              "order": 3
            },
            {
              "id": "o3-mini-2025-01-31-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 74.361,
                "Coding": 65.4765,
                "Mathematics": 76.5543,
                "Data Analysis": 70.64,
                "Language": 56.8553,
                "IF": 84.3585,
                "overall": 71.3743,
                "research": 72.6183,
                "writing": 71.8576
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "o4-mini",
      "name": "o4-mini",
      "provider": "OpenAI",
      "profileSlug": null,
      "studies": [
        {
          "id": "livebench-2025_04_25-thinking",
          "source": "livebench",
          "name": "General tasks \u00b7 LiveBench 2025-04-25 \u00b7 2025-04-16 version",
          "scope": "2025-04-16 version",
          "questionSet": "2025-04-25",
          "dataUrl": "https://livebench.ai/table_2025_04_25.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall benchmark",
              "description": "Equal-weight average of the 6 LiveBench category scores. Each effort uses the same question set."
            },
            {
              "key": "research",
              "label": "Research agent fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "writing",
              "label": "Writing & support fit",
              "description": "AI Agent Store task-weighted category mix. An editorial selection aid, not a measured agent success rate; weights match our published Agent Fit methodology."
            },
            {
              "key": "Reasoning",
              "label": "Reasoning",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Coding",
              "label": "Coding",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Mathematics",
              "label": "Mathematics",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Data Analysis",
              "label": "Data Analysis",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "Language",
              "label": "Language",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            },
            {
              "key": "IF",
              "label": "Instruction following",
              "description": "Published LiveBench task scores averaged within this category; same question set for all effort levels in this sweep."
            }
          ],
          "rows": [
            {
              "id": "o4-mini-2025-04-16-medium",
              "effort": "medium",
              "metrics": {
                "Reasoning": 78.4723,
                "Coding": 74.219,
                "Mathematics": 81.02,
                "Data Analysis": 68.4715,
                "Language": 62.409,
                "IF": 81.825,
                "overall": 74.4028,
                "research": 73.7331,
                "writing": 73.5557
              },
              "order": 3
            },
            {
              "id": "o4-mini-2025-04-16-high",
              "effort": "high",
              "metrics": {
                "Reasoning": 88.111,
                "Coding": 79.9755,
                "Mathematics": 84.8953,
                "Data Analysis": 68.3275,
                "Language": 66.0547,
                "IF": 84.9582,
                "overall": 78.7204,
                "research": 78.2369,
                "writing": 77.8697
              },
              "order": 4
            }
          ]
        }
      ]
    },
    {
      "slug": "glm-5-2",
      "name": "GLM 5.2",
      "provider": "Z.ai",
      "profileSlug": "glm-5-2",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "GLM 5.2 \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 13.3333
              },
              "cost": 0.253753,
              "latency": 136.744,
              "reportedTokens": 10494.41,
              "order": 4
            },
            {
              "id": "GLM 5.2 \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 9.0
              },
              "cost": 0.320113,
              "latency": 183.974,
              "reportedTokens": 19073.47,
              "order": 6
            }
          ]
        },
        {
          "id": "ockbench-200",
          "source": "ockbench",
          "name": "Math, code & science \u00b7 OckBench",
          "scope": "200 selected problems",
          "dataUrl": "https://ockbench.github.io/static/data/top200_model_performance.csv",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Overall accuracy",
              "description": "Correct answers out of 200 problems. Domain proportions are 100 math, 60 coding and 40 science."
            },
            {
              "key": "Math",
              "label": "Mathematics",
              "description": "Accuracy on 100 selected mathematics problems, graded with an LLM judge."
            },
            {
              "key": "Coding",
              "label": "Coding tests",
              "description": "Accuracy on 60 selected programming problems checked with executable tests."
            },
            {
              "key": "Science",
              "label": "Science",
              "description": "Accuracy on 40 selected science problems with checked multiple-choice answers."
            }
          ],
          "rows": [
            {
              "id": "glm-5.2-off",
              "effort": "off",
              "metrics": {
                "overall": 28.5,
                "Math": 13.0,
                "Coding": 40.0,
                "Science": 50.0
              },
              "outputTokens": 2090.0,
              "domainTokens": {
                "Math": 3009.0,
                "Coding": 1014.0,
                "Science": 1404.0
              },
              "sampleSize": 200,
              "order": 0
            },
            {
              "id": "glm-5.2-high",
              "effort": "high",
              "metrics": {
                "overall": 78.5,
                "Math": 76.0,
                "Coding": 85.0,
                "Science": 75.0
              },
              "outputTokens": 38251.0,
              "domainTokens": {
                "Math": 40941.0,
                "Coding": 27061.0,
                "Science": 47749.0
              },
              "sampleSize": 200,
              "order": 4
            },
            {
              "id": "glm-5.2-max",
              "effort": "max",
              "metrics": {
                "overall": 83.5,
                "Math": 79.0,
                "Coding": 88.3,
                "Science": 87.5
              },
              "outputTokens": 41234.0,
              "domainTokens": {
                "Math": 42965.0,
                "Coding": 34340.0,
                "Science": 46900.0
              },
              "sampleSize": 200,
              "order": 6
            }
          ]
        }
      ]
    },
    {
      "slug": "glm-5-3",
      "name": "GLM 5.3",
      "provider": "Z.ai",
      "profileSlug": "glm-5-3",
      "studies": [
        {
          "id": "databench-v1-1",
          "source": "databench",
          "name": "Analytics \u00b7 DataBench v1.1",
          "scope": "Analytics in the Hex agent harness",
          "defaultMetric": "overall",
          "metrics": [
            {
              "key": "overall",
              "label": "Analytics score",
              "description": "Published blended DataBench score, on a 0\u2013100 scale. Judge-based analytical task performance; not a general intelligence or guaranteed success rate."
            }
          ],
          "rows": [
            {
              "id": "GLM 5.3 \u00b7 Low",
              "effort": "low",
              "metrics": {
                "overall": 8.6667
              },
              "cost": 0.199005,
              "latency": 94.651,
              "reportedTokens": 5341.487,
              "order": 2
            },
            {
              "id": "GLM 5.3 \u00b7 High",
              "effort": "high",
              "metrics": {
                "overall": 14.6667
              },
              "cost": 0.356431,
              "latency": 215.078,
              "reportedTokens": 14874.81,
              "order": 4
            },
            {
              "id": "GLM 5.3 \u00b7 Max",
              "effort": "max",
              "metrics": {
                "overall": 16.3333
              },
              "cost": 0.638527,
              "latency": 518.487,
              "reportedTokens": 38848.457,
              "order": 6
            }
          ]
        }
      ]
    }
  ],
  "sourceHashes": {
    "categories_2024_06_24.json": "08fa233f92ca970a786f4ab8ea09639bce4c9dcbc88565254989e0bd34d08bfa",
    "categories_2024_07_26.json": "31382962c060909421a3f2d8fe9f2194daadc252f654647daa8f684efbbe674b",
    "categories_2024_08_31.json": "31382962c060909421a3f2d8fe9f2194daadc252f654647daa8f684efbbe674b",
    "categories_2024_11_25.json": "31382962c060909421a3f2d8fe9f2194daadc252f654647daa8f684efbbe674b",
    "categories_2025_04_02.json": "182ba3b11478a1f1768020fc79b6817b5698054333d768a8c6c800f65cd90cfa",
    "categories_2025_04_25.json": "742ed9e6946db311c91361781d71d3f6c3587b5ade5daf7373eb32a39d21f5fb",
    "categories_2025_05_30.json": "f79b260f8c39b9763007dda694c2ee3f11b9898e17a85ee1c353c8a0b5a8de22",
    "categories_2025_11_25.json": "4b407579b95ffacb519a33afc07affc78096af029ffd29891bd3bcd18b62994f",
    "categories_2025_12_23.json": "e8137ae3bf1baabec1ae22d4815cb3f8cb5ed4abf8bae63c2261f3fd06fa964a",
    "categories_2026_01_08.json": "dad300ad18655b69db720e1b88fc5a5eac06c5b2f0e52c2bf50f10ff057674f3",
    "categories_2026_06_25.json": "dad300ad18655b69db720e1b88fc5a5eac06c5b2f0e52c2bf50f10ff057674f3",
    "cost_2026_06_25.csv": "66bb9fb719d29ebd296da6f04d90a721d33c4aaf588e5ae223eb3c88b299bd4f",
    "databench-rows.json": "1d2e9b1467a56c8b6b8545bbeb3d8a2a5f0bde7952ec9385b8037328270016b1",
    "livebench-files.json": "e98289a852da3f971125c69d5b08ed0745e8863e0b2ab2bf5313f824a97f3dbc",
    "livebench-models.js": "f96c19bf94e15e6ad2cd03b59e4d7efe176292834820058b44c58c0c1098dfef",
    "livebench-releases.js": "ce56014792362d8c36d3e38d00d61709b849d7c572080c58eef53a4aa748f14d",
    "ock.js": "b86813ac84c3eb4fc702867a773d3a9f7d6d7be3d972619c018890f521f00856",
    "per_domain_performance.csv": "796f18559ad0e7669879ac0115f4950fce9e0e71a758111a13749d7be739b808",
    "table_2024_06_24.csv": "df181166a5a670a5b7678ee02f96c6896d8a88a503aec20243234b698d25a8c0",
    "table_2024_07_26.csv": "670dc4af2497d2daca7b5179f9558854163d954a95939bc4a7d71d8fce298a85",
    "table_2024_08_31.csv": "8c66d248110076378f7eb58aaeef524c65bba6dff0b55f190fd9ac8e92d73b2a",
    "table_2024_11_25.csv": "17ca6b3d122921d24b591d0da52566c5552ed4f4001ae9d83a11b2712325aae4",
    "table_2025_04_02.csv": "d7fd6a3e461cff4f96569093553789115e87adeb9cd95dca043bafa51ff2acaa",
    "table_2025_04_25.csv": "251eb7461da69232733b80d1e63f00a9db5b921d90c966d139ce0e2851c762d9",
    "table_2025_05_30.csv": "c1215f3df5b2e0a5fe69907c101d29922bf783b011df9ba82638c0145cb4c64c",
    "table_2025_11_25.csv": "34242000938912ac33369e3b25d49f6eafd3cc64823914b2405cef9a7ddcd81d",
    "table_2025_12_23.csv": "d8a96fd89213f11f144c92846567f8652bc2740511e8b4bc4211f659ff57f670",
    "table_2026_01_08.csv": "bcb52d78b463adeccdd06899452a0bf68ffa84906bd37f68fdc0659d9f241241",
    "table_2026_06_25.csv": "05f59189a82813c446681b0c5902a559ef9c5d55263a2bb0ae2d1868fb5ecb2d",
    "top200_model_performance.csv": "7c685e6418ab46046cdd1307801b946cd586750d831c501da78b09f95700d718"
  }
}
