{
  "apiVersion": "1",
  "methodologyVersion": "2026-07-24-arc-coverage-stability",
  "updatedAt": "2026-08-29T04:03:36.377Z",
  "benchmarks": [
    {
      "id": "aa-lcr",
      "name": "AA Long Chain Reasoning",
      "description": "",
      "category": "iq",
      "dimension": "reliability",
      "direction": "higher_is_better",
      "unit": null,
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/aa-lcr-scores/"
    },
    {
      "id": "aa-omniscience",
      "name": "AA Omniscience",
      "description": "",
      "category": "iq",
      "dimension": "reliability",
      "direction": "higher_is_better",
      "unit": null,
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/aa-omniscience-scores/"
    },
    {
      "id": "agent-arena",
      "name": "Arena.ai Agent Arena",
      "description": "",
      "category": "iq",
      "dimension": "computer-use",
      "direction": "higher_is_better",
      "unit": null,
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/agent-arena-scores/"
    },
    {
      "id": "agents-last-exam",
      "name": "Agents' Last Exam",
      "description": "",
      "category": "iq",
      "dimension": "computer-use",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/agents-last-exam-scores/"
    },
    {
      "id": "aime",
      "name": "AIME",
      "description": "",
      "category": "iq",
      "dimension": "mathematical-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/aime-scores/"
    },
    {
      "id": "arc-agi-1",
      "name": "ARC-AGI-1",
      "description": "",
      "category": "iq",
      "dimension": "abstract-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/arc-agi-1-scores/"
    },
    {
      "id": "arc-agi-2",
      "name": "ARC-AGI-2",
      "description": "",
      "category": "iq",
      "dimension": "abstract-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/arc-agi-2-scores/"
    },
    {
      "id": "arc-agi-3",
      "name": "ARC-AGI-3",
      "description": "",
      "category": "iq",
      "dimension": "abstract-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/arc-agi-3-scores/"
    },
    {
      "id": "browsecomp",
      "name": "BrowseComp",
      "description": "",
      "category": "iq",
      "dimension": "computer-use",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/browsecomp-scores/"
    },
    {
      "id": "bullshitbench-v2",
      "name": "BullshitBench v2",
      "description": "",
      "category": "iq",
      "dimension": "reliability",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/bullshitbench-v2-scores/"
    },
    {
      "id": "critpt",
      "name": "CritPt",
      "description": "",
      "category": "iq",
      "dimension": "academic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/critpt-scores/"
    },
    {
      "id": "facts-grounding",
      "name": "FACTS Grounding",
      "description": "",
      "category": "iq",
      "dimension": "reliability",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/facts-grounding-scores/"
    },
    {
      "id": "frontiermath-t13",
      "name": "FrontierMath Tier 1-3",
      "description": "",
      "category": "iq",
      "dimension": "mathematical-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/frontiermath-tier-1-3-scores/"
    },
    {
      "id": "frontiermath-t4",
      "name": "FrontierMath Tier 4",
      "description": "",
      "category": "iq",
      "dimension": "mathematical-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/frontiermath-tier-4-scores/"
    },
    {
      "id": "frontierswe",
      "name": "FrontierSWE",
      "description": "",
      "category": "iq",
      "dimension": "programmatic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/frontierswe-scores/"
    },
    {
      "id": "gpqa-diamond",
      "name": "GPQA Diamond",
      "description": "",
      "category": "iq",
      "dimension": "academic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/gpqa-diamond-scores/"
    },
    {
      "id": "humanitys-last-exam",
      "name": "Humanity's Last Exam",
      "description": "",
      "category": "iq",
      "dimension": "academic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/humanitys-last-exam-scores/"
    },
    {
      "id": "ifbench",
      "name": "IFBench",
      "description": "",
      "category": "iq",
      "dimension": "reliability",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/ifbench-scores/"
    },
    {
      "id": "ioi",
      "name": "IOI",
      "description": "",
      "category": "iq",
      "dimension": "programmatic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/ioi-scores/"
    },
    {
      "id": "livecodebench",
      "name": "LiveCodeBench",
      "description": "",
      "category": "iq",
      "dimension": "programmatic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/livecodebench-scores/"
    },
    {
      "id": "matharena",
      "name": "MathArena",
      "description": "",
      "category": "iq",
      "dimension": "mathematical-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/matharena-scores/"
    },
    {
      "id": "mcp-atlas",
      "name": "MCP Atlas",
      "description": "",
      "category": "iq",
      "dimension": "computer-use",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/mcp-atlas-scores/"
    },
    {
      "id": "mmlu-pro",
      "name": "MMLU-Pro",
      "description": "",
      "category": "iq",
      "dimension": "academic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/mmlu-pro-scores/"
    },
    {
      "id": "mmmu-pro",
      "name": "MMMU-Pro",
      "description": "",
      "category": "iq",
      "dimension": "academic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/mmmu-pro-scores/"
    },
    {
      "id": "multichallenge",
      "name": "MultiChallenge",
      "description": "",
      "category": "iq",
      "dimension": "reliability",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/multichallenge-scores/"
    },
    {
      "id": "osworld-verified",
      "name": "OSWorld-Verified",
      "description": "",
      "category": "iq",
      "dimension": "computer-use",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/osworld-verified-scores/"
    },
    {
      "id": "programbench-almost-resolved",
      "name": "ProgramBench",
      "description": "",
      "category": "iq",
      "dimension": "programmatic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/programbench-almost-resolved-scores/"
    },
    {
      "id": "proofbench",
      "name": "ProofBench",
      "description": "",
      "category": "iq",
      "dimension": "mathematical-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/proofbench-scores/"
    },
    {
      "id": "scicode",
      "name": "SciCode",
      "description": "",
      "category": "iq",
      "dimension": "academic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/scicode-scores/"
    },
    {
      "id": "simpleqa-verified",
      "name": "SimpleQA Verified",
      "description": "",
      "category": "iq",
      "dimension": "reliability",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/simpleqa-verified-scores/"
    },
    {
      "id": "swe-rebench",
      "name": "SWE-rebench",
      "description": "",
      "category": "iq",
      "dimension": "programmatic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/swe-rebench-scores/"
    },
    {
      "id": "terminal-bench-2-1",
      "name": "Terminal-Bench 2.1",
      "description": "",
      "category": "iq",
      "dimension": "programmatic-reasoning",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/terminal-bench-2-1-scores/"
    },
    {
      "id": "toolathlon",
      "name": "Toolathlon",
      "description": "",
      "category": "iq",
      "dimension": "computer-use",
      "direction": "higher_is_better",
      "unit": "percent",
      "updatedAt": "2026-08-29T04:03:36.377Z",
      "url": "https://www.aiiq.org/charts/toolathlon-scores/"
    }
  ]
}
