{
  "apiVersion": "1",
  "version": "2026-07-24-arc-coverage-stability",
  "summary": "Composite IQ uses six equally weighted capability dimensions mapped through IQ 70-160 expected-score ladders and conservative imputation rules. Abstract Reasoning equally averages the available source-backed ARC projections; coverage count communicates confidence, while zero-coverage models retain conservative fallback estimates. Emotional Reasoning (EQ) is a separate diagnostic domain excluded from Composite IQ.",
  "derivedRankings": [
    {
      "id": "composite-iq",
      "rankingName": "Composite IQ",
      "direction": "higher_is_better",
      "scoreField": "iq",
      "dimensions": [
        {
          "id": "D1",
          "slug": "abstract-reasoning",
          "name": "Abstract Reasoning",
          "minBenchmarks": 1,
          "benchmarks": [
            {
              "field": "arcAgi3",
              "name": "ARC-AGI-3"
            },
            {
              "field": "arcAgi2",
              "name": "ARC-AGI-2"
            },
            {
              "field": "arcAgi1",
              "name": "ARC-AGI-1"
            }
          ]
        },
        {
          "id": "D2",
          "slug": "mathematical-reasoning",
          "name": "Mathematical Reasoning",
          "minBenchmarks": 1,
          "benchmarks": [
            {
              "field": "fmT4Acc",
              "name": "FrontierMath Tier 4"
            },
            {
              "field": "fmT13Acc",
              "name": "FrontierMath Tier 1-3"
            },
            {
              "field": "proofbench",
              "name": "ProofBench"
            },
            {
              "field": "mathArena",
              "name": "MathArena"
            },
            {
              "field": "aime",
              "name": "AIME"
            }
          ]
        },
        {
          "id": "D3",
          "slug": "academic-reasoning",
          "name": "Academic Reasoning",
          "minBenchmarks": 1,
          "benchmarks": [
            {
              "field": "hle",
              "name": "Humanity's Last Exam"
            },
            {
              "field": "gpqa",
              "name": "GPQA Diamond"
            },
            {
              "field": "critPt",
              "name": "CritPt"
            },
            {
              "field": "sciCode",
              "name": "SciCode"
            },
            {
              "field": "mmlupro",
              "name": "MMLU-Pro"
            },
            {
              "field": "mmmuPro",
              "name": "MMMU-Pro"
            }
          ]
        },
        {
          "id": "D4",
          "slug": "programmatic-reasoning",
          "name": "Programmatic Reasoning",
          "minBenchmarks": 1,
          "benchmarks": [
            {
              "field": "livecodebench",
              "name": "LiveCodeBench"
            },
            {
              "field": "ioi",
              "name": "IOI"
            },
            {
              "field": "terminalbench21",
              "name": "Terminal-Bench 2.1"
            },
            {
              "field": "programBenchAlmostResolved",
              "name": "ProgramBench"
            },
            {
              "field": "frontierSWE",
              "name": "FrontierSWE"
            },
            {
              "field": "sweRebench",
              "name": "SWE-rebench"
            }
          ]
        },
        {
          "id": "D5",
          "slug": "computer-use",
          "name": "Computer Use",
          "minBenchmarks": 1,
          "benchmarks": [
            {
              "field": "browseComp",
              "name": "BrowseComp"
            },
            {
              "field": "osworldVerified",
              "name": "OSWorld-Verified"
            },
            {
              "field": "toolathlon",
              "name": "Toolathlon"
            },
            {
              "field": "mcpAtlas",
              "name": "MCP Atlas"
            },
            {
              "field": "agentArena",
              "name": "Arena.ai Agent Arena"
            },
            {
              "field": "agentsLastExam",
              "name": "Agents' Last Exam"
            }
          ]
        },
        {
          "id": "D6",
          "slug": "reliability",
          "name": "Reliability",
          "minBenchmarks": 1,
          "benchmarks": [
            {
              "field": "simpleQAVerified",
              "name": "SimpleQA Verified"
            },
            {
              "field": "aaOmniscience",
              "name": "aaOmniscience"
            },
            {
              "field": "bullshitBench",
              "name": "BullshitBench v2"
            },
            {
              "field": "ifBench",
              "name": "IFBench"
            },
            {
              "field": "multiChallenge",
              "name": "MultiChallenge"
            },
            {
              "field": "aaLCR",
              "name": "AA Long Chain Reasoning"
            },
            {
              "field": "factsGrounding",
              "name": "FACTS Grounding"
            }
          ]
        }
      ]
    },
    {
      "id": "effective-cost",
      "rankingName": "Effective Cost",
      "direction": "lower_is_better",
      "scoreField": "effectiveCost",
      "unit": "USD per 1M I/O Tokens",
      "breakdown": [
        {
          "id": "published-pricing",
          "name": "Published Token Pricing",
          "inputs": [
            {
              "field": "inP",
              "name": "Input token price"
            },
            {
              "field": "outP",
              "name": "Output token price"
            },
            {
              "field": "cacheReadP",
              "name": "Cache read token price"
            },
            {
              "field": "cacheWriteP",
              "name": "Cache write token price"
            },
            {
              "field": "cacheWrite1hP",
              "name": "1-hour cache write token price"
            },
            {
              "field": "cacheStorageHourlyP",
              "name": "Cache storage hourly token price"
            }
          ],
          "summary": "Base effective cost still uses input price plus output price for 1M input tokens and 1M output tokens. Cache read/write/storage fields are exposed separately for cache-aware task-cost views and provider-specific cost decomposition."
        },
        {
          "id": "observed-token-usage",
          "name": "Observed Token Usage",
          "inputs": [
            {
              "field": "aaTokensM",
              "name": "Artificial Analysis token usage"
            },
            {
              "field": "aaInputTokensM",
              "name": "Artificial Analysis input token usage"
            },
            {
              "field": "aaOutputTokensM",
              "name": "Artificial Analysis output token usage"
            }
          ],
          "summary": "Validated token usage calibrates the median workload multiplier when available."
        },
        {
          "id": "task-cost-residuals",
          "name": "Task Cost Residuals",
          "benchmarks": [
            {
              "field": "arcCostPerTask",
              "name": "ARC-AGI cost per task"
            },
            {
              "field": "valsCostPerTest",
              "name": "VALS cost per test"
            },
            {
              "field": "swebenchCostPerTask",
              "name": "SWE-Bench cost per task"
            },
            {
              "field": "hleCostPerQuestion",
              "name": "Humanity's Last Exam cost per question"
            }
          ],
          "summary": "Task-level cost benchmarks adjust for observed cost differences not explained by published token prices alone."
        },
        {
          "id": "usage-multiplier-waterfall",
          "name": "Usage Multiplier Waterfall",
          "steps": [
            "measured benchmark multiplier from validated token usage and price-adjusted task-cost residuals",
            "one-generation-back same-family same-lineage multiplier",
            "two-generations-back same-family same-lineage multiplier",
            "geometric average of the three closest measured peers",
            "assumed 1x fallback for positive-price models"
          ],
          "summary": "Effective cost is sticker price multiplied by the first available usage multiplier in this waterfall."
        }
      ]
    }
  ],
  "updatedAt": "2026-08-29T04:03:36.377Z",
  "url": "https://www.aiiq.org/methodology/"
}
