{
  "schemaVersion": 1,
  "benchmark": {
    "id": "arc-agi-3-standard",
    "name": "ARC-AGI-3 · Standard",
    "version": "3 · semi-private · Standard",
    "category": "reasoning",
    "question": "Can it learn the rules by interacting?",
    "summary": "Explore an unfamiliar environment and solve it through a shared, minimal agent interface.",
    "source": {
      "name": "ARC Prize",
      "url": "https://arcprize.org/leaderboard",
      "methodologyUrl": "https://arcprize.org/blog/astra"
    },
    "measure": "Action-efficiency score relative to the benchmark's human baseline, expressed as a percentage.",
    "comparisonRule": "Standard harness only. Compare the selected Astra, Sol, and Opus 5 configurations within this view.",
    "limitations": [
      "Cost is the full evaluation spend, not the cost of one task.",
      "These are deterministic puzzle environments; they do not measure open-ended real-world competence.",
      "The Provider Adapter harness is a separate comparison."
    ],
    "coverage": "charted",
    "tags": [
      "interactive reasoning",
      "exploration",
      "planning",
      "ARC",
      "AGI",
      "agent"
    ]
  },
  "dataset": {
    "benchmarkId": "arc-agi-3-standard",
    "version": "3 · semi-private · Standard",
    "score": {
      "label": "Action-efficiency score",
      "unit": "%",
      "direction": "higher",
      "minimum": 0,
      "maximum": 100
    },
    "source": {
      "name": "ARC Prize · selected published observations",
      "url": "https://arcprize.org/media/data/leaderboard/v3.json",
      "retrievedAt": "2026-09-09T00:19:36.880Z",
      "revision": "SHA-256 d743d7a731ba6b9f102d3624d5fe82011630b90da4200ceb34bd56cf2b221e1c"
    },
    "observedAt": "2026-09-04T14:38:06.320Z",
    "evidenceLabel": "Benchmark-owner results · selected cohort",
    "configurationLabel": "Model and reasoning effort",
    "comparabilityNote": "Standard harness only. Dollar values cover the full evaluation. Harness cohorts are shown separately.",
    "costLabel": "Total evaluation cost (USD)",
    "points": [
      {
        "id": "anthropic-claude-opus-5-high",
        "label": "Claude Opus 5 (High)",
        "model": "Claude Opus 5",
        "provider": "Anthropic",
        "effort": "High",
        "score": 30.159999999999997,
        "costUsd": 20657.37,
        "sourceUrl": "https://arcprize.org/results/anthropic-claude-opus-5",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-high",
        "label": "GPT-5.6 Sol (High)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "High",
        "score": 2.15,
        "costUsd": 15176.07,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-low",
        "label": "GPT-5.6 Sol (Low)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "Low",
        "score": 0.33,
        "costUsd": 12805.6,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-max",
        "label": "GPT-5.6 Sol (Max)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "Max",
        "score": 7.779999999999999,
        "costUsd": 25064.11,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-medium",
        "label": "GPT-5.6 Sol (Medium)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "Medium",
        "score": 1.0699999999999998,
        "costUsd": 12971.17,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-xhigh",
        "label": "GPT-5.6 Sol (XHigh)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "XHigh",
        "score": 6.99,
        "costUsd": 19216.38,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-high",
        "label": "GPT-6 Astra (High)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "High",
        "score": 54.81905163612231,
        "costUsd": 40704.722830000006,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-low",
        "label": "GPT-6 Astra (Low)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "Low",
        "score": 17.452190401133805,
        "costUsd": 38166.47030000001,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-max",
        "label": "GPT-6 Astra (Max)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "Max",
        "score": 62.71280210060628,
        "costUsd": 26097.501720000007,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-medium",
        "label": "GPT-6 Astra (Medium)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "Medium",
        "score": 38.58757449733784,
        "costUsd": 48090.34598999999,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-none",
        "label": "GPT-6 Astra (None)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "None",
        "score": 35.177880317424,
        "costUsd": 49791.11317000001,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "Standard harness",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-xhigh",
        "label": "GPT-6 Astra (XHigh)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "XHigh",
        "score": 59.342954091460555,
        "costUsd": 37317.38768,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "Standard harness",
        "uncertainty": null
      }
    ]
  }
}
