{
  "schemaVersion": 1,
  "benchmark": {
    "id": "arc-agi-2",
    "name": "ARC-AGI-2",
    "version": "2 · semi-private",
    "category": "reasoning",
    "question": "Can it solve an unfamiliar visual puzzle?",
    "summary": "Infer a rule from a few examples, then apply it to a new grid.",
    "source": {
      "name": "ARC Prize",
      "url": "https://arcprize.org/leaderboard",
      "methodologyUrl": "https://arcprize.org/blog/astra"
    },
    "measure": "Percentage of novel abstract grid tasks solved in ARC Prize's semi-private evaluation.",
    "comparisonRule": "Compare the same dataset split and reasoning effort. This view selects seven recent model families.",
    "limitations": [
      "Strong puzzle performance does not establish general intelligence.",
      "Selected configurations have no published cost in this source.",
      "Scores near the ceiling leave less room to distinguish leading systems."
    ],
    "coverage": "charted",
    "tags": [
      "abstract reasoning",
      "novel problems",
      "visual puzzles",
      "ARC",
      "AGI"
    ]
  },
  "dataset": {
    "benchmarkId": "arc-agi-2",
    "version": "2 · semi-private",
    "score": {
      "label": "Tasks solved",
      "unit": "%",
      "direction": "higher",
      "minimum": 0,
      "maximum": 100
    },
    "source": {
      "name": "ARC Prize · selected published observations",
      "url": "https://arcprize.org/media/data/leaderboard/v2.json",
      "retrievedAt": "2026-09-09T00:19:36.880Z",
      "revision": "SHA-256 b5cb5ec6e8c7547c4f11a4e61e5ed7e0f28a26208f87bd16c83d1703777a2627"
    },
    "observedAt": "2026-09-04T14:38:06.319Z",
    "evidenceLabel": "Benchmark-owner results · selected cohort",
    "configurationLabel": "Model and reasoning effort",
    "comparabilityNote": "Semi-private set. Seven selected recent model families; cost is unpublished for these configurations.",
    "points": [
      {
        "id": "anthropic-claude-fable-5-1-high",
        "label": "Claude Fable 5.1 (High)",
        "model": "Claude Fable 5.1",
        "provider": "Anthropic",
        "effort": "High",
        "score": 88.75,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/anthropic-claude-fable-5-1",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "anthropic-claude-fable-5-1-low",
        "label": "Claude Fable 5.1 (Low)",
        "model": "Claude Fable 5.1",
        "provider": "Anthropic",
        "effort": "Low",
        "score": 78.33333333333333,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/anthropic-claude-fable-5-1",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "anthropic-claude-fable-5-1-max",
        "label": "Claude Fable 5.1 (Max)",
        "model": "Claude Fable 5.1",
        "provider": "Anthropic",
        "effort": "Max",
        "score": 90,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/anthropic-claude-fable-5-1",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "anthropic-claude-fable-5-1-medium",
        "label": "Claude Fable 5.1 (Medium)",
        "model": "Claude Fable 5.1",
        "provider": "Anthropic",
        "effort": "Medium",
        "score": 86.25,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/anthropic-claude-fable-5-1",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "anthropic-claude-fable-5-1-xhigh",
        "label": "Claude Fable 5.1 (XHigh)",
        "model": "Claude Fable 5.1",
        "provider": "Anthropic",
        "effort": "XHigh",
        "score": 90,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/anthropic-claude-fable-5-1",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "anthropic-claude-opus-5-high",
        "label": "Claude Opus 5 (High)",
        "model": "Claude Opus 5",
        "provider": "Anthropic",
        "effort": "High",
        "score": 88.33,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/anthropic-claude-opus-5",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "anthropic-claude-opus-5-max",
        "label": "Claude Opus 5 (Max)",
        "model": "Claude Opus 5",
        "provider": "Anthropic",
        "effort": "Max",
        "score": 90.42,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/anthropic-claude-opus-5",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "deepseek-v4-pro-0813-high",
        "label": "DeepSeek V4 Pro 0813 (High)",
        "model": "DeepSeek V4 Pro 0813",
        "provider": "DeepSeek",
        "effort": "High",
        "score": 59.72222222222221,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/deepseek-v4-pro-0813",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "deepseek-v4-pro-0813-low",
        "label": "DeepSeek V4 Pro 0813 (Low)",
        "model": "DeepSeek V4 Pro 0813",
        "provider": "DeepSeek",
        "effort": "Low",
        "score": 56.25,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/deepseek-v4-pro-0813",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "deepseek-v4-pro-0813-max",
        "label": "DeepSeek V4 Pro 0813 (Max)",
        "model": "DeepSeek V4 Pro 0813",
        "provider": "DeepSeek",
        "effort": "Max",
        "score": 61.25000000000001,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/deepseek-v4-pro-0813",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "deepseek-v4-pro-0813-none",
        "label": "DeepSeek V4 Pro 0813 (None)",
        "model": "DeepSeek V4 Pro 0813",
        "provider": "DeepSeek",
        "effort": "None",
        "score": 0.8333333333333334,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/deepseek-v4-pro-0813",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "google-gemini-3-7-flash-high",
        "label": "Gemini 3.7 Flash (High)",
        "model": "Gemini 3.7 Flash",
        "provider": "Google",
        "effort": "High",
        "score": 84.58333333333333,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/google-gemini-3-7-flash",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "google-gemini-3-7-flash-low",
        "label": "Gemini 3.7 Flash (Low)",
        "model": "Gemini 3.7 Flash",
        "provider": "Google",
        "effort": "Low",
        "score": 52.916666666666664,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/google-gemini-3-7-flash",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "google-gemini-3-7-flash-medium",
        "label": "Gemini 3.7 Flash (Medium)",
        "model": "Gemini 3.7 Flash",
        "provider": "Google",
        "effort": "Medium",
        "score": 63.74999999999999,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/google-gemini-3-7-flash",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "moonshot-kimi-k3-high",
        "label": "Kimi K3 (High)",
        "model": "Kimi K3",
        "provider": "Moonshot AI",
        "effort": "High",
        "score": 55.00000000000001,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/moonshot-kimi-k3",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "moonshot-kimi-k3-low",
        "label": "Kimi K3 (Low)",
        "model": "Kimi K3",
        "provider": "Moonshot AI",
        "effort": "Low",
        "score": 12.36111111111111,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/moonshot-kimi-k3",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "moonshot-kimi-k3-max",
        "label": "Kimi K3 (Max)",
        "model": "Kimi K3",
        "provider": "Moonshot AI",
        "effort": "Max",
        "score": 60.416666666666664,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/moonshot-kimi-k3",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-high",
        "label": "GPT-5.6 Sol (High)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "High",
        "score": 85.42,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-low",
        "label": "GPT-5.6 Sol (Low)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "Low",
        "score": 42.5,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-max",
        "label": "GPT-5.6 Sol (Max)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "Max",
        "score": 92.5,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-medium",
        "label": "GPT-5.6 Sol (Medium)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "Medium",
        "score": 67.08,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-5-6-sol-xhigh",
        "label": "GPT-5.6 Sol (XHigh)",
        "model": "GPT-5.6 Sol",
        "provider": "OpenAI",
        "effort": "XHigh",
        "score": 90,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-5-6-sol",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-high",
        "label": "GPT-6 Astra (High)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "High",
        "score": 92.08333333333333,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-low",
        "label": "GPT-6 Astra (Low)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "Low",
        "score": 85.41666666666666,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-max",
        "label": "GPT-6 Astra (Max)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "Max",
        "score": 95,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-medium",
        "label": "GPT-6 Astra (Medium)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "Medium",
        "score": 92.08333333333333,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-none",
        "label": "GPT-6 Astra (None)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "None",
        "score": 59.583333333333336,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      },
      {
        "id": "openai-gpt-6-astra-xhigh",
        "label": "GPT-6 Astra (XHigh)",
        "model": "GPT-6 Astra",
        "provider": "OpenAI",
        "effort": "XHigh",
        "score": 93.33333333333333,
        "costUsd": null,
        "sourceUrl": "https://arcprize.org/results/openai-gpt-6-astra",
        "harness": "ARC Prize reasoning evaluation",
        "uncertainty": null
      }
    ]
  }
}
