{
  "schemaVersion": 1,
  "name": "AI Charts benchmark atlas",
  "description": "Versioned AI benchmark charts and source guides across coding, reasoning, research, memory, images, video, audio, and world models. Each measured cohort retains its source, configuration, score unit, and comparison limits.",
  "contentModifiedAt": "2026-09-09T02:50:00Z",
  "comparisonPolicy": "Each dataset is a separate evaluation cohort. Scores, cost bases, harnesses, and versions are not pooled into a universal rank. Source-only entries have no charted observations.",
  "reuseNotice": "AI Charts software is MIT-licensed. Third-party measurements and methodology retain their source terms; this distribution does not grant a new license to them. Cite the source and named evaluation version.",
  "entries": [
    {
      "id": "terminal-bench-4",
      "name": "Terminal-Bench 4",
      "version": "4.0.0",
      "category": "coding",
      "question": "Which agent can finish difficult terminal work?",
      "summary": "Software, systems, CAD, scientific computing, and formal proof tasks executed in a terminal.",
      "source": {
        "name": "Harbor Framework",
        "url": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/latest?leaderboard=4-0-0&tab=leaderboard",
        "methodologyUrl": "https://github.com/harbor-framework/terminal-bench/releases/tag/v4.0.0"
      },
      "measure": "Task success over 66 tasks and five trials per configuration, with source-reported 95% confidence intervals.",
      "comparisonRule": "AI Charts’ primary terminal benchmark. Compare exact version 4.0.0 with the named model, agent version, and effort.",
      "limitations": [
        "An agent and model are evaluated together; this is not a model-only ranking.",
        "Different harnesses and effort settings remain separate configurations.",
        "Cost covers the full 330-trial evaluation. The source does not specify its pricing or confidence-interval method."
      ],
      "coverage": "charted",
      "tags": [
        "terminal",
        "coding",
        "software engineering",
        "agents",
        "computer",
        "standard"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=terminal-bench-4#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/terminal-bench-4",
        "configurationCount": 13,
        "score": {
          "label": "Task success",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "Harbor Framework",
          "url": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/latest?leaderboard=4-0-0&tab=leaderboard",
          "retrievedAt": "2026-09-03T21:57:53.225Z",
          "revision": "83c7a6172d629c6575b785ab12c8db787bb2e323"
        },
        "evidenceLabel": "Owner-published submissions",
        "costLabel": "USD for the full 330-trial evaluation"
      }
    },
    {
      "id": "aa-intelligence-4-3",
      "name": "Artificial Analysis Intelligence Index",
      "version": "4.3",
      "category": "general",
      "question": "How much capability do I get for the cost?",
      "summary": "Broad model capability alongside the cost and output tokens used to achieve it, within the same evaluation cohort.",
      "source": {
        "name": "Artificial Analysis",
        "url": "https://artificialanalysis.ai/models",
        "methodologyUrl": "https://artificialanalysis.ai/methodology/intelligence-benchmarking"
      },
      "measure": "The publisher’s v4.3 index across ten evaluations: agents 30%, coding 20%, scientific reasoning 20%, and general capability 30%.",
      "comparisonRule": "Compare configurations only within v4.3. Its evaluation roster and weights differ from v4.1.1; the historical scores are not on a continuous scale with these scores.",
      "limitations": [
        "An index reflects the publisher’s task mix and weights, not every use case. Reasoning effort changes the evaluated configuration.",
        "Both resources are publisher-reported per-task measures. Output tokens include answer and reasoning; cost also includes input and cache traffic.",
        "Only current, non-estimated configurations with complete native measurements are retained. Missing positive cost is not free. No per-configuration uncertainty is supplied.",
        "The Terminal-Bench component uses Artificial Analysis’s own evaluation harness and must not be merged with the Harbor submission leaderboard."
      ],
      "coverage": "charted",
      "tags": [
        "language",
        "text",
        "intelligence",
        "efficiency",
        "cost",
        "reasoning effort",
        "pareto"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=aa-intelligence-4-3#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/aa-intelligence-4-3",
        "configurationCount": 97,
        "score": {
          "label": "Intelligence Index",
          "unit": "index points",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "Artificial Analysis",
          "url": "https://artificialanalysis.ai/models",
          "retrievedAt": "2026-09-09T01:54:26.626Z"
        },
        "evidenceLabel": "Independent evaluation",
        "costLabel": "USD per Intelligence Index task"
      }
    },
    {
      "id": "image-arena",
      "name": "Image generation · Arena",
      "version": "Overall · Bradley–Terry",
      "category": "image",
      "question": "Which image generators lead Arena’s preference ratings?",
      "summary": "Compare published preference ratings for images generated from text, with uncertainty and vote counts.",
      "source": {
        "name": "Arena",
        "url": "https://arena.ai/leaderboard/text-to-image",
        "methodologyUrl": "https://arena.ai/blog/arena-rank"
      },
      "measure": "Published Bradley–Terry preference rating; higher is better. Source confidence intervals and vote counts included.",
      "comparisonRule": "Text-to-image, overall category only. One complete, dated publisher cohort; no blending with other Arena tracks or diagnostic benchmarks.",
      "limitations": [
        "Ratings depend on the opponent pool and prompt mix. Compare only within this track and publication date; ratings are not percentages.",
        "Overlapping confidence intervals do not establish a clear winner. Vote count is per model, not a count of unique people or prompts.",
        "The licensed export omits preliminary and AutoEval status flags. A row’s evidence status cannot be inferred from its vote count.",
        "This extract has no matched price or latency data. Keep audio, resolution, effort, web-search, and agent settings in the model labels.",
        "Arena supports AutoEval proxy votes for image generation; the export does not identify affected rows. Preference does not guarantee correct composition or faithful editing."
      ],
      "coverage": "charted",
      "tags": [
        "text to image",
        "image generation",
        "preference",
        "GPT Image",
        "MAI",
        "Muse",
        "Grok",
        "Reve",
        "Nano Banana",
        "Seedream"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=image-arena#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/image-arena",
        "configurationCount": 76,
        "score": {
          "label": "Arena preference rating",
          "unit": "Arena points",
          "direction": "higher"
        },
        "source": {
          "name": "Arena · CC BY 4.0",
          "url": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset/blob/bc50ae8cd12e8e0fd019ada807f35e7c04e7d317/text_to_image/latest-00000-of-00001.parquet",
          "revision": "bc50ae8cd12e8e0fd019ada807f35e7c04e7d317",
          "retrievedAt": "2026-09-09T01:57:01.851Z"
        },
        "observedAt": "2026-09-04",
        "evidenceLabel": "Published preference ratings · evidence flags unavailable"
      }
    },
    {
      "id": "image-edit-arena",
      "name": "Image editing · Arena",
      "version": "Overall · Bradley–Terry",
      "category": "image",
      "question": "Which image editors lead Arena’s preference ratings?",
      "summary": "A separate preference comparison for editing an existing image, preserving each published model setting.",
      "source": {
        "name": "Arena",
        "url": "https://arena.ai/leaderboard/image-edit",
        "methodologyUrl": "https://arena.ai/blog/arena-rank"
      },
      "measure": "Published Bradley–Terry preference rating; higher is better. Source confidence intervals and vote counts included.",
      "comparisonRule": "Image editing, overall category only. One complete, dated publisher cohort; no blending with other Arena tracks or diagnostic benchmarks.",
      "limitations": [
        "Ratings depend on the opponent pool and prompt mix. Compare only within this track and publication date; ratings are not percentages.",
        "Overlapping confidence intervals do not establish a clear winner. Vote count is per model, not a count of unique people or prompts.",
        "The licensed export omits preliminary and AutoEval status flags. A row’s evidence status cannot be inferred from its vote count.",
        "This extract has no matched price or latency data. Keep audio, resolution, effort, web-search, and agent settings in the model labels.",
        "Arena supports AutoEval proxy votes for image generation; the export does not identify affected rows. Preference does not guarantee correct composition or faithful editing."
      ],
      "coverage": "charted",
      "tags": [
        "image editing",
        "preference",
        "GPT Image",
        "MAI",
        "Muse",
        "Grok",
        "Seedream",
        "Nano Banana"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=image-edit-arena#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/image-edit-arena",
        "configurationCount": 53,
        "score": {
          "label": "Arena preference rating",
          "unit": "Arena points",
          "direction": "higher"
        },
        "source": {
          "name": "Arena · CC BY 4.0",
          "url": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset/blob/bc50ae8cd12e8e0fd019ada807f35e7c04e7d317/image_edit/latest-00000-of-00001.parquet",
          "revision": "bc50ae8cd12e8e0fd019ada807f35e7c04e7d317",
          "retrievedAt": "2026-09-09T01:57:01.851Z"
        },
        "observedAt": "2026-09-04",
        "evidenceLabel": "Published preference ratings · evidence flags unavailable"
      }
    },
    {
      "id": "video-arena",
      "name": "Video generation · Arena",
      "version": "Overall · Bradley–Terry",
      "category": "video",
      "question": "Which text-to-video systems lead Arena’s preference ratings?",
      "summary": "Compare video generators in the same text-prompt preference pool, keeping audio, resolution, and agent labels intact.",
      "source": {
        "name": "Arena",
        "url": "https://arena.ai/leaderboard/text-to-video",
        "methodologyUrl": "https://arena.ai/blog/arena-rank"
      },
      "measure": "Published Bradley–Terry preference rating; higher is better. Source confidence intervals and vote counts included.",
      "comparisonRule": "Text-to-video, overall category only. One complete, dated publisher cohort; no blending with other Arena tracks or diagnostic benchmarks.",
      "limitations": [
        "Ratings depend on the opponent pool and prompt mix. Compare only within this track and publication date; ratings are not percentages.",
        "Overlapping confidence intervals do not establish a clear winner. Vote count is per model, not a count of unique people or prompts.",
        "The licensed export omits preliminary and AutoEval status flags. A row’s evidence status cannot be inferred from its vote count.",
        "This extract has no matched price or latency data. Keep audio, resolution, effort, web-search, and agent settings in the model labels.",
        "Video preference does not measure physical plausibility, interactive world simulation, or camera-control accuracy."
      ],
      "coverage": "charted",
      "tags": [
        "text to video",
        "video generation",
        "preference",
        "Gemini Omni",
        "Wan",
        "FLUX",
        "Grok",
        "Seedance",
        "MiniMax",
        "Muse",
        "Sora",
        "Veo"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=video-arena#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/video-arena",
        "configurationCount": 48,
        "score": {
          "label": "Arena preference rating",
          "unit": "Arena points",
          "direction": "higher"
        },
        "source": {
          "name": "Arena · CC BY 4.0",
          "url": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset/blob/bc50ae8cd12e8e0fd019ada807f35e7c04e7d317/text_to_video/latest-00000-of-00001.parquet",
          "revision": "bc50ae8cd12e8e0fd019ada807f35e7c04e7d317",
          "retrievedAt": "2026-09-09T01:57:01.851Z"
        },
        "observedAt": "2026-09-04",
        "evidenceLabel": "Published preference ratings · evidence flags unavailable"
      }
    },
    {
      "id": "image-to-video-arena",
      "name": "Image-to-video · Arena",
      "version": "Overall · Bradley–Terry",
      "category": "video",
      "question": "Which systems turn an image into a preferred video?",
      "summary": "An image-conditioned video preference comparison, separate from generation that starts with a text prompt alone.",
      "source": {
        "name": "Arena",
        "url": "https://arena.ai/leaderboard/image-to-video",
        "methodologyUrl": "https://arena.ai/blog/arena-rank"
      },
      "measure": "Published Bradley–Terry preference rating; higher is better. Source confidence intervals and vote counts included.",
      "comparisonRule": "Image-to-video, overall category only. One complete, dated publisher cohort; no blending with other Arena tracks or diagnostic benchmarks.",
      "limitations": [
        "Ratings depend on the opponent pool and prompt mix. Compare only within this track and publication date; ratings are not percentages.",
        "Overlapping confidence intervals do not establish a clear winner. Vote count is per model, not a count of unique people or prompts.",
        "The licensed export omits preliminary and AutoEval status flags. A row’s evidence status cannot be inferred from its vote count.",
        "This extract has no matched price or latency data. Keep audio, resolution, effort, web-search, and agent settings in the model labels.",
        "Video preference does not measure physical plausibility, interactive world simulation, or camera-control accuracy."
      ],
      "coverage": "charted",
      "tags": [
        "image to video",
        "video generation",
        "preference",
        "MiniMax",
        "Gemini Omni",
        "Wan",
        "Seedance",
        "FLUX",
        "Grok",
        "Veo"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=image-to-video-arena#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/image-to-video-arena",
        "configurationCount": 47,
        "score": {
          "label": "Arena preference rating",
          "unit": "Arena points",
          "direction": "higher"
        },
        "source": {
          "name": "Arena · CC BY 4.0",
          "url": "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset/blob/bc50ae8cd12e8e0fd019ada807f35e7c04e7d317/image_to_video/latest-00000-of-00001.parquet",
          "revision": "bc50ae8cd12e8e0fd019ada807f35e7c04e7d317",
          "retrievedAt": "2026-09-09T01:57:01.851Z"
        },
        "observedAt": "2026-09-02",
        "evidenceLabel": "Published preference ratings · evidence flags unavailable"
      }
    },
    {
      "id": "aa-intelligence",
      "name": "Artificial Analysis Intelligence Index",
      "version": "4.1.1",
      "category": "general",
      "question": "How much capability do I get for the cost?",
      "summary": "A versioned snapshot of broad model capability, with the output tokens and cost used to achieve it.",
      "source": {
        "name": "Artificial Analysis",
        "url": "https://artificialanalysis.ai/models",
        "methodologyUrl": "https://artificialanalysis.ai/methodology/intelligence-benchmarking"
      },
      "measure": "An owner-defined index across nine evaluations of agents, coding, scientific reasoning, and general knowledge.",
      "comparisonRule": "Compare models within the retained v4.1.1 cohort. Later index versions use different evaluations or weights and require a separate chart.",
      "limitations": [
        "The publisher introduced v4.3 on September 7, 2026. This historical v4.1.1 chart is not a current-version ranking.",
        "The index reflects the publisher’s task mix and weights, not every use case.",
        "Reasoning effort changes the system being evaluated. Missing cost does not mean free."
      ],
      "coverage": "charted",
      "tags": [
        "language",
        "text",
        "intelligence",
        "efficiency",
        "cost",
        "reasoning effort"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=aa-intelligence#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/aa-intelligence",
        "configurationCount": 135,
        "score": {
          "label": "Intelligence Index",
          "unit": "index points",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "Artificial Analysis",
          "url": "https://artificialanalysis.ai/models",
          "retrievedAt": "2026-09-04T18:58:55.465Z"
        },
        "evidenceLabel": "Independent evaluation · historical version",
        "costLabel": "USD per Intelligence Index task"
      }
    },
    {
      "id": "terminal-bench-science",
      "name": "Terminal-Bench-Science",
      "version": "0.1.0",
      "category": "science",
      "question": "Can an agent carry out scientific work?",
      "summary": "Research tasks that require scientific software, computation, and evidence across five domains.",
      "source": {
        "name": "Terminal-Bench-Science",
        "url": "https://hub.harborframework.com/datasets/terminal-bench-science/terminal-bench-science/0.1.0?leaderboard=v0-1-eval&tab=leaderboard",
        "methodologyUrl": "https://www.terminal-bench-science.ai/announcement"
      },
      "measure": "Resolution rate across 70 tasks and three trials, with separate domain results.",
      "comparisonRule": "Compare the exact 0.1.0 cohort. Error bars show one standard error, which is narrower than a 95% confidence interval.",
      "limitations": [
        "Model and agent versions and several run limits are not specified by the source.",
        "Cost covers the full 210-trial evaluation; source aggregate and domain costs can differ.",
        "A benchmark success does not establish the scientific validity of arbitrary new research."
      ],
      "coverage": "charted",
      "tags": [
        "scientific computing",
        "research",
        "engineering",
        "life science",
        "physics",
        "mathematics"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=terminal-bench-science#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/terminal-bench-science",
        "configurationCount": 11,
        "score": {
          "label": "Resolution rate",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "Terminal-Bench-Science and Harbor Framework",
          "url": "https://hub.harborframework.com/datasets/terminal-bench-science/terminal-bench-science/0.1.0?leaderboard=v0-1-eval&tab=leaderboard",
          "retrievedAt": "2026-09-04T18:59:01.281Z",
          "revision": "f81afac4f11048e77a15dfc8fb1dbfb897fea0ce"
        },
        "observedAt": "2026-09-04T18:07:28.264353+00:00",
        "evidenceLabel": "Owner leaderboard",
        "costLabel": "USD for the full 210-trial evaluation"
      }
    },
    {
      "id": "deep-swe",
      "name": "DeepSWE · AA evaluation",
      "version": "deep-swe",
      "category": "coding",
      "question": "Can an agent complete long software changes?",
      "summary": "Long-horizon software engineering, measured inside Artificial Analysis’s coding-agent evaluation.",
      "source": {
        "name": "Artificial Analysis",
        "url": "https://artificialanalysis.ai/agents/coding-agents/"
      },
      "measure": "Success on software engineering tasks with automated code verification.",
      "comparisonRule": "Keep this Artificial Analysis cohort separate from DataCurve’s direct mini-swe-agent leaderboard and other harnesses.",
      "limitations": [
        "The retained source identifies the dataset as deep-swe without a finer immutable revision.",
        "Cost and time describe the AA coding suite, not the isolated DeepSWE task set.",
        "The checked source does not report per-configuration uncertainty."
      ],
      "coverage": "charted",
      "tags": [
        "coding",
        "software engineering",
        "repository",
        "long horizon",
        "agent"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=deep-swe#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/deep-swe",
        "configurationCount": 68,
        "score": {
          "label": "DeepSWE",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "Artificial Analysis",
          "url": "https://artificialanalysis.ai/agents/coding-agents/",
          "retrievedAt": "2026-09-05T13:33:35.430Z"
        },
        "evidenceLabel": "Independent evaluation",
        "costLabel": "USD per task across the AA coding suite"
      }
    },
    {
      "id": "aa-coding-index",
      "name": "Artificial Analysis Coding Agent Index",
      "version": "AA coding suite",
      "category": "coding",
      "question": "Which coding configuration covers a range of tasks?",
      "summary": "The publisher’s combined view of software changes, terminal work, and repository understanding.",
      "source": {
        "name": "Artificial Analysis",
        "url": "https://artificialanalysis.ai/agents/coding-agents/"
      },
      "measure": "The source’s Coding Agent Index alongside mean evaluation cost, time, and token use.",
      "comparisonRule": "Read the source-defined index within this coding-suite snapshot. It is distinct from the general Intelligence Index.",
      "limitations": [
        "The combined score reflects the evaluator’s task mix.",
        "Different agent harnesses and effort settings affect both performance and resource use.",
        "The suite contains Terminal-Bench 2.1, not Terminal-Bench 4."
      ],
      "coverage": "charted",
      "tags": [
        "coding",
        "agent",
        "index",
        "cost",
        "time",
        "tokens"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=aa-coding-index#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/aa-coding-index",
        "configurationCount": 68,
        "score": {
          "label": "Coding Agent Index",
          "unit": "index points",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "Artificial Analysis",
          "url": "https://artificialanalysis.ai/agents/coding-agents/",
          "retrievedAt": "2026-09-05T13:33:35.430Z"
        },
        "evidenceLabel": "Independent evaluation",
        "costLabel": "USD per task across the AA coding suite"
      }
    },
    {
      "id": "terminal-bench-2-1",
      "name": "Terminal-Bench 2.1 · AA evaluation",
      "version": "2.1",
      "category": "coding",
      "question": "How did these agents perform on the earlier terminal test?",
      "summary": "The earlier terminal task set retained for the Artificial Analysis coding-suite comparison.",
      "source": {
        "name": "Artificial Analysis",
        "url": "https://artificialanalysis.ai/agents/coding-agents/"
      },
      "measure": "Task success on Terminal-Bench 2.1 with the named agent and model.",
      "comparisonRule": "Historical comparison only. Its scores are not comparable with Terminal-Bench 4, the current AI Charts terminal standard.",
      "limitations": [
        "Older task sets can saturate and cover different capabilities.",
        "Cost and time cover the AA coding suite, not this isolated benchmark."
      ],
      "coverage": "charted",
      "tags": [
        "terminal",
        "coding",
        "historical",
        "legacy",
        "agents"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=terminal-bench-2-1#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/terminal-bench-2-1",
        "configurationCount": 68,
        "score": {
          "label": "Terminal-Bench v2.1",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "Artificial Analysis",
          "url": "https://artificialanalysis.ai/agents/coding-agents/",
          "retrievedAt": "2026-09-05T13:33:35.430Z"
        },
        "evidenceLabel": "Independent evaluation",
        "costLabel": "USD per task across the AA coding suite"
      }
    },
    {
      "id": "swe-atlas",
      "name": "SWE-Atlas-QnA",
      "version": "swe-atlas-qna",
      "category": "coding",
      "question": "Can an agent understand a software repository?",
      "summary": "Repository questions evaluated with a strict answer verifier.",
      "source": {
        "name": "Artificial Analysis",
        "url": "https://artificialanalysis.ai/agents/coding-agents/"
      },
      "measure": "Successful answers to repository-understanding questions in the AA evaluation.",
      "comparisonRule": "Compare the retained SWE-Atlas-QnA source cohort and exact agent configuration.",
      "limitations": [
        "Repository understanding is narrower than successfully changing and shipping software.",
        "The source does not expose a finer immutable dataset revision in this snapshot.",
        "Cost and time cover the AA coding suite, not this isolated benchmark."
      ],
      "coverage": "charted",
      "tags": [
        "coding",
        "repository understanding",
        "questions",
        "retrieval",
        "agent"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=swe-atlas#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/swe-atlas",
        "configurationCount": 68,
        "score": {
          "label": "SWE-Atlas-QnA",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "Artificial Analysis",
          "url": "https://artificialanalysis.ai/agents/coding-agents/",
          "retrievedAt": "2026-09-05T13:33:35.430Z"
        },
        "evidenceLabel": "Independent evaluation",
        "costLabel": "USD per task across the AA coding suite"
      }
    },
    {
      "id": "arc-agi-2",
      "name": "ARC-AGI-2",
      "version": "2 · semi-private",
      "category": "reasoning",
      "question": "Can it solve an unfamiliar visual puzzle?",
      "summary": "Infer a rule from a few examples, then apply it to a new grid.",
      "source": {
        "name": "ARC Prize",
        "url": "https://arcprize.org/leaderboard",
        "methodologyUrl": "https://arcprize.org/blog/astra"
      },
      "measure": "Percentage of novel abstract grid tasks solved in ARC Prize's semi-private evaluation.",
      "comparisonRule": "Compare the same dataset split and reasoning effort. This view selects seven recent model families.",
      "limitations": [
        "Strong puzzle performance does not establish general intelligence.",
        "Selected configurations have no published cost in this source.",
        "Scores near the ceiling leave less room to distinguish leading systems."
      ],
      "coverage": "charted",
      "tags": [
        "abstract reasoning",
        "novel problems",
        "visual puzzles",
        "ARC",
        "AGI"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=arc-agi-2#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/arc-agi-2",
        "configurationCount": 28,
        "score": {
          "label": "Tasks solved",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "ARC Prize · selected published observations",
          "url": "https://arcprize.org/media/data/leaderboard/v2.json",
          "retrievedAt": "2026-09-09T00:19:36.880Z",
          "revision": "SHA-256 b5cb5ec6e8c7547c4f11a4e61e5ed7e0f28a26208f87bd16c83d1703777a2627"
        },
        "observedAt": "2026-09-04T14:38:06.319Z",
        "evidenceLabel": "Benchmark-owner results · selected cohort"
      }
    },
    {
      "id": "arc-agi-3-standard",
      "name": "ARC-AGI-3 · Standard",
      "version": "3 · semi-private · Standard",
      "category": "reasoning",
      "question": "Can it learn the rules by interacting?",
      "summary": "Explore an unfamiliar environment and solve it through a shared, minimal agent interface.",
      "source": {
        "name": "ARC Prize",
        "url": "https://arcprize.org/leaderboard",
        "methodologyUrl": "https://arcprize.org/blog/astra"
      },
      "measure": "Action-efficiency score relative to the benchmark's human baseline, expressed as a percentage.",
      "comparisonRule": "Standard harness only. Compare the selected Astra, Sol, and Opus 5 configurations within this view.",
      "limitations": [
        "Cost is the full evaluation spend, not the cost of one task.",
        "These are deterministic puzzle environments; they do not measure open-ended real-world competence.",
        "The Provider Adapter harness is a separate comparison."
      ],
      "coverage": "charted",
      "tags": [
        "interactive reasoning",
        "exploration",
        "planning",
        "ARC",
        "AGI",
        "agent"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=arc-agi-3-standard#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/arc-agi-3-standard",
        "configurationCount": 12,
        "score": {
          "label": "Action-efficiency score",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "ARC Prize · selected published observations",
          "url": "https://arcprize.org/media/data/leaderboard/v3.json",
          "retrievedAt": "2026-09-09T00:19:36.880Z",
          "revision": "SHA-256 d743d7a731ba6b9f102d3624d5fe82011630b90da4200ceb34bd56cf2b221e1c"
        },
        "observedAt": "2026-09-04T14:38:06.320Z",
        "evidenceLabel": "Benchmark-owner results · selected cohort",
        "costLabel": "Total evaluation cost (USD)"
      }
    },
    {
      "id": "arc-agi-3-adapter",
      "name": "ARC-AGI-3 · Provider Adapter",
      "version": "3 · semi-private · Provider Adapter",
      "category": "reasoning",
      "question": "How much does native context management help?",
      "summary": "Astra's reasoning settings with its provider's conversation and compaction features enabled.",
      "source": {
        "name": "ARC Prize",
        "url": "https://arcprize.org/leaderboard",
        "methodologyUrl": "https://arcprize.org/blog/astra"
      },
      "measure": "ARC-AGI-3 action-efficiency score using the Provider Adapter harness.",
      "comparisonRule": "Compare Astra's six effort settings here. These scores are not a ranking against Standard-harness runs.",
      "limitations": [
        "Only Astra is included in this selected adapter cohort.",
        "Cost is total evaluation spend.",
        "Near-perfect performance on this bounded test is not proof of general intelligence."
      ],
      "coverage": "charted",
      "tags": [
        "interactive reasoning",
        "harness",
        "context management",
        "ARC",
        "AGI"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=arc-agi-3-adapter#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/arc-agi-3-adapter",
        "configurationCount": 6,
        "score": {
          "label": "Action-efficiency score",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "ARC Prize · selected published observations",
          "url": "https://arcprize.org/media/data/leaderboard/v3.json",
          "retrievedAt": "2026-09-09T00:19:36.880Z",
          "revision": "SHA-256 d743d7a731ba6b9f102d3624d5fe82011630b90da4200ceb34bd56cf2b221e1c"
        },
        "observedAt": "2026-09-04T14:38:06.320Z",
        "evidenceLabel": "Benchmark-owner results · selected cohort",
        "costLabel": "Total evaluation cost (USD)"
      }
    },
    {
      "id": "deepresearch-bench-ii",
      "name": "DeepResearch Bench II",
      "version": "II · 132 tasks",
      "category": "research",
      "question": "Which research agent produces a useful report?",
      "summary": "Compare information gathering, analysis, and presentation against expert-written research rubrics.",
      "source": {
        "name": "USTC / Metastone",
        "url": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "methodologyUrl": "https://arxiv.org/abs/2601.08536"
      },
      "measure": "Owner-reported weighted rubric score across 132 tasks and 9,430 criteria; component scores appear in the inspector.",
      "comparisonRule": "Keep the exact research product and model generation. An o3-era research run does not represent today's OpenAI product.",
      "limitations": [
        "The source includes historical products and does not date every run.",
        "A model judge evaluates reports; no intervals or comparable costs are published in this table.",
        "This is a selected set of nine systems, not the full source leaderboard."
      ],
      "coverage": "charted",
      "tags": [
        "deep research",
        "reports",
        "citations",
        "analysis",
        "search"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=deepresearch-bench-ii#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/deepresearch-bench-ii",
        "configurationCount": 9,
        "score": {
          "label": "Weighted rubric score",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "USTC / Metastone · selected published observations",
          "url": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
          "retrievedAt": "2026-09-09T00:19:36.880Z",
          "revision": "SHA-256 ba0ccb0a89ff6c8f5d3c97861b072cdb8459c27fdaef2db90e725537f2869562"
        },
        "evidenceLabel": "Benchmark-owner table · mixed historical product versions"
      }
    },
    {
      "id": "longmemeval-v2-small",
      "name": "LongMemEval-V2 · Small",
      "version": "2 · small · paper baselines",
      "category": "memory",
      "question": "Can an agent reuse what it learned before?",
      "summary": "Compare memory systems that turn previous web-agent work into useful evidence for a later question.",
      "source": {
        "name": "UCLA LongMemEval team",
        "url": "https://xiaowu0162.github.io/longmemeval-v2/",
        "methodologyUrl": "https://github.com/xiaowu0162/LongMemEval-V2"
      },
      "measure": "Answer accuracy with a fixed reader. Query latency is shown for each memory method.",
      "comparisonRule": "Compare memory methods within the same history tier and reader configuration. Small and Medium are separate sets.",
      "limitations": [
        "These are six paper baselines; the public submission leaderboard is still empty.",
        "This measures memory-system plus reader performance, not a general model ranking.",
        "Query latency excludes the wider application experience; no comparable dollar costs are published."
      ],
      "coverage": "charted",
      "tags": [
        "memory",
        "retrieval",
        "agent memory",
        "latency",
        "long context"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=longmemeval-v2-small#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/longmemeval-v2-small",
        "configurationCount": 6,
        "score": {
          "label": "Answer accuracy",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "UCLA LongMemEval-V2 paper baselines",
          "url": "https://xiaowu0162.github.io/longmemeval-v2/",
          "retrievedAt": "2026-09-09T00:19:36.880Z",
          "revision": "Paper baselines; evaluation code 2cc8c540bdb87fe6761629b585e727e1c4704520"
        },
        "evidenceLabel": "Paper baselines · community submissions pending"
      }
    },
    {
      "id": "longmemeval-v2-medium",
      "name": "LongMemEval-V2 · Medium",
      "version": "2 · medium · paper baselines",
      "category": "memory",
      "question": "Can an agent reuse what it learned before?",
      "summary": "Compare memory systems that turn previous web-agent work into useful evidence for a later question.",
      "source": {
        "name": "UCLA LongMemEval team",
        "url": "https://xiaowu0162.github.io/longmemeval-v2/",
        "methodologyUrl": "https://github.com/xiaowu0162/LongMemEval-V2"
      },
      "measure": "Answer accuracy with a fixed reader. Query latency is shown for each memory method.",
      "comparisonRule": "Compare memory methods within the same history tier and reader configuration. Small and Medium are separate sets.",
      "limitations": [
        "These are six paper baselines; the public submission leaderboard is still empty.",
        "This measures memory-system plus reader performance, not a general model ranking.",
        "Query latency excludes the wider application experience; no comparable dollar costs are published."
      ],
      "coverage": "charted",
      "tags": [
        "memory",
        "retrieval",
        "agent memory",
        "latency",
        "long context"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=longmemeval-v2-medium#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/longmemeval-v2-medium",
        "configurationCount": 6,
        "score": {
          "label": "Answer accuracy",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "UCLA LongMemEval-V2 paper baselines",
          "url": "https://xiaowu0162.github.io/longmemeval-v2/",
          "retrievedAt": "2026-09-09T00:19:36.880Z",
          "revision": "Paper baselines; evaluation code 2cc8c540bdb87fe6761629b585e727e1c4704520"
        },
        "evidenceLabel": "Paper baselines · community submissions pending"
      }
    },
    {
      "id": "longmemeval",
      "name": "LongMemEval",
      "version": "1 · S / M",
      "category": "memory",
      "question": "Will it remember changing facts across conversations?",
      "summary": "Tests recall, updates, time, reasoning across sessions, and knowing when an answer is absent.",
      "source": {
        "name": "LongMemEval authors",
        "url": "https://xiaowu0162.github.io/long-mem-eval/",
        "methodologyUrl": "https://github.com/xiaowu0162/LongMemEval"
      },
      "measure": "Answer accuracy on 500 questions across long conversation histories.",
      "comparisonRule": "Keep S and M separate; match reader, judge, retrieval budget, and history construction.",
      "limitations": [
        "Retrieval recall@k is not answer accuracy.",
        "Vendor memory-system scores often use different readers and judges."
      ],
      "coverage": "source-only",
      "tags": [
        "memory",
        "conversation",
        "personalization",
        "knowledge updates"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=longmemeval#explore",
      "dataset": null
    },
    {
      "id": "locomo",
      "name": "LoCoMo",
      "version": "ACL 2024 · public 10-conversation set",
      "category": "memory",
      "question": "Can it recall details from a long-running conversation?",
      "summary": "A widely cited test for conversational memory, event summaries, and dialogue continuity.",
      "source": {
        "name": "Snap Research / UNC",
        "url": "https://snap-research.github.io/locomo/",
        "methodologyUrl": "https://github.com/snap-research/locomo"
      },
      "measure": "Question-answering and conversation tasks with long, multi-session histories.",
      "comparisonRule": "Name the public subset, included question categories, scoring method, reader, and retrieval budget.",
      "limitations": [
        "F1, model-judged correctness, and retrieval recall are different scores.",
        "The small conversation set and differing category exclusions limit cross-report comparisons."
      ],
      "coverage": "source-only",
      "tags": [
        "memory",
        "conversation",
        "personalization",
        "retrieval"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=locomo#explore",
      "dataset": null
    },
    {
      "id": "longbench-v2",
      "name": "LongBench v2",
      "version": "2 · 503 questions",
      "category": "memory",
      "question": "Can it reason over a very long document?",
      "summary": "Reading comprehension across documents, conversations, code repositories, and structured data.",
      "source": {
        "name": "LongBench team",
        "url": "https://longbench2.github.io/",
        "methodologyUrl": "https://github.com/THUDM/LongBench"
      },
      "measure": "Multiple-choice accuracy, with short, medium, and long context breakdowns.",
      "comparisonRule": "Match chain-of-thought setting, input truncation, length bucket, and model context limit.",
      "limitations": [
        "Long-context reading does not measure persistent memory between sessions.",
        "Advertised context capacity does not guarantee useful recall at that length."
      ],
      "coverage": "source-only",
      "tags": [
        "long context",
        "documents",
        "reading",
        "code understanding"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=longbench-v2#explore",
      "dataset": null
    },
    {
      "id": "browsecomp",
      "name": "BrowseComp",
      "version": "2025 · 1,266 questions",
      "category": "research",
      "question": "Can it find a hard-to-locate fact on the web?",
      "summary": "Persistent search and multi-step browsing, scored through short factual answers.",
      "source": {
        "name": "OpenAI",
        "url": "https://openai.com/index/browsecomp/",
        "methodologyUrl": "https://github.com/openai/simple-evals"
      },
      "measure": "Accuracy on web questions that require extensive information seeking.",
      "comparisonRule": "Match search tools, context management, agent count, and retry policy; label developer-reported scores.",
      "limitations": [
        "It does not directly evaluate the quality of a long research report.",
        "Live search results and different harnesses make cross-release scores difficult to compare."
      ],
      "coverage": "source-only",
      "tags": [
        "browsing",
        "search",
        "deep research",
        "facts"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=browsecomp#explore",
      "dataset": null
    },
    {
      "id": "browsecomp-plus",
      "name": "BrowseComp-Plus",
      "version": "ACL 2026 · fixed corpus",
      "category": "research",
      "question": "How good is the research agent when search access is controlled?",
      "summary": "A reproducible information-retrieval setting for difficult research questions.",
      "source": {
        "name": "Waterloo / CSIRO / collaborators",
        "url": "https://texttron.github.io/BrowseComp-Plus/",
        "methodologyUrl": "https://github.com/texttron/BrowseComp-Plus"
      },
      "measure": "Answer accuracy and retrieval effectiveness over a released document collection.",
      "comparisonRule": "Keep corpus revision, retriever, retrieval budget, and agent configuration fixed.",
      "limitations": [
        "A fixed corpus cannot represent the freshness or changing access conditions of the live web.",
        "BrowseComp-Plus scores cannot be substituted for BrowseComp scores."
      ],
      "coverage": "source-only",
      "tags": [
        "retrieval",
        "search",
        "deep research",
        "reproducible"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=browsecomp-plus#explore",
      "dataset": null
    },
    {
      "id": "frontiermath",
      "name": "FrontierMath",
      "version": "Tiers 1–3 / Tier 4 · v2",
      "category": "reasoning",
      "question": "Can it solve a difficult research-level math problem?",
      "summary": "Expert-authored mathematical problems with automatically checkable answers.",
      "source": {
        "name": "Epoch AI",
        "url": "https://epoch.ai/frontiermath/tiers-1-4",
        "methodologyUrl": "https://epoch.ai/frontiermath/tiers-1-4/about"
      },
      "measure": "Solved-problem rate under the evaluator's tools and compute budget.",
      "comparisonRule": "Keep version, difficulty tier, holdout set, and tool budget explicit.",
      "limitations": [
        "Tier 4 and Tiers 1–3 answer different difficulty questions.",
        "OpenAI funded the original benchmark and has access to some problems; Epoch describes the held-out subsets."
      ],
      "coverage": "source-only",
      "tags": [
        "math",
        "reasoning",
        "proof",
        "research"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=frontiermath#explore",
      "dataset": null
    },
    {
      "id": "astabench",
      "name": "AstaBench",
      "version": "Scientific research suite",
      "category": "science",
      "question": "Can it carry out the steps of scientific research?",
      "summary": "Evidence spanning literature work, code execution, data analysis, and discovery.",
      "source": {
        "name": "Allen Institute for AI",
        "url": "https://allenai.org/blog/astabench-update-spring-2026",
        "methodologyUrl": "https://github.com/allenai/asta-bench"
      },
      "measure": "Task-specific research outcomes across a suite of scientific evaluations.",
      "comparisonRule": "Select a named sub-benchmark and matching tools; retain agent configuration and uncertainty.",
      "limitations": [
        "Not every submitted agent supports every task family.",
        "A suite average can hide a strong specialty and a missing capability."
      ],
      "coverage": "source-only",
      "tags": [
        "science",
        "literature",
        "data analysis",
        "discovery",
        "research"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=astabench#explore",
      "dataset": null
    },
    {
      "id": "scicode",
      "name": "SciCode",
      "version": "2024 · main problems",
      "category": "science",
      "question": "Can it translate scientific knowledge into working code?",
      "summary": "Coding problems drawn from numerical methods, simulations, and scientific calculations.",
      "source": {
        "name": "SciCode authors",
        "url": "https://scicode-bench.github.io/",
        "methodologyUrl": "https://github.com/scicode-bench/SciCode"
      },
      "measure": "Main-problem or subproblem correctness against scientific test cases.",
      "comparisonRule": "Keep background-information setting, subproblem assistance, and model-generated versus gold earlier steps separate.",
      "limitations": [
        "An assisted subproblem score is not an end-to-end scientific workflow score.",
        "This adds more value inside the science category than as another homepage coding total."
      ],
      "coverage": "source-only",
      "tags": [
        "science",
        "coding",
        "physics",
        "simulation"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=scicode#explore",
      "dataset": null
    },
    {
      "id": "critpt",
      "name": "CritPt",
      "version": "Research-level physics",
      "category": "science",
      "question": "Can it reason through an unfamiliar physics research problem?",
      "summary": "Challenging physics tasks that test reasoning beyond standard academic exams.",
      "source": {
        "name": "Artificial Analysis / CritPt authors",
        "url": "https://artificialanalysis.ai/evaluations/critpt",
        "methodologyUrl": "https://artificialanalysis.ai/methodology/intelligence-benchmarking"
      },
      "measure": "Accuracy on the evaluator's research-level physics problems.",
      "comparisonRule": "Keep the evaluation revision, tools, reasoning budget, and grading protocol fixed.",
      "limitations": [
        "A specialist physics score should not stand in for general scientific usefulness.",
        "Difficulty and domain coverage differ from HLE and Terminal-Bench-Science."
      ],
      "coverage": "source-only",
      "tags": [
        "science",
        "physics",
        "reasoning",
        "research"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=critpt#explore",
      "dataset": null
    },
    {
      "id": "livebench",
      "name": "LiveBench",
      "version": "2026-06-25",
      "category": "general",
      "question": "How does it handle fresh, objectively scored tasks?",
      "summary": "A periodically refreshed collection covering reasoning, language, data, instruction following, and coding.",
      "source": {
        "name": "LiveBench team",
        "url": "https://livebench.ai/",
        "methodologyUrl": "https://github.com/LiveBench/LiveBench"
      },
      "measure": "Objective task scores and category results on a named release.",
      "comparisonRule": "Compare only the same release and task categories; refreshes change the exam.",
      "limitations": [
        "Its broad score overlaps with other general-purpose indices.",
        "Freshness reduces contamination risk; it does not establish its absence."
      ],
      "coverage": "source-only",
      "tags": [
        "general",
        "reasoning",
        "instruction following",
        "language",
        "data analysis"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=livebench#explore",
      "dataset": null
    },
    {
      "id": "livecodebench",
      "name": "LiveCodeBench",
      "version": "Release v6 · dated windows",
      "category": "coding",
      "question": "Can it solve a new programming problem?",
      "summary": "Competitive-programming problems published over time, with executable tests.",
      "source": {
        "name": "LiveCodeBench authors",
        "url": "https://livecodebench.github.io/",
        "methodologyUrl": "https://github.com/LiveCodeBench/LiveCodeBench"
      },
      "measure": "Code-generation correctness on a specified problem-date window and scenario.",
      "comparisonRule": "Match release, start/end dates, scenario, sampling count, and execution budget.",
      "limitations": [
        "Algorithmic problem solving does not establish repository-editing skill.",
        "Different date windows are different cohorts, even if both are called v6."
      ],
      "coverage": "source-only",
      "tags": [
        "coding",
        "algorithms",
        "competitive programming",
        "fresh tasks"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=livecodebench#explore",
      "dataset": null
    },
    {
      "id": "humanitys-last-exam",
      "name": "Humanity’s Last Exam",
      "version": "Classic · 2025-04-03 final set",
      "category": "general",
      "question": "Can it answer difficult questions across expert fields?",
      "summary": "An academic breadth check spanning mathematics, science, and the humanities, with text and image questions.",
      "source": {
        "name": "Center for AI Safety / Scale AI",
        "url": "https://www.lastexam.ai/",
        "methodologyUrl": "https://huggingface.co/datasets/cais/hle"
      },
      "measure": "Answer accuracy on the finalized 2,500-question classic set; calibration is a separate measure.",
      "comparisonRule": "Pin the dataset revision, full versus text-only set, tools, reasoning effort, and answer judge. HLE-Rolling is a different exam.",
      "limitations": [
        "Closed-ended expert questions do not measure open-ended discovery or professional work.",
        "Tool-assisted and no-tool scores cannot be pooled.",
        "The dataset is gated; this guide links to it without redistributing its questions."
      ],
      "coverage": "source-only",
      "tags": [
        "HLE",
        "knowledge",
        "academic",
        "reasoning",
        "multimodal",
        "science"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=humanitys-last-exam#explore",
      "dataset": null
    },
    {
      "id": "gpqa-diamond",
      "name": "GPQA Diamond",
      "version": "Diamond · 198 questions",
      "category": "science",
      "question": "Can it reason through an expert science question?",
      "summary": "A compact multiple-choice test in biology, chemistry, and physics, selected through expert and non-expert review.",
      "source": {
        "name": "GPQA authors",
        "url": "https://github.com/idavidrein/gpqa",
        "methodologyUrl": "https://arxiv.org/html/2311.12022v1"
      },
      "measure": "Multiple-choice answer accuracy on the 198-question Diamond subset; random choice has a 25% baseline.",
      "comparisonRule": "Keep Diamond separate from Main and Extended. Match prompts, answer shuffling, tools, reasoning budget, and sampling policy.",
      "limitations": [
        "A small fixed exam does not establish scientific discovery or laboratory competence.",
        "Near-ceiling scores and small gaps require uncertainty, not a confident rank order.",
        "Best-of-many success is not single-attempt accuracy."
      ],
      "coverage": "source-only",
      "tags": [
        "GPQA",
        "science",
        "physics",
        "chemistry",
        "biology",
        "reasoning"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=gpqa-diamond#explore",
      "dataset": null
    },
    {
      "id": "swe-bench-verified",
      "name": "SWE-bench Verified",
      "version": "Verified · 500 tasks",
      "category": "coding",
      "question": "Can it repair an issue in an existing repository?",
      "summary": "Real repository issues with executable tests, useful for understanding a coding agent's patching ability.",
      "source": {
        "name": "SWE-bench team",
        "url": "https://www.swebench.com/",
        "methodologyUrl": "https://www.swebench.com/verified.html"
      },
      "measure": "Percentage of the 500 human-filtered task instances resolved.",
      "comparisonRule": "Use one task revision, agent, budget, and attempt policy. The Bash Only view controls the agent; the full board mixes systems.",
      "limitations": [
        "mini-SWE-agent 1.x and 2.x change action handling and sampling and are not automatically comparable.",
        "Public task exposure and test quality limit conclusions about new, unseen work.",
        "A repair score does not measure an entire software development workflow."
      ],
      "coverage": "source-only",
      "tags": [
        "SWE-bench",
        "coding",
        "repository",
        "bug fixing",
        "agent",
        "software engineering"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=swe-bench-verified#explore",
      "dataset": null
    },
    {
      "id": "swe-bench-pro",
      "name": "SWE-bench Pro",
      "version": "Public · 731 tasks",
      "category": "coding",
      "question": "Can it make a larger change in a complex codebase?",
      "summary": "Longer software-engineering tasks across public application and developer-tool repositories.",
      "source": {
        "name": "Scale AI",
        "url": "https://labs.scale.com/leaderboard/swe_bench_pro_public",
        "methodologyUrl": "https://github.com/scaleapi/SWE-bench_Pro-os"
      },
      "measure": "Resolve rate: the percentage of public tasks whose patches pass the required new and regression tests.",
      "comparisonRule": "Keep Public, Private, and Held-out sets separate. Match dataset revision, agent harness, turn limit, cost cap, and attempts.",
      "limitations": [
        "The owner board mixes harnesses and capped versus uncapped runs; its rows are not one controlled cohort.",
        "Public repository licensing is not evidence that models have never seen the code.",
        "Task-quality disputes make this supporting evidence, not a universal replacement for Verified or Terminal-Bench."
      ],
      "coverage": "source-only",
      "tags": [
        "SWE-bench Pro",
        "coding",
        "repository",
        "long horizon",
        "software engineering"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=swe-bench-pro#explore",
      "dataset": null
    },
    {
      "id": "cursorbench",
      "name": "CursorBench",
      "version": "3.2",
      "category": "coding",
      "question": "Which model handles the kind of work done inside Cursor?",
      "summary": "Ambiguous multi-file tasks drawn from Cursor usage, including instruction following and advanced tool use.",
      "source": {
        "name": "Cursor · vendor-reported",
        "url": "https://cursor.com/cursorbench",
        "methodologyUrl": "https://cursor.com/blog/cursorbench"
      },
      "measure": "Cursor's task-correctness score in percent, alongside average API-priced cost per task, tokens, and steps.",
      "comparisonRule": "Compare only version 3.2 with its Cursor agent setup and exact model effort. Preserve the pricing revision for cost comparisons.",
      "limitations": [
        "This is a vendor-owned internal evaluation, not an independent cross-product ranking.",
        "Private tasks and agentic grading limit external reproduction.",
        "The task set changed from 3.1; small score gaps may reflect evaluation variance."
      ],
      "coverage": "source-only",
      "tags": [
        "CursorBench",
        "coding",
        "Cursor",
        "production",
        "multi-file",
        "agent",
        "cost"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=cursorbench#explore",
      "dataset": null
    },
    {
      "id": "gdpval",
      "name": "GDPval",
      "version": "2025 · original evaluation",
      "category": "work",
      "question": "Can it deliver work an experienced professional would accept?",
      "summary": "Occupational tasks with reference files and finished deliverables, including documents, presentations, and spreadsheets.",
      "source": {
        "name": "OpenAI",
        "url": "https://openai.com/index/gdpval/",
        "methodologyUrl": "https://huggingface.co/datasets/openai/gdpval"
      },
      "measure": "Quality of completed work compared with expert deliverables across 44 occupations; the public gold set contains 220 tasks.",
      "comparisonRule": "Name the full or gold task set, grading protocol, scaffolding, and whether ties count toward the reported win rate.",
      "limitations": [
        "The original evaluation and Artificial Analysis's GDPval-AA use different evaluation protocols.",
        "Producing one deliverable is not the same as doing an entire job or handling its organizational context.",
        "Developer-reported results need to retain their exact human or model-judge protocol."
      ],
      "coverage": "source-only",
      "tags": [
        "professional work",
        "documents",
        "spreadsheets",
        "presentations",
        "knowledge work",
        "deliverables"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=gdpval#explore",
      "dataset": null
    },
    {
      "id": "gdpval-aa",
      "name": "GDPval-AA",
      "version": "v2",
      "category": "work",
      "question": "Which tool-using model produces the strongest professional deliverable?",
      "summary": "Artificial Analysis evaluates GDPval work products in its Stirrup agent environment, then compares the outputs head to head.",
      "source": {
        "name": "Artificial Analysis",
        "url": "https://artificialanalysis.ai/evaluations/gdpval-aa",
        "methodologyUrl": "https://artificialanalysis.ai/methodology/intelligence-benchmarking"
      },
      "measure": "Pairwise Elo rating anchored to a human-expert baseline of 1,000, with source-reported uncertainty.",
      "comparisonRule": "Keep v2, the Stirrup environment, reasoning effort, judge panel, and rating pool together. Elo is not percent correct.",
      "limitations": [
        "The v2 environment, turn limit, and panel of judges differ from v1.",
        "Model-judged preferences are not a direct measurement of business value or worker replacement.",
        "Per-task cost must not be mixed with full-evaluation spend."
      ],
      "coverage": "source-only",
      "tags": [
        "professional work",
        "office",
        "deliverables",
        "Elo",
        "cost",
        "agents"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=gdpval-aa#explore",
      "dataset": null
    },
    {
      "id": "osworld-v2",
      "name": "OSWorld 2.0",
      "version": "osworld-v2-2026.08.08",
      "category": "computer-use",
      "question": "Can an agent finish a workflow across desktop and web apps?",
      "summary": "Long computer-use tasks with verifiable outcomes, not just recognizing a button in a screenshot.",
      "source": {
        "name": "OSWorld / XLang Lab",
        "url": "https://osworld-v2.xlang.ai/",
        "methodologyUrl": "https://github.com/xlang-ai/OSWorld-V2/blob/v2026.08.08/benchmark_releases/osworld-v2-2026.08.08.json"
      },
      "measure": "Task completion and partial reward across the pinned 108-workflow release, reported separately.",
      "comparisonRule": "Pin code, tasks, assets, website, provider image, step budget, and input/action interface before comparing systems.",
      "limitations": [
        "OSWorld-Verified and OSWorld 2.0 are different task cohorts.",
        "Partial progress is not a completed workflow.",
        "Gated environment assets and long runs affect reproducibility; the release manifest identifies the required versions."
      ],
      "coverage": "source-only",
      "tags": [
        "computer use",
        "desktop",
        "browser",
        "GUI",
        "workflow",
        "agents"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=osworld-v2#explore",
      "dataset": null
    },
    {
      "id": "tau-bench-3",
      "name": "τ³-bench",
      "version": "3 · v1.0.1 grading",
      "category": "work",
      "question": "Can a service agent solve the issue while following policy?",
      "summary": "Simulated customer-service conversations combine tool actions, user coordination, and domain rules; newer tracks add knowledge retrieval and voice.",
      "source": {
        "name": "Sierra Research",
        "url": "https://taubench.com/",
        "methodologyUrl": "https://github.com/sierra-research/tau2-bench"
      },
      "measure": "Task success and repeated-trial reliability within a named domain and communication mode.",
      "comparisonRule": "Match domain, task split, user simulator, trials, and text or voice mode. pass^k consistency is not pass@k best-of-k success.",
      "limitations": [
        "The repository retains the tau2-bench name while the current suite is τ³-bench.",
        "Banking-knowledge results before v1.0.1 are not comparable with the corrected grading.",
        "Simulated service interactions do not cover every live customer or organizational policy."
      ],
      "coverage": "source-only",
      "tags": [
        "customer service",
        "tool use",
        "policy",
        "reliability",
        "voice",
        "tau-bench",
        "tau2",
        "knowledge retrieval"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=tau-bench-3#explore",
      "dataset": null
    },
    {
      "id": "wise-verified",
      "name": "WISE Verified",
      "version": "Verified · Qwen3.5-35B-A3B",
      "category": "image",
      "question": "Can it draw what a prompt implies?",
      "summary": "Image generation that needs world knowledge: culture, time, space, biology, physics, and chemistry.",
      "source": {
        "name": "WISE benchmark team",
        "url": "https://github.com/PKU-YuanGroup/WISE/blob/main/leadboard.md",
        "methodologyUrl": "https://github.com/PKU-YuanGroup/WISE"
      },
      "measure": "Weighted knowledge-consistency score, 0–1; higher is better.",
      "comparisonRule": "Only the Verified prompts and Qwen3.5-35B-A3B judge. Keep agent and chain-of-thought configurations named.",
      "limitations": [
        "Not an aesthetic preference or image-editing test.",
        "The Verified prompts and judge changed in 2026; legacy WISE scores are not comparable.",
        "Published research cohort, not a complete inventory of today's image models."
      ],
      "coverage": "charted",
      "tags": [
        "image generation",
        "text to image",
        "world knowledge",
        "reasoning",
        "Qwen",
        "Nano Banana",
        "GPT Image"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=wise-verified#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/wise-verified",
        "configurationCount": 29,
        "score": {
          "label": "Knowledge consistency",
          "unit": "score",
          "direction": "higher",
          "minimum": 0,
          "maximum": 1
        },
        "source": {
          "name": "WISE benchmark team",
          "url": "https://github.com/PKU-YuanGroup/WISE/blob/main/leadboard.md",
          "retrievedAt": "2026-09-09T00:05:33.978Z",
          "revision": "sha256:1b06a1e2697244ff1f29417662bc34b5708587c6bf335e0de65c80d984972263"
        },
        "evidenceLabel": "2026 Verified research cohort"
      }
    },
    {
      "id": "geditbench-2",
      "name": "GEditBench",
      "version": "2",
      "category": "image",
      "question": "Can it make an edit without breaking the rest?",
      "summary": "Instruction following, visual quality, and preservation of the original image across 23 editing tasks.",
      "source": {
        "name": "GEditBench v2 team",
        "url": "https://zhangqijiang07.github.io/gedit2_web/",
        "methodologyUrl": "https://github.com/ZhangqiJiang07/GEditBench_v2"
      },
      "measure": "Overall pairwise Elo; higher is better. Publisher bootstrap intervals included.",
      "comparisonRule": "GPT-4o judges instruction and quality; PVC-Judge scores consistency. Compare within this fixed evaluation pool.",
      "limitations": [
        "Automated, human-aligned judging is not direct human voting.",
        "This is the 2026 paper cohort; the dated API names are retained.",
        "Safety refusals and failures leave some models with fewer evaluated samples.",
        "Elo is pool-relative, not a percentage; overlapping intervals do not establish a winner."
      ],
      "coverage": "charted",
      "tags": [
        "image editing",
        "identity",
        "preservation",
        "instruction following",
        "FLUX",
        "Seedream",
        "Qwen Image Edit"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=geditbench-2#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/geditbench-2",
        "configurationCount": 16,
        "score": {
          "label": "Editing overall",
          "unit": "Elo",
          "direction": "higher"
        },
        "source": {
          "name": "GEditBench v2 team",
          "url": "https://zhangqijiang07.github.io/gedit2_web/",
          "retrievedAt": "2026-09-09T00:05:33.978Z",
          "revision": "sha256:87a97fa869d8d879e2f834520a109632e41c7f8e43724aabae6bb13882f838cc"
        },
        "evidenceLabel": "2026 paper cohort · model-judged"
      }
    },
    {
      "id": "videophy-2",
      "name": "VideoPhy",
      "version": "2 · human evaluation",
      "category": "video",
      "question": "Does the action obey basic physics?",
      "summary": "Human reviewers check whether a generated video both follows the prompt and respects physical commonsense.",
      "source": {
        "name": "VideoPhy2 team",
        "url": "https://videophy2.github.io/",
        "methodologyUrl": "https://github.com/Hritikbansal/videophy/tree/main/VIDEOPHY2"
      },
      "measure": "Joint semantic and physical adherence, %; higher is better.",
      "comparisonRule": "Human-evaluated All subset only. Hard, physical-activity, and object-interaction scores remain separate details.",
      "limitations": [
        "A 2025 research cohort, not a current ranking of video generators.",
        "Physical plausibility is different from cinematic appeal.",
        "Automatic VideoPhy2-eval scores must not be mixed with these human judgments."
      ],
      "coverage": "charted",
      "tags": [
        "video generation",
        "physics",
        "motion",
        "Sora",
        "Wan",
        "Cosmos",
        "world models"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=videophy-2#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/videophy-2",
        "configurationCount": 7,
        "score": {
          "label": "Prompt + physical adherence",
          "unit": "%",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "VideoPhy2 team",
          "url": "https://videophy2.github.io/",
          "retrievedAt": "2026-09-09T00:05:33.978Z",
          "revision": "sha256:bad716a62a9247323932ca8ea305feb4a2f373ebe01233beece8e3dbea4fbd34"
        },
        "evidenceLabel": "2025 research cohort · human-evaluated"
      }
    },
    {
      "id": "omnidocbench",
      "name": "OmniDocBench",
      "version": "1.6_full",
      "category": "image",
      "question": "Can it turn a difficult document into usable text?",
      "summary": "Document extraction across text, formulas, tables, and reading order, comparing specialist pipelines with general vision models.",
      "source": {
        "name": "OpenDataLab",
        "url": "https://github.com/opendatalab/OmniDocBench",
        "methodologyUrl": "https://github.com/opendatalab/OmniDocBench#evaluation"
      },
      "measure": "Publisher overall score, 0–100; higher is better. Component error rates retain their native direction.",
      "comparisonRule": "Use the exact v1.6_full result table; do not pool v1.0, v1.5, or other document datasets.",
      "limitations": [
        "The repository advertises v1.7 but still labels this model table v1.6_full; its table label is preserved.",
        "A parsing system and a general vision model are different deployment choices.",
        "No matched latency or cost measurements are provided in this extract."
      ],
      "coverage": "charted",
      "tags": [
        "OCR",
        "PDF",
        "documents",
        "vision",
        "multimodal understanding",
        "tables",
        "formulas",
        "PaddleOCR",
        "GLM OCR"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=omnidocbench#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/omnidocbench",
        "configurationCount": 32,
        "score": {
          "label": "Document extraction overall",
          "unit": "/ 100",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "OpenDataLab",
          "url": "https://github.com/opendatalab/OmniDocBench",
          "retrievedAt": "2026-09-09T00:05:33.978Z",
          "revision": "sha256:1c3f6f7e0dc6027e96161fd368f68b41b1e797b2329a4decbefdd4f2d2ec31eb"
        },
        "evidenceLabel": "Published v1.6_full results · run dates unspecified"
      }
    },
    {
      "id": "worldscore-static",
      "name": "WorldScore",
      "version": "Static · 2025 author cohort",
      "category": "world",
      "question": "Can a generated world stay coherent as the camera moves?",
      "summary": "A common scene-generation protocol compares camera control, scene consistency, and quality across video, 3D, and 4D systems.",
      "source": {
        "name": "WorldScore team",
        "url": "https://huggingface.co/spaces/Howieeeee/WorldScore_Leaderboard",
        "methodologyUrl": "https://haoyi-duan.github.io/WorldScore/"
      },
      "measure": "WorldScore-Static, 0–100; higher is better.",
      "comparisonRule": "Only systems sampled and evaluated by the WorldScore authors on March 30, 2025. Input and system types stay visible.",
      "limitations": [
        "Historical research comparison; newer model-team submissions are not in this chart.",
        "Static and Dynamic are distinct scores; this chart does not measure interactive control latency or physical simulation accuracy.",
        "A strong generated video is not evidence of an action-conditioned world simulator."
      ],
      "coverage": "charted",
      "tags": [
        "world models",
        "3D",
        "4D",
        "camera",
        "scene consistency",
        "spatial",
        "WonderWorld",
        "CogVideoX"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=worldscore-static#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/worldscore-static",
        "configurationCount": 19,
        "score": {
          "label": "WorldScore-Static",
          "unit": "/ 100",
          "direction": "higher",
          "minimum": 0,
          "maximum": 100
        },
        "source": {
          "name": "WorldScore team",
          "url": "https://huggingface.co/spaces/Howieeeee/WorldScore_Leaderboard",
          "retrievedAt": "2026-09-09T00:05:33.978Z",
          "revision": "sha256:6e97e0bc57b0c2472d3685bcd6f5ab96b64d17f0a6188fb8eb29acdedcba14ff"
        },
        "observedAt": "2025-03-30T00:00:00Z",
        "evidenceLabel": "March 2025 benchmark-author cohort"
      }
    },
    {
      "id": "geneval-2",
      "name": "GenEval2",
      "version": "2025-12",
      "category": "image",
      "question": "Are the objects, attributes, and relationships correct?",
      "summary": "Checks the detailed compositional content of generated images, beyond whether an image looks good.",
      "source": {
        "name": "GenEval2 authors / Meta",
        "url": "https://github.com/facebookresearch/GenEval2",
        "methodologyUrl": "https://arxiv.org/abs/2512.16853"
      },
      "measure": "Soft-TIFA compositional alignment; higher is better.",
      "comparisonRule": "Keep arithmetic and geometric aggregation separate; GenEval2 is not the original GenEval score.",
      "limitations": [
        "Judge and prompt-processing settings affect results.",
        "Use the same benchmark release and scoring configuration when comparing results."
      ],
      "coverage": "source-only",
      "tags": [
        "image generation",
        "composition",
        "counting",
        "objects",
        "attributes",
        "prompt adherence"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=geneval-2#explore",
      "dataset": null
    },
    {
      "id": "dpg-bench",
      "name": "DPG-Bench",
      "version": "2024-03",
      "category": "image",
      "question": "Does a detailed image prompt survive intact?",
      "summary": "Breaks dense text-to-image prompts into questions about the requested content.",
      "source": {
        "name": "ELLA / DPG-Bench authors",
        "url": "https://github.com/TencentQQGYLab/ELLA/tree/main/dpg_bench",
        "methodologyUrl": "https://ella-diffusion.github.io/"
      },
      "measure": "Dense-prompt alignment score; higher is better.",
      "comparisonRule": "Require identical prompts, question dependencies, VQA evaluator, and rewriting policy.",
      "limitations": [
        "Prompt rewriting can change what is being tested.",
        "A legacy compositional test, not evidence of current aesthetic preference or editing quality."
      ],
      "coverage": "source-only",
      "tags": [
        "image generation",
        "dense prompts",
        "instruction following",
        "semantic alignment"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=dpg-bench#explore",
      "dataset": null
    },
    {
      "id": "t2i-compbench",
      "name": "T2I-CompBench++",
      "version": "++ · 2024 suite",
      "category": "image",
      "question": "Can it bind the right attribute to the right object?",
      "summary": "Separates color, shape, texture, relationships, counting, and complex scene composition.",
      "source": {
        "name": "T2I-CompBench authors",
        "url": "https://github.com/Karine-Huang/T2I-CompBench"
      },
      "measure": "Dimension-specific composition scores; higher is better.",
      "comparisonRule": "Use the same ++ task split and evaluator per dimension; avoid a made-up average of unlike evaluators.",
      "limitations": [
        "The original suite and ++ extension have different task coverage.",
        "Detector and visual-judge errors can look like generation failures."
      ],
      "coverage": "source-only",
      "tags": [
        "image generation",
        "composition",
        "attribute binding",
        "counting",
        "spatial relations"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=t2i-compbench#explore",
      "dataset": null
    },
    {
      "id": "vbench-2",
      "name": "VBench",
      "version": "2.0 · 2025",
      "category": "video",
      "question": "Where does a video generator break down?",
      "summary": "A diagnostic view of human fidelity, creativity, controllability, physics, and commonsense across 18 dimensions.",
      "source": {
        "name": "VBench authors",
        "url": "https://huggingface.co/spaces/Vchitect/VBench_Leaderboard",
        "methodologyUrl": "https://github.com/Vchitect/VBench/tree/master/VBench-2.0"
      },
      "measure": "Dimension and five-aspect aggregate scores; higher is better.",
      "comparisonRule": "Keep VBench 2.0 separate from VBench 1.0, VBench++, and image-to-video tracks.",
      "limitations": [
        "The overall score gives equal weight to five aspects, not to every dimension.",
        "Submission settings and evaluator revisions affect comparability.",
        "Automatic diagnostics do not replace viewer preference."
      ],
      "coverage": "source-only",
      "tags": [
        "video generation",
        "quality",
        "physics",
        "human fidelity",
        "controllability",
        "creativity"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=vbench-2#explore",
      "dataset": null
    },
    {
      "id": "worldmodelbench",
      "name": "WorldModelBench",
      "version": "2025",
      "category": "world",
      "question": "Does a predicted world follow the instruction and physical rules?",
      "summary": "Image-conditioned world generation judged across everyday, simulated, and embodied environments.",
      "source": {
        "name": "WorldModelBench team",
        "url": "https://worldmodelbench-team.github.io/",
        "methodologyUrl": "https://github.com/WorldModelBench-Team/WorldModelBench"
      },
      "measure": "Instruction, commonsense, and physical-adherence scores; higher is better.",
      "comparisonRule": "Hold the input images, environment domains, and world-model evaluator fixed.",
      "limitations": [
        "A small research suite is not an interactive-agent deployment test.",
        "The judge and evaluated model cohort matter; a newer demo is not a benchmark result."
      ],
      "coverage": "source-only",
      "tags": [
        "world models",
        "physics",
        "embodied",
        "simulation",
        "instruction following"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=worldmodelbench#explore",
      "dataset": null
    },
    {
      "id": "open-asr",
      "name": "Open ASR Leaderboard",
      "version": "Public evaluation tracks",
      "category": "audio",
      "question": "Which system transcribes speech accurately and quickly?",
      "summary": "Speech recognition across datasets and languages, with accuracy and inference speed reported separately.",
      "source": {
        "name": "Hugging Face audio team",
        "url": "https://huggingface.co/spaces/hf-audio/open_asr_leaderboard",
        "methodologyUrl": "https://github.com/huggingface/open_asr_leaderboard"
      },
      "measure": "Word error rate: lower is better. Real-time factor speedup: higher is better.",
      "comparisonRule": "Match language, dataset, chunking, decoding settings, and hardware before comparing speed.",
      "limitations": [
        "Short-form, long-form, and multilingual tracks are different cohorts.",
        "GPU throughput is not end-to-end API latency.",
        "Hardware, chunking, and decoding settings must match for a fair efficiency comparison."
      ],
      "coverage": "source-only",
      "tags": [
        "speech",
        "transcription",
        "ASR",
        "audio",
        "WER",
        "latency",
        "multilingual"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=open-asr#explore",
      "dataset": null
    },
    {
      "id": "seed-tts-eval",
      "name": "SEED-TTS-Eval",
      "version": "2024",
      "category": "audio",
      "question": "Can it say the right words in the requested voice?",
      "summary": "Zero-shot speech synthesis evaluated for intelligibility and similarity to a reference speaker in English and Mandarin.",
      "source": {
        "name": "ByteDance Speech",
        "url": "https://github.com/BytedanceSpeech/seed-tts-eval"
      },
      "measure": "Word error rate: lower is better. Speaker similarity: higher is better.",
      "comparisonRule": "Keep language, ASR scorer, speaker encoder, and voice-conditioning protocol fixed.",
      "limitations": [
        "Speaker similarity and correct words do not measure expressive naturalness.",
        "Voice-cloning evaluation is distinct from choosing a ready-made production voice."
      ],
      "coverage": "source-only",
      "tags": [
        "audio generation",
        "speech synthesis",
        "TTS",
        "voice cloning",
        "speaker similarity"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=seed-tts-eval#explore",
      "dataset": null
    },
    {
      "id": "voicearena",
      "name": "Voice Arena",
      "version": "TTS v1",
      "category": "audio",
      "question": "Which synthetic voice do listeners prefer?",
      "summary": "Pairwise listener preference for speech synthesis, separated by language.",
      "source": {
        "name": "Voice Arena",
        "url": "https://voicearena.com/",
        "methodologyUrl": "https://voicearena.com/tts-methodology"
      },
      "measure": "Language-specific preference rating; higher is better.",
      "comparisonRule": "Compare within the same language and voice setup; retain uncertainty and vote counts.",
      "limitations": [
        "Different language pools are not on one universal scale.",
        "Listener preference does not establish word-perfect transcription, speaker cloning, or conversational latency."
      ],
      "coverage": "source-only",
      "tags": [
        "audio generation",
        "TTS",
        "voice",
        "naturalness",
        "human preference",
        "multilingual"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=voicearena#explore",
      "dataset": null
    },
    {
      "id": "mmau-pro",
      "name": "MMAU-Pro",
      "version": "2025-08",
      "category": "audio",
      "question": "Can it reason about what it hears?",
      "summary": "Audio understanding across speech, environmental sounds, and music, including long and multiple recordings.",
      "source": {
        "name": "MMAU-Pro authors",
        "url": "https://github.com/sonalkum/MMAUPro",
        "methodologyUrl": "https://arxiv.org/abs/2508.13992"
      },
      "measure": "Task-specific audio reasoning accuracy; higher is better.",
      "comparisonRule": "Keep multiple-choice, open-ended, and instruction-following evaluation modes distinct.",
      "limitations": [
        "Not a speech-generation or music-generation preference test.",
        "Transcription-only systems do not receive the same sensory input as audio-native models."
      ],
      "coverage": "source-only",
      "tags": [
        "audio understanding",
        "music",
        "sound",
        "reasoning",
        "multimodal understanding"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=mmau-pro#explore",
      "dataset": null
    },
    {
      "id": "mmmu-pro",
      "name": "MMMU-Pro",
      "version": "2024 suite",
      "category": "image",
      "question": "Can it reason from a diagram, not just read its text?",
      "summary": "Expert-level multimodal problems designed to reduce text-only shortcuts.",
      "source": {
        "name": "MMMU authors",
        "url": "https://mmmu-benchmark.github.io/",
        "methodologyUrl": "https://github.com/MMMU-Benchmark/MMMU/tree/main/mmmu-pro"
      },
      "measure": "Accuracy, %; higher is better.",
      "comparisonRule": "Keep Standard 10-option and Vision settings named; original MMMU and MMMU-Pro are not interchangeable.",
      "limitations": [
        "Prompting, resolution, tools, and reasoning effort affect the result.",
        "Academic question answering is not document parsing or visual design quality."
      ],
      "coverage": "source-only",
      "tags": [
        "vision",
        "multimodal understanding",
        "diagrams",
        "visual reasoning",
        "science"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=mmmu-pro#explore",
      "dataset": null
    },
    {
      "id": "video-mme",
      "name": "Video-MME",
      "version": "2024 / CVPR 2025",
      "category": "video",
      "question": "Can it understand a long video?",
      "summary": "Video question answering at short, medium, and long durations.",
      "source": {
        "name": "Video-MME authors",
        "url": "https://video-mme.github.io/",
        "methodologyUrl": "https://github.com/MME-Benchmarks/Video-MME"
      },
      "measure": "Question-answer accuracy, %; higher is better.",
      "comparisonRule": "Keep with-subtitle and without-subtitle tracks separate; record frame sampling and audio access.",
      "limitations": [
        "Extra frames, subtitles, and audio change the information supplied to the model.",
        "Understanding video is different from generating it."
      ],
      "coverage": "source-only",
      "tags": [
        "video understanding",
        "long context",
        "multimodal understanding",
        "subtitles",
        "perception"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=video-mme#explore",
      "dataset": null
    },
    {
      "id": "aa-image-arena",
      "name": "Image Arena",
      "version": "Live · text-to-image",
      "category": "image",
      "question": "Which generated image do people prefer?",
      "summary": "Blind human preference provides an aesthetic and overall-utility view alongside diagnostic image benchmarks.",
      "source": {
        "name": "Artificial Analysis",
        "url": "https://artificialanalysis.ai/image/leaderboard/text-to-image"
      },
      "measure": "Pairwise Elo with confidence intervals; higher is better.",
      "comparisonRule": "Text-to-image and editing have separate opponent pools; retain model settings, votes, and rating uncertainty.",
      "limitations": [
        "Preference is not a factuality or exact-composition guarantee.",
        "Use the publisher’s current pool, vote counts, and intervals when comparing its scores."
      ],
      "coverage": "source-only",
      "tags": [
        "image generation",
        "human preference",
        "aesthetics",
        "price",
        "latency"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=aa-image-arena#explore",
      "dataset": null
    },
    {
      "id": "aa-video-arena",
      "name": "Video Arena",
      "version": "Live · generation tracks",
      "category": "video",
      "question": "Which video looks best to viewers?",
      "summary": "Human preference for generated video, with separate input and audio tracks.",
      "source": {
        "name": "Artificial Analysis",
        "url": "https://artificialanalysis.ai/video/leaderboard/text-to-video",
        "methodologyUrl": "https://artificialanalysis.ai/video/methodology"
      },
      "measure": "Pairwise Elo with confidence intervals; higher is better.",
      "comparisonRule": "Keep text-to-video, image-to-video, and with-audio pools separate; match resolution, duration, and frame rate for cost.",
      "limitations": [
        "A silent-video score does not describe audio quality.",
        "Use the publisher’s matching input, duration, resolution, and audio track when comparing results."
      ],
      "coverage": "source-only",
      "tags": [
        "video generation",
        "human preference",
        "aesthetics",
        "audio",
        "price",
        "latency"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=aa-video-arena#explore",
      "dataset": null
    },
    {
      "id": "open-asr-ami-cleaned",
      "name": "Open ASR · meeting transcription",
      "version": "AMI-Cleaned · English test · 2026-09-04",
      "category": "audio",
      "question": "Which open-weight speech models make fewer errors in English meetings?",
      "summary": "Ten selected configurations on the same cleaned meeting-transcription test. These are publisher-reported results with different inference pipelines.",
      "source": {
        "name": "Hugging Face Open ASR Leaderboard",
        "url": "https://huggingface.co/datasets/hf-audio/open-asr-leaderboard-results/blob/ba5712d5ace8f785fa0daae1aecea8561ecd87c9/english_short_latest.csv",
        "methodologyUrl": "https://github.com/huggingface/open_asr_leaderboard"
      },
      "measure": "Word error rate (WER), %; lower is better. Counts substituted, missing, and extra words.",
      "comparisonRule": "Compare only the AMI-Cleaned English test in this September 4, 2026 snapshot. Keep the model version and publisher scoring protocol fixed.",
      "limitations": [
        "A selected open-weight comparison, not the full leaderboard or a top-ten list.",
        "Inference pipelines differ. The published rows do not identify exact run dates, decoding settings, checkpoint revisions, or execution records.",
        "Short speech clips do not test whole-meeting speaker attribution, punctuation quality, other languages, or live response time.",
        "No uncertainty is published; small differences do not establish a reliable winner. WER can exceed 100% when extra words are inserted."
      ],
      "coverage": "charted",
      "tags": [
        "audio",
        "speech",
        "transcription",
        "ASR",
        "WER",
        "English",
        "meetings",
        "AMI",
        "open weights",
        "Whisper",
        "Parakeet",
        "Cohere",
        "Qwen",
        "Granite",
        "Voxtral"
      ],
      "explorationUrl": "https://aicharts.io/benchmarks?atlas=open-asr-ami-cleaned#explore",
      "dataset": {
        "url": "https://aicharts.io/data/benchmark-atlas/open-asr-ami-cleaned",
        "configurationCount": 10,
        "score": {
          "label": "Word error rate",
          "unit": "% WER",
          "direction": "lower",
          "minimum": 0
        },
        "source": {
          "name": "Hugging Face Open ASR Leaderboard",
          "url": "https://huggingface.co/datasets/hf-audio/open-asr-leaderboard-results/blob/ba5712d5ace8f785fa0daae1aecea8561ecd87c9/english_short_latest.csv",
          "revision": "ba5712d5ace8f785fa0daae1aecea8561ecd87c9",
          "retrievedAt": "2026-09-09T01:53:06.273Z"
        },
        "observedAt": "2026-09-04",
        "evidenceLabel": "Open ASR owner-reported · selected open-weight configurations"
      }
    }
  ]
}
