{
  "schemaVersion": 1,
  "benchmark": {
    "id": "deepresearch-bench-ii",
    "name": "DeepResearch Bench II",
    "version": "II · 132 tasks",
    "category": "research",
    "question": "Which research agent produces a useful report?",
    "summary": "Compare information gathering, analysis, and presentation against expert-written research rubrics.",
    "source": {
      "name": "USTC / Metastone",
      "url": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
      "methodologyUrl": "https://arxiv.org/abs/2601.08536"
    },
    "measure": "Owner-reported weighted rubric score across 132 tasks and 9,430 criteria; component scores appear in the inspector.",
    "comparisonRule": "Keep the exact research product and model generation. An o3-era research run does not represent today's OpenAI product.",
    "limitations": [
      "The source includes historical products and does not date every run.",
      "A model judge evaluates reports; no intervals or comparable costs are published in this table.",
      "This is a selected set of nine systems, not the full source leaderboard."
    ],
    "coverage": "charted",
    "tags": [
      "deep research",
      "reports",
      "citations",
      "analysis",
      "search"
    ]
  },
  "dataset": {
    "benchmarkId": "deepresearch-bench-ii",
    "version": "II · 132 tasks",
    "score": {
      "label": "Weighted rubric score",
      "unit": "%",
      "direction": "higher",
      "minimum": 0,
      "maximum": 100
    },
    "source": {
      "name": "USTC / Metastone · selected published observations",
      "url": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
      "retrievedAt": "2026-09-09T00:19:36.880Z",
      "revision": "SHA-256 ba0ccb0a89ff6c8f5d3c97861b072cdb8459c27fdaef2db90e725537f2869562"
    },
    "evidenceLabel": "Benchmark-owner table · mixed historical product versions",
    "configurationLabel": "Research product as named by the source",
    "comparabilityNote": "Nine selected systems. The table includes older products; run dates, exact model versions, costs, and confidence intervals are not consistently supplied.",
    "points": [
      {
        "id": "ai21-deepresearch",
        "label": "AI21-DeepResearch",
        "model": "AI21-DeepResearch",
        "provider": "AI21 Labs",
        "harness": "Research product's own agent",
        "effort": null,
        "score": 64.38,
        "costUsd": null,
        "uncertainty": null,
        "sourceUrl": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "details": [
          {
            "label": "Information recall",
            "value": "60.35%"
          },
          {
            "label": "Analysis",
            "value": "71%"
          },
          {
            "label": "Presentation",
            "value": "92.89%"
          },
          {
            "label": "Run date",
            "value": "Not consistently published"
          }
        ]
      },
      {
        "id": "dalpha-deepresearch",
        "label": "Dalpha DeepResearch",
        "model": "Dalpha DeepResearch",
        "provider": "Dalpha",
        "harness": "Research product's own agent",
        "effort": null,
        "score": 61.01,
        "costUsd": null,
        "uncertainty": null,
        "sourceUrl": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "details": [
          {
            "label": "Information recall",
            "value": "58.62%"
          },
          {
            "label": "Analysis",
            "value": "61.36%"
          },
          {
            "label": "Presentation",
            "value": "93.41%"
          },
          {
            "label": "Run date",
            "value": "Not consistently published"
          }
        ]
      },
      {
        "id": "gemini-3-pro-deep-research",
        "label": "Gemini-3-Pro Deep Research",
        "model": "Gemini-3-Pro Deep Research",
        "provider": "Google",
        "harness": "Research product's own agent",
        "effort": null,
        "score": 44.6,
        "costUsd": null,
        "uncertainty": null,
        "sourceUrl": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "details": [
          {
            "label": "Information recall",
            "value": "39.09%"
          },
          {
            "label": "Analysis",
            "value": "48.94%"
          },
          {
            "label": "Presentation",
            "value": "91.85%"
          },
          {
            "label": "Run date",
            "value": "Not consistently published"
          }
        ]
      },
      {
        "id": "iflow-researcher",
        "label": "iFlow-Researcher",
        "model": "iFlow-Researcher",
        "provider": "NJU&Alibaba",
        "harness": "Research product's own agent",
        "effort": null,
        "score": 59.91,
        "costUsd": null,
        "uncertainty": null,
        "sourceUrl": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "details": [
          {
            "label": "Information recall",
            "value": "54.99%"
          },
          {
            "label": "Analysis",
            "value": "69.54%"
          },
          {
            "label": "Presentation",
            "value": "92.56%"
          },
          {
            "label": "Run date",
            "value": "Not consistently published"
          }
        ]
      },
      {
        "id": "nvidia-aiq-nemotron-3-opus-4-6",
        "label": "nvidia-aiq (Nemotron 3, Opus 4.6)",
        "model": "nvidia-aiq (Nemotron 3, Opus 4.6)",
        "provider": "NVIDIA",
        "harness": "Research product's own agent",
        "effort": null,
        "score": 54.5,
        "costUsd": null,
        "uncertainty": null,
        "sourceUrl": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "details": [
          {
            "label": "Information recall",
            "value": "49.23%"
          },
          {
            "label": "Analysis",
            "value": "61.55%"
          },
          {
            "label": "Presentation",
            "value": "93.15%"
          },
          {
            "label": "Run date",
            "value": "Not consistently published"
          }
        ]
      },
      {
        "id": "openai-gpt-o3-deep-research",
        "label": "OpenAI-GPT-o3 Deep Research",
        "model": "OpenAI-GPT-o3 Deep Research",
        "provider": "OpenAI",
        "harness": "Research product's own agent",
        "effort": null,
        "score": 45.4,
        "costUsd": null,
        "uncertainty": null,
        "sourceUrl": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "details": [
          {
            "label": "Information recall",
            "value": "39.98%"
          },
          {
            "label": "Analysis",
            "value": "49.85%"
          },
          {
            "label": "Presentation",
            "value": "89.16%"
          },
          {
            "label": "Run date",
            "value": "Not consistently published"
          }
        ]
      },
      {
        "id": "perplexity-research",
        "label": "Perplexity Research",
        "model": "Perplexity Research",
        "provider": "Perplexity AI",
        "harness": "Research product's own agent",
        "effort": null,
        "score": 38.58,
        "costUsd": null,
        "uncertainty": null,
        "sourceUrl": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "details": [
          {
            "label": "Information recall",
            "value": "33.05%"
          },
          {
            "label": "Analysis",
            "value": "44.47%"
          },
          {
            "label": "Presentation",
            "value": "79.34%"
          },
          {
            "label": "Run date",
            "value": "Not consistently published"
          }
        ]
      },
      {
        "id": "tongyi-deep-research",
        "label": "Tongyi Deep Research",
        "model": "Tongyi Deep Research",
        "provider": "Alibaba",
        "harness": "Research product's own agent",
        "effort": null,
        "score": 29.89,
        "costUsd": null,
        "uncertainty": null,
        "sourceUrl": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "details": [
          {
            "label": "Information recall",
            "value": "22.95%"
          },
          {
            "label": "Analysis",
            "value": "35.89%"
          },
          {
            "label": "Presentation",
            "value": "86.13%"
          },
          {
            "label": "Run date",
            "value": "Not consistently published"
          }
        ]
      },
      {
        "id": "xiaoyi-deepresearch-6-0",
        "label": "Xiaoyi DeepResearch 6.0",
        "model": "Xiaoyi DeepResearch 6.0",
        "provider": "Huawei",
        "harness": "Research product's own agent",
        "effort": null,
        "score": 58.72,
        "costUsd": null,
        "uncertainty": null,
        "sourceUrl": "https://agentresearchlab.com/benchmarks/deepresearch-bench-ii/index.html",
        "details": [
          {
            "label": "Information recall",
            "value": "53.05%"
          },
          {
            "label": "Analysis",
            "value": "69.9%"
          },
          {
            "label": "Presentation",
            "value": "91.12%"
          },
          {
            "label": "Run date",
            "value": "Not consistently published"
          }
        ]
      }
    ]
  }
}
