{
  "schemaVersion": 1,
  "benchmark": {
    "id": "vals-tax-agent",
    "name": "Tax Agent Bench · Vals AI",
    "version": "1 · private test set",
    "category": "work",
    "question": "Can an agent answer a research-grade tax question?",
    "summary": "US corporate tax questions covering fact patterns, rule lookup, calculations, forms, and controversy.",
    "source": {
      "name": "Vals AI",
      "url": "https://www.vals.ai/benchmarks/tax_agent_bench",
      "methodologyUrl": "https://www.vals.ai/about"
    },
    "measure": "Accuracy across six tax task families, with a stricter all-pass variant reported alongside the headline score.",
    "comparisonRule": "Compare within Tax Agent Bench version 1. This board covers 22 configurations, fewer than the other Vals boards.",
    "limitations": [
      "The test set is private, so no independent party can reproduce these scores.",
      "US corporate tax only, and tax rules change with the filing year.",
      "A score here is not tax advice and does not establish that an answer is filing-ready.",
      "Model names are the identifiers Vals publishes, not aicharts canonical names."
    ],
    "coverage": "charted",
    "tags": [
      "tax",
      "research",
      "agentic",
      "compliance",
      "professional work",
      "Vals AI",
      "independent evaluator"
    ]
  },
  "dataset": {
    "benchmarkId": "vals-tax-agent",
    "version": "1 · private test set",
    "score": {
      "label": "Accuracy",
      "unit": "%",
      "direction": "higher",
      "minimum": 0,
      "maximum": 100
    },
    "source": {
      "name": "Vals AI · Tax Agent Bench",
      "url": "https://www.vals.ai/benchmarks/tax_agent_bench",
      "retrievedAt": "2026-09-28T18:16:23.422Z",
      "revision": "tax_agent_bench v1"
    },
    "observedAt": "2026-09-27T00:00:00.000Z",
    "evidenceLabel": "Independent evaluator · private test set",
    "configurationLabel": "Model at the effort Vals selected",
    "comparabilityNote": "Private evaluation run by Vals. Dollar values are the publisher's average cost per test. Model names are the identifiers Vals publishes, not aicharts canonical names.",
    "costLabel": "Average cost per test (USD)",
    "points": [
      {
        "id": "alibaba-qwen3-6-plus",
        "label": "alibaba/qwen3.6-plus",
        "model": "alibaba/qwen3.6-plus",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 32.844,
        "costUsd": 0.298017,
        "uncertainty": {
          "lower": 30.092000000000002,
          "upper": 35.596000000000004,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.6-plus"
          },
          {
            "label": "Standard error",
            "value": "±2.75 points"
          },
          {
            "label": "Mean latency",
            "value": "378 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "8.8%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "32.0%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "32.2%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "30.5%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "49.1%"
          },
          {
            "label": "Forms & Filings",
            "value": "31.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "22.7%"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-7-max",
        "label": "alibaba/qwen3.7-max",
        "model": "alibaba/qwen3.7-max",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 47.407,
        "costUsd": 0.435494,
        "uncertainty": {
          "lower": 44.169999999999995,
          "upper": 50.644,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.7-max"
          },
          {
            "label": "Standard error",
            "value": "±3.24 points"
          },
          {
            "label": "Mean latency",
            "value": "210 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "16.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "42.2%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "47.5%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "49.7%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "60.6%"
          },
          {
            "label": "Forms & Filings",
            "value": "46.2%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "40.4%"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-7-plus",
        "label": "alibaba/qwen3.7-plus",
        "model": "alibaba/qwen3.7-plus",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 38.713,
        "costUsd": 0.093885,
        "uncertainty": {
          "lower": 35.903,
          "upper": 41.523,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.7-plus"
          },
          {
            "label": "Standard error",
            "value": "±2.81 points"
          },
          {
            "label": "Mean latency",
            "value": "310 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "13.0%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "28.1%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "44.2%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "46.1%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "52.1%"
          },
          {
            "label": "Forms & Filings",
            "value": "30.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "36.3%"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-8-27b",
        "label": "alibaba/qwen3.8-27b",
        "model": "alibaba/qwen3.8-27b",
        "provider": "Alibaba",
        "harness": null,
        "effort": "xhigh",
        "score": 60.038,
        "costUsd": 0.81533,
        "uncertainty": {
          "lower": 56.757999999999996,
          "upper": 63.318,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.8-27b"
          },
          {
            "label": "Standard error",
            "value": "±3.28 points"
          },
          {
            "label": "Mean latency",
            "value": "1,334 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "31.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "57.7%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "68.0%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "58.6%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "66.2%"
          },
          {
            "label": "Forms & Filings",
            "value": "62.4%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "48.3%"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-8-max",
        "label": "alibaba/qwen3.8-max",
        "model": "alibaba/qwen3.8-max",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 65.96,
        "costUsd": 1.552558,
        "uncertainty": {
          "lower": 62.778999999999996,
          "upper": 69.14099999999999,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.8-max"
          },
          {
            "label": "Standard error",
            "value": "±3.18 points"
          },
          {
            "label": "Mean latency",
            "value": "2,436 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "32.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "62.1%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "60.8%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "71.4%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "75.6%"
          },
          {
            "label": "Forms & Filings",
            "value": "64.9%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "62.6%"
          }
        ]
      },
      {
        "id": "ant-ling-3-0-flash-af-rc3",
        "label": "ant/ling-3.0-flash-af-rc3",
        "model": "ant/ling-3.0-flash-af-rc3",
        "provider": "Ant",
        "harness": null,
        "effort": null,
        "score": 45.697,
        "costUsd": 0.025079,
        "uncertainty": {
          "lower": 42.379000000000005,
          "upper": 49.015,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "ant/ling-3.0-flash-af-rc3"
          },
          {
            "label": "Standard error",
            "value": "±3.32 points"
          },
          {
            "label": "Mean latency",
            "value": "334 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "20.7%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "33.0%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "53.8%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "51.0%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "66.6%"
          },
          {
            "label": "Forms & Filings",
            "value": "43.0%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "32.3%"
          }
        ]
      },
      {
        "id": "anthropic-claude-fable-5",
        "label": "anthropic/claude-fable-5",
        "model": "anthropic/claude-fable-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 69.822,
        "costUsd": 6.696087,
        "uncertainty": {
          "lower": 66.70400000000001,
          "upper": 72.94,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-fable-5"
          },
          {
            "label": "Standard error",
            "value": "±3.12 points"
          },
          {
            "label": "Mean latency",
            "value": "867 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "38.3%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "74.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "55.9%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "71.0%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "83.4%"
          },
          {
            "label": "Forms & Filings",
            "value": "64.0%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "68.4%"
          }
        ]
      },
      {
        "id": "anthropic-claude-fable-5-1",
        "label": "anthropic/claude-fable-5-1",
        "model": "anthropic/claude-fable-5-1",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 77.642,
        "costUsd": 13.180073,
        "uncertainty": {
          "lower": 74.807,
          "upper": 80.47699999999999,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-fable-5-1"
          },
          {
            "label": "Standard error",
            "value": "±2.83 points"
          },
          {
            "label": "Mean latency",
            "value": "1,714 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "49.2%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "79.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "72.9%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "81.2%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "88.0%"
          },
          {
            "label": "Forms & Filings",
            "value": "73.2%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "70.6%"
          }
        ]
      },
      {
        "id": "anthropic-claude-haiku-4-5-20251001-thinking",
        "label": "anthropic/claude-haiku-4-5-20251001-thinking",
        "model": "anthropic/claude-haiku-4-5-20251001-thinking",
        "provider": "Anthropic",
        "harness": null,
        "effort": null,
        "score": 28.314,
        "costUsd": 0.226416,
        "uncertainty": {
          "lower": 25.762,
          "upper": 30.866,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-haiku-4-5-20251001-thinking"
          },
          {
            "label": "Standard error",
            "value": "±2.55 points"
          },
          {
            "label": "Mean latency",
            "value": "155 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "7.3%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "22.9%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "27.0%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "25.4%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "50.4%"
          },
          {
            "label": "Forms & Filings",
            "value": "25.8%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "20.8%"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-7",
        "label": "anthropic/claude-opus-4-7",
        "model": "anthropic/claude-opus-4-7",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 63.864,
        "costUsd": 1.274806,
        "uncertainty": {
          "lower": 60.727,
          "upper": 67.00099999999999,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-7"
          },
          {
            "label": "Standard error",
            "value": "±3.14 points"
          },
          {
            "label": "Mean latency",
            "value": "279 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "23.8%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "61.6%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "61.0%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "66.5%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "77.0%"
          },
          {
            "label": "Forms & Filings",
            "value": "60.4%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "57.6%"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-8",
        "label": "anthropic/claude-opus-4-8",
        "model": "anthropic/claude-opus-4-8",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 64.906,
        "costUsd": 1.68964,
        "uncertainty": {
          "lower": 61.732000000000006,
          "upper": 68.08000000000001,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-8"
          },
          {
            "label": "Standard error",
            "value": "±3.17 points"
          },
          {
            "label": "Mean latency",
            "value": "479 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "30.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "65.4%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "54.7%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "76.0%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "78.6%"
          },
          {
            "label": "Forms & Filings",
            "value": "55.8%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "58.7%"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-5",
        "label": "anthropic/claude-opus-5",
        "model": "anthropic/claude-opus-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 75.06,
        "costUsd": 5.136554,
        "uncertainty": {
          "lower": 72.104,
          "upper": 78.016,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-5"
          },
          {
            "label": "Standard error",
            "value": "±2.96 points"
          },
          {
            "label": "Mean latency",
            "value": "1,008 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "45.6%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "77.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "78.0%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "71.6%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "85.2%"
          },
          {
            "label": "Forms & Filings",
            "value": "67.2%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "70.0%"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-5-5",
        "label": "anthropic/claude-opus-5-5",
        "model": "anthropic/claude-opus-5-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 70.499,
        "costUsd": 15.086184,
        "uncertainty": {
          "lower": 67.354,
          "upper": 73.64399999999999,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-5-5"
          },
          {
            "label": "Standard error",
            "value": "±3.15 points"
          },
          {
            "label": "Mean latency",
            "value": "4,013 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "45.6%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "72.9%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "56.2%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "73.3%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "88.9%"
          },
          {
            "label": "Forms & Filings",
            "value": "63.8%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "66.9%"
          }
        ]
      },
      {
        "id": "anthropic-claude-sonnet-4-6",
        "label": "anthropic/claude-sonnet-4-6",
        "model": "anthropic/claude-sonnet-4-6",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 64.854,
        "costUsd": 1.193501,
        "uncertainty": {
          "lower": 61.744,
          "upper": 67.964,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-sonnet-4-6"
          },
          {
            "label": "Standard error",
            "value": "±3.11 points"
          },
          {
            "label": "Mean latency",
            "value": "707 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "25.9%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "56.7%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "65.9%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "66.1%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "80.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "65.6%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "57.9%"
          }
        ]
      },
      {
        "id": "anthropic-claude-sonnet-5",
        "label": "anthropic/claude-sonnet-5",
        "model": "anthropic/claude-sonnet-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 62.273,
        "costUsd": 1.794774,
        "uncertainty": {
          "lower": 59.087,
          "upper": 65.459,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-sonnet-5"
          },
          {
            "label": "Standard error",
            "value": "±3.19 points"
          },
          {
            "label": "Mean latency",
            "value": "1,159 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "30.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "59.9%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "61.5%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "60.0%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "70.6%"
          },
          {
            "label": "Forms & Filings",
            "value": "65.2%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "57.4%"
          }
        ]
      },
      {
        "id": "anthropic-claude-sonnet-5-5",
        "label": "anthropic/claude-sonnet-5-5",
        "model": "anthropic/claude-sonnet-5-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 73.395,
        "costUsd": 9.245466,
        "uncertainty": {
          "lower": 70.414,
          "upper": 76.37599999999999,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-sonnet-5-5"
          },
          {
            "label": "Standard error",
            "value": "±2.98 points"
          },
          {
            "label": "Mean latency",
            "value": "3,561 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "42.0%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "73.9%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "67.8%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "77.2%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "85.4%"
          },
          {
            "label": "Forms & Filings",
            "value": "70.7%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "65.2%"
          }
        ]
      },
      {
        "id": "deepseek-deepseek-v4-1-flash",
        "label": "deepseek/deepseek-v4.1-flash",
        "model": "deepseek/deepseek-v4.1-flash",
        "provider": "DeepSeek",
        "harness": null,
        "effort": "high",
        "score": 62.458,
        "costUsd": 0.134662,
        "uncertainty": {
          "lower": 59.303,
          "upper": 65.613,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "deepseek/deepseek-v4.1-flash"
          },
          {
            "label": "Standard error",
            "value": "±3.15 points"
          },
          {
            "label": "Mean latency",
            "value": "349 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "27.5%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "60.6%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "56.9%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "71.9%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "76.6%"
          },
          {
            "label": "Forms & Filings",
            "value": "56.8%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "52.8%"
          }
        ]
      },
      {
        "id": "deepseek-deepseek-v4-flash-0731",
        "label": "deepseek/deepseek-v4-flash-0731",
        "model": "deepseek/deepseek-v4-flash-0731",
        "provider": "DeepSeek",
        "harness": null,
        "effort": "high",
        "score": 62.563,
        "costUsd": 0.156388,
        "uncertainty": {
          "lower": 59.427,
          "upper": 65.699,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "deepseek/deepseek-v4-flash-0731"
          },
          {
            "label": "Standard error",
            "value": "±3.14 points"
          },
          {
            "label": "Mean latency",
            "value": "337 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "29.5%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "54.7%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "70.2%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "70.5%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "78.0%"
          },
          {
            "label": "Forms & Filings",
            "value": "59.4%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "45.8%"
          }
        ]
      },
      {
        "id": "deepseek-deepseek-v4-pro",
        "label": "deepseek/deepseek-v4-pro",
        "model": "deepseek/deepseek-v4-pro",
        "provider": "DeepSeek",
        "harness": null,
        "effort": "max",
        "score": 58.499,
        "costUsd": 0.719152,
        "uncertainty": {
          "lower": 55.367000000000004,
          "upper": 61.631,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "deepseek/deepseek-v4-pro"
          },
          {
            "label": "Standard error",
            "value": "±3.13 points"
          },
          {
            "label": "Mean latency",
            "value": "1,245 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "26.4%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "55.6%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "55.1%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "65.7%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "62.9%"
          },
          {
            "label": "Forms & Filings",
            "value": "58.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "54.8%"
          }
        ]
      },
      {
        "id": "deepseek-deepseek-v4-pro-0813",
        "label": "deepseek/deepseek-v4-pro-0813",
        "model": "deepseek/deepseek-v4-pro-0813",
        "provider": "DeepSeek",
        "harness": null,
        "effort": "max",
        "score": 58.663,
        "costUsd": 0.729136,
        "uncertainty": {
          "lower": 55.525,
          "upper": 61.800999999999995,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "deepseek/deepseek-v4-pro-0813"
          },
          {
            "label": "Standard error",
            "value": "±3.14 points"
          },
          {
            "label": "Mean latency",
            "value": "1,597 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "28.5%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "59.7%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "49.8%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "67.9%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "70.2%"
          },
          {
            "label": "Forms & Filings",
            "value": "52.4%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "51.5%"
          }
        ]
      },
      {
        "id": "fireworks-nemotron-lightning-3p5-30b-a3b",
        "label": "fireworks/nemotron-lightning-3p5-30b-a3b",
        "model": "fireworks/nemotron-lightning-3p5-30b-a3b",
        "provider": "Fireworks AI",
        "harness": null,
        "effort": null,
        "score": 13.002,
        "costUsd": 0.030533,
        "uncertainty": {
          "lower": 10.953000000000001,
          "upper": 15.051,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "fireworks/nemotron-lightning-3p5-30b-a3b"
          },
          {
            "label": "Standard error",
            "value": "±2.05 points"
          },
          {
            "label": "Mean latency",
            "value": "1,180 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "2.6%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "6.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "14.7%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "10.5%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "31.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "6.7%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "11.2%"
          }
        ]
      },
      {
        "id": "google-gemini-3-1-flash-lite-preview",
        "label": "google/gemini-3.1-flash-lite-preview",
        "model": "google/gemini-3.1-flash-lite-preview",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 20.085,
        "costUsd": 0.014308,
        "uncertainty": {
          "lower": 17.628,
          "upper": 22.542,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.1-flash-lite-preview"
          },
          {
            "label": "Standard error",
            "value": "±2.46 points"
          },
          {
            "label": "Mean latency",
            "value": "19 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "3.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "21.7%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "20.7%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "12.3%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "39.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "14.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "11.5%"
          }
        ]
      },
      {
        "id": "google-gemini-3-1-pro-preview",
        "label": "google/gemini-3.1-pro-preview",
        "model": "google/gemini-3.1-pro-preview",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 37.842,
        "costUsd": 0.301555,
        "uncertainty": {
          "lower": 34.744,
          "upper": 40.94,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.1-pro-preview"
          },
          {
            "label": "Standard error",
            "value": "±3.10 points"
          },
          {
            "label": "Mean latency",
            "value": "101 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "9.3%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "32.9%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "28.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "36.5%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "66.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "30.7%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "34.4%"
          }
        ]
      },
      {
        "id": "google-gemini-3-5-flash",
        "label": "google/gemini-3.5-flash",
        "model": "google/gemini-3.5-flash",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 53.526,
        "costUsd": 0.859679,
        "uncertainty": {
          "lower": 50.224000000000004,
          "upper": 56.828,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.5-flash"
          },
          {
            "label": "Standard error",
            "value": "±3.30 points"
          },
          {
            "label": "Mean latency",
            "value": "233 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "21.8%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "58.1%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "46.8%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "61.6%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "72.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "45.9%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "34.2%"
          }
        ]
      },
      {
        "id": "google-gemini-3-5-flash-lite",
        "label": "google/gemini-3.5-flash-lite",
        "model": "google/gemini-3.5-flash-lite",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 37.451,
        "costUsd": 0.081536,
        "uncertainty": {
          "lower": 34.336,
          "upper": 40.566,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.5-flash-lite"
          },
          {
            "label": "Standard error",
            "value": "±3.12 points"
          },
          {
            "label": "Mean latency",
            "value": "47 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "7.3%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "32.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "34.0%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "24.9%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "69.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "31.7%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "34.6%"
          }
        ]
      },
      {
        "id": "google-gemini-3-6-flash",
        "label": "google/gemini-3.6-flash",
        "model": "google/gemini-3.6-flash",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 49.895,
        "costUsd": 0.356293,
        "uncertainty": {
          "lower": 46.593,
          "upper": 53.197,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.6-flash"
          },
          {
            "label": "Standard error",
            "value": "±3.30 points"
          },
          {
            "label": "Mean latency",
            "value": "105 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "18.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "49.8%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "43.6%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "48.6%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "77.9%"
          },
          {
            "label": "Forms & Filings",
            "value": "42.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "37.4%"
          }
        ]
      },
      {
        "id": "google-gemini-3-7-flash",
        "label": "google/gemini-3.7-flash",
        "model": "google/gemini-3.7-flash",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 57.665,
        "costUsd": 0.483978,
        "uncertainty": {
          "lower": 54.352,
          "upper": 60.978,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.7-flash"
          },
          {
            "label": "Standard error",
            "value": "±3.31 points"
          },
          {
            "label": "Mean latency",
            "value": "87 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "22.8%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "49.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "57.9%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "68.8%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "78.8%"
          },
          {
            "label": "Forms & Filings",
            "value": "48.8%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "46.0%"
          }
        ]
      },
      {
        "id": "google-gemini-3-8-flash",
        "label": "google/gemini-3.8-flash",
        "model": "google/gemini-3.8-flash",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 66.771,
        "costUsd": 0.857214,
        "uncertainty": {
          "lower": 63.623,
          "upper": 69.919,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.8-flash"
          },
          {
            "label": "Standard error",
            "value": "±3.15 points"
          },
          {
            "label": "Mean latency",
            "value": "178 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "32.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "68.0%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "60.2%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "64.8%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "83.1%"
          },
          {
            "label": "Forms & Filings",
            "value": "65.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "59.0%"
          }
        ]
      },
      {
        "id": "google-gemini-3-flash-preview",
        "label": "google/gemini-3-flash-preview",
        "model": "google/gemini-3-flash-preview",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 37.557,
        "costUsd": 0.100699,
        "uncertainty": {
          "lower": 34.427,
          "upper": 40.687000000000005,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3-flash-preview"
          },
          {
            "label": "Standard error",
            "value": "±3.13 points"
          },
          {
            "label": "Mean latency",
            "value": "101 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "11.9%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "38.8%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "29.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "31.9%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "67.1%"
          },
          {
            "label": "Forms & Filings",
            "value": "27.5%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "30.1%"
          }
        ]
      },
      {
        "id": "grok-grok-4-20-0309-reasoning",
        "label": "grok/grok-4.20-0309-reasoning",
        "model": "grok/grok-4.20-0309-reasoning",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": null,
        "score": 31.241,
        "costUsd": 0.119262,
        "uncertainty": {
          "lower": 28.537,
          "upper": 33.945,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.20-0309-reasoning"
          },
          {
            "label": "Standard error",
            "value": "±2.70 points"
          },
          {
            "label": "Mean latency",
            "value": "89 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "7.8%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "25.1%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "24.8%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "37.7%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "48.8%"
          },
          {
            "label": "Forms & Filings",
            "value": "28.8%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "24.9%"
          }
        ]
      },
      {
        "id": "grok-grok-4-3",
        "label": "grok/grok-4.3",
        "model": "grok/grok-4.3",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": "high",
        "score": 34.416,
        "costUsd": 0.116616,
        "uncertainty": {
          "lower": 31.421999999999997,
          "upper": 37.41,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.3"
          },
          {
            "label": "Standard error",
            "value": "±2.99 points"
          },
          {
            "label": "Mean latency",
            "value": "98 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "6.2%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "36.7%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "26.1%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "34.6%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "51.4%"
          },
          {
            "label": "Forms & Filings",
            "value": "33.8%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "22.9%"
          }
        ]
      },
      {
        "id": "grok-grok-4-5",
        "label": "grok/grok-4.5",
        "model": "grok/grok-4.5",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": "high",
        "score": 64.323,
        "costUsd": 0.473604,
        "uncertainty": {
          "lower": 61.166999999999994,
          "upper": 67.479,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.5"
          },
          {
            "label": "Standard error",
            "value": "±3.16 points"
          },
          {
            "label": "Mean latency",
            "value": "284 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "24.9%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "59.5%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "60.6%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "70.6%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "72.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "64.2%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "60.7%"
          }
        ]
      },
      {
        "id": "grok-grok-4-6",
        "label": "grok/grok-4.6",
        "model": "grok/grok-4.6",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": "high",
        "score": 70.788,
        "costUsd": 0.978328,
        "uncertainty": {
          "lower": 67.783,
          "upper": 73.79299999999999,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.6"
          },
          {
            "label": "Standard error",
            "value": "±3.00 points"
          },
          {
            "label": "Mean latency",
            "value": "658 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "33.2%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "66.0%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "67.5%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "73.4%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "79.8%"
          },
          {
            "label": "Forms & Filings",
            "value": "72.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "68.1%"
          }
        ]
      },
      {
        "id": "grok-grok-4-7",
        "label": "grok/grok-4.7",
        "model": "grok/grok-4.7",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": "xhigh",
        "score": 65.596,
        "costUsd": 1.796539,
        "uncertainty": {
          "lower": 62.358000000000004,
          "upper": 68.834,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.7"
          },
          {
            "label": "Standard error",
            "value": "±3.24 points"
          },
          {
            "label": "Mean latency",
            "value": "996 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "34.2%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "64.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "52.2%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "74.5%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "76.7%"
          },
          {
            "label": "Forms & Filings",
            "value": "63.8%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "62.7%"
          }
        ]
      },
      {
        "id": "inception-mercury-2-5",
        "label": "inception/mercury-2.5",
        "model": "inception/mercury-2.5",
        "provider": "Inception",
        "harness": null,
        "effort": "high",
        "score": 12.773,
        "costUsd": 0.023633,
        "uncertainty": {
          "lower": 10.945,
          "upper": 14.600999999999999,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "inception/mercury-2.5"
          },
          {
            "label": "Standard error",
            "value": "±1.83 points"
          },
          {
            "label": "Mean latency",
            "value": "30 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "3.6%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "11.7%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "8.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "13.2%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "31.1%"
          },
          {
            "label": "Forms & Filings",
            "value": "6.9%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "5.9%"
          }
        ]
      },
      {
        "id": "kimi-kimi-k2-6",
        "label": "kimi/kimi-k2.6",
        "model": "kimi/kimi-k2.6",
        "provider": "Moonshot AI",
        "harness": null,
        "effort": null,
        "score": 45.135,
        "costUsd": 0.565511,
        "uncertainty": {
          "lower": 41.91,
          "upper": 48.36,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "kimi/kimi-k2.6"
          },
          {
            "label": "Standard error",
            "value": "±3.23 points"
          },
          {
            "label": "Mean latency",
            "value": "924 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "12.4%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "46.7%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "30.4%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "51.2%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "61.8%"
          },
          {
            "label": "Forms & Filings",
            "value": "37.6%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "42.3%"
          }
        ]
      },
      {
        "id": "kimi-kimi-k3",
        "label": "kimi/kimi-k3",
        "model": "kimi/kimi-k3",
        "provider": "Moonshot AI",
        "harness": null,
        "effort": "max",
        "score": 68.672,
        "costUsd": 2.787196,
        "uncertainty": {
          "lower": 65.589,
          "upper": 71.755,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "kimi/kimi-k3"
          },
          {
            "label": "Standard error",
            "value": "±3.08 points"
          },
          {
            "label": "Mean latency",
            "value": "2,415 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "33.7%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "62.2%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "58.4%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "72.9%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "84.0%"
          },
          {
            "label": "Forms & Filings",
            "value": "68.6%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "68.8%"
          }
        ]
      },
      {
        "id": "meta-muse-spark-1-2",
        "label": "meta/muse_spark_1_2",
        "model": "meta/muse_spark_1_2",
        "provider": "Meta",
        "harness": null,
        "effort": "xhigh",
        "score": 56.86,
        "costUsd": 0.269212,
        "uncertainty": {
          "lower": 54.542,
          "upper": 59.178,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "meta/muse_spark_1_2"
          },
          {
            "label": "Standard error",
            "value": "±2.32 points"
          },
          {
            "label": "Mean latency",
            "value": "120 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "43.5%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "52.5%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "55.1%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "55.2%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "76.3%"
          },
          {
            "label": "Forms & Filings",
            "value": "59.5%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "44.4%"
          }
        ]
      },
      {
        "id": "meta-muse-spark-1-3",
        "label": "meta/muse_spark_1_3",
        "model": "meta/muse_spark_1_3",
        "provider": "Meta",
        "harness": null,
        "effort": "xhigh",
        "score": 71.934,
        "costUsd": 0.319558,
        "uncertainty": {
          "lower": 69.017,
          "upper": 74.851,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "meta/muse_spark_1_3"
          },
          {
            "label": "Standard error",
            "value": "±2.92 points"
          },
          {
            "label": "Mean latency",
            "value": "505 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "36.8%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "75.1%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "67.7%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "73.1%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "81.8%"
          },
          {
            "label": "Forms & Filings",
            "value": "71.7%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "60.9%"
          }
        ]
      },
      {
        "id": "meta-muse-spark-1-3-max",
        "label": "meta/muse_spark_1_3_max",
        "model": "meta/muse_spark_1_3_max",
        "provider": "Meta",
        "harness": null,
        "effort": "max",
        "score": 72.441,
        "costUsd": 0.389917,
        "uncertainty": {
          "lower": 69.562,
          "upper": 75.32000000000001,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "meta/muse_spark_1_3_max"
          },
          {
            "label": "Standard error",
            "value": "±2.88 points"
          },
          {
            "label": "Mean latency",
            "value": "249 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "43.5%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "70.2%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "74.7%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "69.6%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "81.1%"
          },
          {
            "label": "Forms & Filings",
            "value": "71.4%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "68.7%"
          }
        ]
      },
      {
        "id": "minimax-minimax-m2-7",
        "label": "minimax/MiniMax-M2.7",
        "model": "minimax/MiniMax-M2.7",
        "provider": "MiniMax",
        "harness": null,
        "effort": null,
        "score": 25.382,
        "costUsd": 0.120058,
        "uncertainty": {
          "lower": 22.792,
          "upper": 27.972,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "minimax/MiniMax-M2.7"
          },
          {
            "label": "Standard error",
            "value": "±2.59 points"
          },
          {
            "label": "Mean latency",
            "value": "303 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "8.3%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "20.2%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "25.2%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "21.5%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "47.2%"
          },
          {
            "label": "Forms & Filings",
            "value": "19.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "21.3%"
          }
        ]
      },
      {
        "id": "minimax-minimax-m3",
        "label": "minimax/MiniMax-M3",
        "model": "minimax/MiniMax-M3",
        "provider": "MiniMax",
        "harness": null,
        "effort": null,
        "score": 49.691,
        "costUsd": 0.164063,
        "uncertainty": {
          "lower": 46.443000000000005,
          "upper": 52.939,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "minimax/MiniMax-M3"
          },
          {
            "label": "Standard error",
            "value": "±3.25 points"
          },
          {
            "label": "Mean latency",
            "value": "315 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "21.2%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "33.8%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "57.9%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "52.9%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "54.9%"
          },
          {
            "label": "Forms & Filings",
            "value": "53.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "52.4%"
          }
        ]
      },
      {
        "id": "mistralai-mistral-medium-3-5",
        "label": "mistralai/mistral-medium-3.5",
        "model": "mistralai/mistral-medium-3.5",
        "provider": "Mistral AI",
        "harness": null,
        "effort": "high",
        "score": 27.291,
        "costUsd": 1.443655,
        "uncertainty": {
          "lower": 24.467,
          "upper": 30.115000000000002,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "mistralai/mistral-medium-3.5"
          },
          {
            "label": "Standard error",
            "value": "±2.82 points"
          },
          {
            "label": "Mean latency",
            "value": "431 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "4.7%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "24.4%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "26.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "29.4%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "45.3%"
          },
          {
            "label": "Forms & Filings",
            "value": "21.3%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "18.4%"
          }
        ]
      },
      {
        "id": "openai-gpt-5-4-mini-2026-03-17",
        "label": "openai/gpt-5.4-mini-2026-03-17",
        "model": "openai/gpt-5.4-mini-2026-03-17",
        "provider": "OpenAI",
        "harness": null,
        "effort": "xhigh",
        "score": 36.586,
        "costUsd": 0.806171,
        "uncertainty": {
          "lower": 33.491,
          "upper": 39.681,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.4-mini-2026-03-17"
          },
          {
            "label": "Standard error",
            "value": "±3.10 points"
          },
          {
            "label": "Mean latency",
            "value": "1,006 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "8.8%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "36.5%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "26.5%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "38.6%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "52.3%"
          },
          {
            "label": "Forms & Filings",
            "value": "33.5%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "32.2%"
          }
        ]
      },
      {
        "id": "openai-gpt-5-4-nano-2026-03-17",
        "label": "openai/gpt-5.4-nano-2026-03-17",
        "model": "openai/gpt-5.4-nano-2026-03-17",
        "provider": "OpenAI",
        "harness": null,
        "effort": "high",
        "score": 26.578,
        "costUsd": 0.069662,
        "uncertainty": {
          "lower": 24.06,
          "upper": 29.096,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.4-nano-2026-03-17"
          },
          {
            "label": "Standard error",
            "value": "±2.52 points"
          },
          {
            "label": "Mean latency",
            "value": "220 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "6.2%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "23.4%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "23.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "22.5%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "42.8%"
          },
          {
            "label": "Forms & Filings",
            "value": "22.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "26.8%"
          }
        ]
      },
      {
        "id": "openai-gpt-5-5",
        "label": "openai/gpt-5.5",
        "model": "openai/gpt-5.5",
        "provider": "OpenAI",
        "harness": null,
        "effort": "xhigh",
        "score": 60.456,
        "costUsd": 4.073277,
        "uncertainty": {
          "lower": 57.24,
          "upper": 63.672000000000004,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.5"
          },
          {
            "label": "Standard error",
            "value": "±3.22 points"
          },
          {
            "label": "Mean latency",
            "value": "952 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "21.8%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "59.2%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "44.2%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "68.2%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "76.8%"
          },
          {
            "label": "Forms & Filings",
            "value": "59.9%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "54.9%"
          }
        ]
      },
      {
        "id": "openai-gpt-5-6-luna",
        "label": "openai/gpt-5.6-luna",
        "model": "openai/gpt-5.6-luna",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 60.815,
        "costUsd": 0.465359,
        "uncertainty": {
          "lower": 57.552,
          "upper": 64.078,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.6-luna"
          },
          {
            "label": "Standard error",
            "value": "±3.26 points"
          },
          {
            "label": "Mean latency",
            "value": "2,566 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "24.9%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "59.6%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "54.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "60.1%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "71.8%"
          },
          {
            "label": "Forms & Filings",
            "value": "58.5%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "61.2%"
          }
        ]
      },
      {
        "id": "openai-gpt-5-6-sol",
        "label": "openai/gpt-5.6-sol",
        "model": "openai/gpt-5.6-sol",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 67.955,
        "costUsd": 7.513965,
        "uncertainty": {
          "lower": 64.832,
          "upper": 71.078,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.6-sol"
          },
          {
            "label": "Standard error",
            "value": "±3.12 points"
          },
          {
            "label": "Mean latency",
            "value": "2,681 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "31.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "60.4%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "53.8%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "74.4%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "81.4%"
          },
          {
            "label": "Forms & Filings",
            "value": "72.5%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "68.5%"
          }
        ]
      },
      {
        "id": "openai-gpt-5-6-terra",
        "label": "openai/gpt-5.6-terra",
        "model": "openai/gpt-5.6-terra",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 65.199,
        "costUsd": 4.829225,
        "uncertainty": {
          "lower": 62.010999999999996,
          "upper": 68.387,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.6-terra"
          },
          {
            "label": "Standard error",
            "value": "±3.19 points"
          },
          {
            "label": "Mean latency",
            "value": "3,631 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "30.1%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "63.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "53.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "60.3%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "81.6%"
          },
          {
            "label": "Forms & Filings",
            "value": "68.7%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "64.8%"
          }
        ]
      },
      {
        "id": "openai-gpt-6-astra",
        "label": "openai/gpt-6-astra",
        "model": "openai/gpt-6-astra",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 63.335,
        "costUsd": 5.7963,
        "uncertainty": {
          "lower": 60.188,
          "upper": 66.482,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-6-astra"
          },
          {
            "label": "Standard error",
            "value": "±3.15 points"
          },
          {
            "label": "Mean latency",
            "value": "901 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "20.7%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "66.0%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "49.4%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "66.2%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "77.9%"
          },
          {
            "label": "Forms & Filings",
            "value": "60.8%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "58.6%"
          }
        ]
      },
      {
        "id": "openai-gpt-6-luna",
        "label": "openai/gpt-6-luna",
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 58.864,
        "costUsd": 0.190956,
        "uncertainty": {
          "lower": 55.653,
          "upper": 62.074999999999996,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-6-luna"
          },
          {
            "label": "Standard error",
            "value": "±3.21 points"
          },
          {
            "label": "Mean latency",
            "value": "1,996 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "20.2%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "60.2%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "45.9%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "55.3%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "73.8%"
          },
          {
            "label": "Forms & Filings",
            "value": "62.9%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "54.4%"
          }
        ]
      },
      {
        "id": "openai-gpt-6-sol",
        "label": "openai/gpt-6-sol",
        "model": "openai/gpt-6-sol",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 53.045,
        "costUsd": 2.277613,
        "uncertainty": {
          "lower": 49.755,
          "upper": 56.335,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-6-sol"
          },
          {
            "label": "Standard error",
            "value": "±3.29 points"
          },
          {
            "label": "Mean latency",
            "value": "1,052 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "15.0%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "49.9%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "39.8%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "56.1%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "69.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "53.7%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "50.7%"
          }
        ]
      },
      {
        "id": "tencent-hy4-preview",
        "label": "tencent/hy4-preview",
        "model": "tencent/hy4-preview",
        "provider": "Tencent",
        "harness": null,
        "effort": null,
        "score": 63.712,
        "costUsd": 0.585277,
        "uncertainty": {
          "lower": 60.452000000000005,
          "upper": 66.97200000000001,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "tencent/hy4-preview"
          },
          {
            "label": "Standard error",
            "value": "±3.26 points"
          },
          {
            "label": "Mean latency",
            "value": "2,464 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "32.6%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "55.8%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "57.5%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "63.7%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "81.0%"
          },
          {
            "label": "Forms & Filings",
            "value": "68.5%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "59.1%"
          }
        ]
      },
      {
        "id": "thinkingmachines-inkling",
        "label": "thinkingmachines/inkling",
        "model": "thinkingmachines/inkling",
        "provider": "Thinkingmachines",
        "harness": null,
        "effort": "0.99",
        "score": 43.516,
        "costUsd": 0.353757,
        "uncertainty": {
          "lower": 40.568999999999996,
          "upper": 46.463,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "thinkingmachines/inkling"
          },
          {
            "label": "Standard error",
            "value": "±2.95 points"
          },
          {
            "label": "Mean latency",
            "value": "811 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "17.6%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "44.1%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "31.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "34.4%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "61.9%"
          },
          {
            "label": "Forms & Filings",
            "value": "42.4%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "46.7%"
          }
        ]
      },
      {
        "id": "thinkingmachines-inkling-small",
        "label": "thinkingmachines/inkling-small",
        "model": "thinkingmachines/inkling-small",
        "provider": "Thinkingmachines",
        "harness": null,
        "effort": "0.99",
        "score": 41.461,
        "costUsd": 0.091767,
        "uncertainty": {
          "lower": 38.589999999999996,
          "upper": 44.332,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "thinkingmachines/inkling-small"
          },
          {
            "label": "Standard error",
            "value": "±2.87 points"
          },
          {
            "label": "Mean latency",
            "value": "385 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "18.7%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "43.9%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "32.1%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "47.5%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "51.2%"
          },
          {
            "label": "Forms & Filings",
            "value": "43.0%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "30.2%"
          }
        ]
      },
      {
        "id": "xiaomi-mimo-v2-5",
        "label": "xiaomi/mimo-v2.5",
        "model": "xiaomi/mimo-v2.5",
        "provider": "Xiaomi",
        "harness": null,
        "effort": null,
        "score": 29.163,
        "costUsd": 0.017023,
        "uncertainty": {
          "lower": 26.45,
          "upper": 31.876,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "xiaomi/mimo-v2.5"
          },
          {
            "label": "Standard error",
            "value": "±2.71 points"
          },
          {
            "label": "Mean latency",
            "value": "501 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "9.3%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "29.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "21.6%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "30.1%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "43.2%"
          },
          {
            "label": "Forms & Filings",
            "value": "19.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "31.7%"
          }
        ]
      },
      {
        "id": "xiaomi-mimo-v2-5-pro",
        "label": "xiaomi/mimo-v2.5-pro",
        "model": "xiaomi/mimo-v2.5-pro",
        "provider": "Xiaomi",
        "harness": null,
        "effort": null,
        "score": 34.787,
        "costUsd": 0.040042,
        "uncertainty": {
          "lower": 32.166,
          "upper": 37.408,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "xiaomi/mimo-v2.5-pro"
          },
          {
            "label": "Standard error",
            "value": "±2.62 points"
          },
          {
            "label": "Mean latency",
            "value": "377 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "15.5%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "23.2%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "41.1%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "42.2%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "43.1%"
          },
          {
            "label": "Forms & Filings",
            "value": "31.3%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "32.7%"
          }
        ]
      },
      {
        "id": "xiaomi-mimo-v2-6-flash",
        "label": "xiaomi/mimo-v2.6-flash",
        "model": "xiaomi/mimo-v2.6-flash",
        "provider": "Xiaomi",
        "harness": null,
        "effort": null,
        "score": 59.904,
        "costUsd": 0.048982,
        "uncertainty": {
          "lower": 56.774,
          "upper": 63.034000000000006,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "xiaomi/mimo-v2.6-flash"
          },
          {
            "label": "Standard error",
            "value": "±3.13 points"
          },
          {
            "label": "Mean latency",
            "value": "1,039 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "30.6%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "66.0%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "56.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "61.4%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "62.4%"
          },
          {
            "label": "Forms & Filings",
            "value": "54.9%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "55.7%"
          }
        ]
      },
      {
        "id": "xiaomi-mimo-v2-6-pro",
        "label": "xiaomi/mimo-v2.6-pro",
        "model": "xiaomi/mimo-v2.6-pro",
        "provider": "Xiaomi",
        "harness": null,
        "effort": null,
        "score": 64.939,
        "costUsd": 0.134306,
        "uncertainty": {
          "lower": 61.724999999999994,
          "upper": 68.15299999999999,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "xiaomi/mimo-v2.6-pro"
          },
          {
            "label": "Standard error",
            "value": "±3.21 points"
          },
          {
            "label": "Mean latency",
            "value": "1,548 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "32.6%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "58.4%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "59.8%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "73.9%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "75.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "59.1%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "65.7%"
          }
        ]
      },
      {
        "id": "zai-glm-5-3",
        "label": "zai/glm-5.3",
        "model": "zai/glm-5.3",
        "provider": "Fireworks AI",
        "harness": null,
        "effort": "max",
        "score": 73.089,
        "costUsd": 1.80097,
        "uncertainty": {
          "lower": 70.116,
          "upper": 76.062,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "zai/glm-5.3"
          },
          {
            "label": "Standard error",
            "value": "±2.97 points"
          },
          {
            "label": "Mean latency",
            "value": "3,053 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "39.9%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "58.2%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "81.3%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "78.2%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "84.5%"
          },
          {
            "label": "Forms & Filings",
            "value": "71.7%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "71.1%"
          }
        ]
      },
      {
        "id": "zai-glm-5-3-flash",
        "label": "zai/glm-5.3-flash",
        "model": "zai/glm-5.3-flash",
        "provider": "Zhipu AI",
        "harness": null,
        "effort": "max",
        "score": 61.956,
        "costUsd": 0.138062,
        "uncertainty": {
          "lower": 58.7,
          "upper": 65.212,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/tax_agent_bench",
        "details": [
          {
            "label": "Publisher model id",
            "value": "zai/glm-5.3-flash"
          },
          {
            "label": "Standard error",
            "value": "±3.26 points"
          },
          {
            "label": "Mean latency",
            "value": "1,297 seconds per test"
          },
          {
            "label": "Evaluation mode",
            "value": "Agentic · many steps allowed"
          },
          {
            "label": "All-Pass",
            "value": "31.6%"
          },
          {
            "label": "Fact-Pattern Analysis",
            "value": "57.3%"
          },
          {
            "label": "Rule & Source Lookup",
            "value": "60.2%"
          },
          {
            "label": "Current & Temporal Analysis",
            "value": "61.4%"
          },
          {
            "label": "Numbers & Calculations",
            "value": "81.0%"
          },
          {
            "label": "Forms & Filings",
            "value": "61.4%"
          },
          {
            "label": "Controversy & Precedence Analysis",
            "value": "52.4%"
          }
        ]
      }
    ]
  }
}
