{
  "schemaVersion": 1,
  "benchmark": {
    "id": "vals-medcode",
    "name": "MedCode · Vals AI",
    "version": "1 · private test set",
    "category": "work",
    "question": "Can a model assign the right medical billing code?",
    "summary": "Medical coding for the billing process, scored one-shot rather than as an agent.",
    "source": {
      "name": "Vals AI",
      "url": "https://www.vals.ai/benchmarks/medcode",
      "methodologyUrl": "https://www.vals.ai/about"
    },
    "measure": "Accuracy on medical billing code assignment in a single-response setting.",
    "comparisonRule": "One-shot only. These scores are not comparable with the agentic Vals boards, which let a model take many steps.",
    "limitations": [
      "Vals does not present a comparable cost for this board, so no cost axis is offered.",
      "The board retains configurations from 2025 alongside current models, so the range spans more than one model generation.",
      "The test set is private, so no independent party can reproduce these scores.",
      "Model names are the identifiers Vals publishes, not aicharts canonical names."
    ],
    "coverage": "charted",
    "tags": [
      "healthcare",
      "medical coding",
      "billing",
      "one-shot",
      "professional work",
      "Vals AI",
      "independent evaluator"
    ]
  },
  "dataset": {
    "benchmarkId": "vals-medcode",
    "version": "1 · private test set",
    "score": {
      "label": "Accuracy",
      "unit": "%",
      "direction": "higher",
      "minimum": 0,
      "maximum": 100
    },
    "source": {
      "name": "Vals AI · MedCode",
      "url": "https://www.vals.ai/benchmarks/medcode",
      "retrievedAt": "2026-09-28T18:16:23.422Z",
      "revision": "medcode v1"
    },
    "observedAt": "2026-09-26T00:00:00.000Z",
    "evidenceLabel": "Independent evaluator · private test set",
    "configurationLabel": "Model at the effort Vals selected",
    "comparabilityNote": "Private one-shot evaluation run by Vals. The publisher does not present a comparable cost for this board. Model names are the identifiers Vals publishes, not aicharts canonical names.",
    "points": [
      {
        "id": "alibaba-qwen3-5-flash",
        "label": "alibaba/qwen3.5-flash",
        "model": "alibaba/qwen3.5-flash",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 32.997,
        "costUsd": null,
        "uncertainty": {
          "lower": 31.21,
          "upper": 34.784,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.5-flash"
          },
          {
            "label": "Standard error",
            "value": "±1.79 points"
          },
          {
            "label": "Mean latency",
            "value": "63 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-6-plus",
        "label": "alibaba/qwen3.6-plus",
        "model": "alibaba/qwen3.6-plus",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 36.894,
        "costUsd": null,
        "uncertainty": {
          "lower": 34.876999999999995,
          "upper": 38.911,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.6-plus"
          },
          {
            "label": "Standard error",
            "value": "±2.02 points"
          },
          {
            "label": "Mean latency",
            "value": "56 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-7-max",
        "label": "alibaba/qwen3.7-max",
        "model": "alibaba/qwen3.7-max",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 38.751,
        "costUsd": null,
        "uncertainty": {
          "lower": 36.555,
          "upper": 40.946999999999996,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.7-max"
          },
          {
            "label": "Standard error",
            "value": "±2.20 points"
          },
          {
            "label": "Mean latency",
            "value": "29 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-8-27b",
        "label": "alibaba/qwen3.8-27b",
        "model": "alibaba/qwen3.8-27b",
        "provider": "Alibaba",
        "harness": null,
        "effort": "xhigh",
        "score": 28.698,
        "costUsd": null,
        "uncertainty": {
          "lower": 26.727,
          "upper": 30.669,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.8-27b"
          },
          {
            "label": "Standard error",
            "value": "±1.97 points"
          },
          {
            "label": "Mean latency",
            "value": "89 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-8-max",
        "label": "alibaba/qwen3.8-max",
        "model": "alibaba/qwen3.8-max",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 40.668,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.638999999999996,
          "upper": 42.697,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3.8-max"
          },
          {
            "label": "Standard error",
            "value": "±2.03 points"
          },
          {
            "label": "Mean latency",
            "value": "378 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-max-2026-01-23",
        "label": "alibaba/qwen3-max-2026-01-23",
        "model": "alibaba/qwen3-max-2026-01-23",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 31.373,
        "costUsd": null,
        "uncertainty": {
          "lower": 29.485,
          "upper": 33.261,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3-max-2026-01-23"
          },
          {
            "label": "Standard error",
            "value": "±1.89 points"
          },
          {
            "label": "Mean latency",
            "value": "182 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "alibaba-qwen3-vl-plus-2025-09-23",
        "label": "alibaba/qwen3-vl-plus-2025-09-23",
        "model": "alibaba/qwen3-vl-plus-2025-09-23",
        "provider": "Alibaba",
        "harness": null,
        "effort": null,
        "score": 31.651,
        "costUsd": null,
        "uncertainty": {
          "lower": 29.806,
          "upper": 33.496,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "alibaba/qwen3-vl-plus-2025-09-23"
          },
          {
            "label": "Standard error",
            "value": "±1.84 points"
          },
          {
            "label": "Mean latency",
            "value": "10 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "ant-ling-3-0-flash-2607",
        "label": "ant/ling-3.0-flash-2607",
        "model": "ant/ling-3.0-flash-2607",
        "provider": "Ant",
        "harness": null,
        "effort": null,
        "score": 32.275,
        "costUsd": null,
        "uncertainty": {
          "lower": 30.366999999999997,
          "upper": 34.183,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "ant/ling-3.0-flash-2607"
          },
          {
            "label": "Standard error",
            "value": "±1.91 points"
          },
          {
            "label": "Mean latency",
            "value": "10 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "ant-ling-3-0-flash-af-rc3",
        "label": "ant/ling-3.0-flash-af-rc3",
        "model": "ant/ling-3.0-flash-af-rc3",
        "provider": "Ant",
        "harness": null,
        "effort": null,
        "score": 29.305,
        "costUsd": null,
        "uncertainty": {
          "lower": 27.362,
          "upper": 31.248,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "ant/ling-3.0-flash-af-rc3"
          },
          {
            "label": "Standard error",
            "value": "±1.94 points"
          },
          {
            "label": "Mean latency",
            "value": "31 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-fable-5",
        "label": "anthropic/claude-fable-5",
        "model": "anthropic/claude-fable-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 56.07,
        "costUsd": null,
        "uncertainty": {
          "lower": 53.867,
          "upper": 58.273,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-fable-5"
          },
          {
            "label": "Standard error",
            "value": "±2.20 points"
          },
          {
            "label": "Mean latency",
            "value": "91 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-fable-5-1",
        "label": "anthropic/claude-fable-5-1",
        "model": "anthropic/claude-fable-5-1",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 53.509,
        "costUsd": null,
        "uncertainty": {
          "lower": 51.344,
          "upper": 55.674,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-fable-5-1"
          },
          {
            "label": "Standard error",
            "value": "±2.17 points"
          },
          {
            "label": "Mean latency",
            "value": "215 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-haiku-4-5-20251001-thinking",
        "label": "anthropic/claude-haiku-4-5-20251001-thinking",
        "model": "anthropic/claude-haiku-4-5-20251001-thinking",
        "provider": "Anthropic",
        "harness": null,
        "effort": null,
        "score": 32.678,
        "costUsd": null,
        "uncertainty": {
          "lower": 30.679999999999996,
          "upper": 34.675999999999995,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-haiku-4-5-20251001-thinking"
          },
          {
            "label": "Standard error",
            "value": "±2.00 points"
          },
          {
            "label": "Mean latency",
            "value": "34 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-1-20250805",
        "label": "anthropic/claude-opus-4-1-20250805",
        "model": "anthropic/claude-opus-4-1-20250805",
        "provider": "Anthropic",
        "harness": null,
        "effort": null,
        "score": 41.372,
        "costUsd": null,
        "uncertainty": {
          "lower": 39.414,
          "upper": 43.33,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-1-20250805"
          },
          {
            "label": "Standard error",
            "value": "±1.96 points"
          },
          {
            "label": "Mean latency",
            "value": "13 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-1-20250805-thinking",
        "label": "anthropic/claude-opus-4-1-20250805-thinking",
        "model": "anthropic/claude-opus-4-1-20250805-thinking",
        "provider": "Anthropic",
        "harness": null,
        "effort": null,
        "score": 47.235,
        "costUsd": null,
        "uncertainty": {
          "lower": 45.168,
          "upper": 49.302,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-1-20250805-thinking"
          },
          {
            "label": "Standard error",
            "value": "±2.07 points"
          },
          {
            "label": "Mean latency",
            "value": "33 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-5-20251101",
        "label": "anthropic/claude-opus-4-5-20251101",
        "model": "anthropic/claude-opus-4-5-20251101",
        "provider": "Anthropic",
        "harness": null,
        "effort": "high",
        "score": 45.174,
        "costUsd": null,
        "uncertainty": {
          "lower": 43.286,
          "upper": 47.062,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-5-20251101"
          },
          {
            "label": "Standard error",
            "value": "±1.89 points"
          },
          {
            "label": "Mean latency",
            "value": "5 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-5-20251101-thinking",
        "label": "anthropic/claude-opus-4-5-20251101-thinking",
        "model": "anthropic/claude-opus-4-5-20251101-thinking",
        "provider": "Anthropic",
        "harness": null,
        "effort": "high",
        "score": 49.156,
        "costUsd": null,
        "uncertainty": {
          "lower": 47.144,
          "upper": 51.168,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-5-20251101-thinking"
          },
          {
            "label": "Standard error",
            "value": "±2.01 points"
          },
          {
            "label": "Mean latency",
            "value": "61 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-6",
        "label": "anthropic/claude-opus-4-6",
        "model": "anthropic/claude-opus-4-6",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 48.244,
        "costUsd": null,
        "uncertainty": {
          "lower": 46.194,
          "upper": 50.294,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-6"
          },
          {
            "label": "Standard error",
            "value": "±2.05 points"
          },
          {
            "label": "Mean latency",
            "value": "4 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-6-thinking",
        "label": "anthropic/claude-opus-4-6-thinking",
        "model": "anthropic/claude-opus-4-6-thinking",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 49.129,
        "costUsd": null,
        "uncertainty": {
          "lower": 47.044,
          "upper": 51.214,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-6-thinking"
          },
          {
            "label": "Standard error",
            "value": "±2.08 points"
          },
          {
            "label": "Mean latency",
            "value": "156 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-7",
        "label": "anthropic/claude-opus-4-7",
        "model": "anthropic/claude-opus-4-7",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 54.858,
        "costUsd": null,
        "uncertainty": {
          "lower": 52.653,
          "upper": 57.062999999999995,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-7"
          },
          {
            "label": "Standard error",
            "value": "±2.21 points"
          },
          {
            "label": "Mean latency",
            "value": "54 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-4-8",
        "label": "anthropic/claude-opus-4-8",
        "model": "anthropic/claude-opus-4-8",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 53.217,
        "costUsd": null,
        "uncertainty": {
          "lower": 51.052,
          "upper": 55.382,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-4-8"
          },
          {
            "label": "Standard error",
            "value": "±2.17 points"
          },
          {
            "label": "Mean latency",
            "value": "106 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-5",
        "label": "anthropic/claude-opus-5",
        "model": "anthropic/claude-opus-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 63.57,
        "costUsd": null,
        "uncertainty": {
          "lower": 61.577,
          "upper": 65.563,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-5"
          },
          {
            "label": "Standard error",
            "value": "±1.99 points"
          },
          {
            "label": "Mean latency",
            "value": "25 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-opus-5-5",
        "label": "anthropic/claude-opus-5-5",
        "model": "anthropic/claude-opus-5-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 49.797,
        "costUsd": null,
        "uncertainty": {
          "lower": 47.523999999999994,
          "upper": 52.07,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-opus-5-5"
          },
          {
            "label": "Standard error",
            "value": "±2.27 points"
          },
          {
            "label": "Mean latency",
            "value": "247 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-sonnet-4-20250514",
        "label": "anthropic/claude-sonnet-4-20250514",
        "model": "anthropic/claude-sonnet-4-20250514",
        "provider": "Anthropic",
        "harness": null,
        "effort": null,
        "score": 33.943,
        "costUsd": null,
        "uncertainty": {
          "lower": 32.037,
          "upper": 35.849,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-sonnet-4-20250514"
          },
          {
            "label": "Standard error",
            "value": "±1.91 points"
          },
          {
            "label": "Mean latency",
            "value": "7 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-sonnet-4-20250514-thinking",
        "label": "anthropic/claude-sonnet-4-20250514-thinking",
        "model": "anthropic/claude-sonnet-4-20250514-thinking",
        "provider": "Anthropic",
        "harness": null,
        "effort": null,
        "score": 34.959,
        "costUsd": null,
        "uncertainty": {
          "lower": 33.02,
          "upper": 36.898,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-sonnet-4-20250514-thinking"
          },
          {
            "label": "Standard error",
            "value": "±1.94 points"
          },
          {
            "label": "Mean latency",
            "value": "40 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-sonnet-4-5-20250929",
        "label": "anthropic/claude-sonnet-4-5-20250929",
        "model": "anthropic/claude-sonnet-4-5-20250929",
        "provider": "Anthropic",
        "harness": null,
        "effort": null,
        "score": 40.569,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.574000000000005,
          "upper": 42.564,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-sonnet-4-5-20250929"
          },
          {
            "label": "Standard error",
            "value": "±2.00 points"
          },
          {
            "label": "Mean latency",
            "value": "12 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-sonnet-4-5-20250929-thinking",
        "label": "anthropic/claude-sonnet-4-5-20250929-thinking",
        "model": "anthropic/claude-sonnet-4-5-20250929-thinking",
        "provider": "Anthropic",
        "harness": null,
        "effort": null,
        "score": 44.134,
        "costUsd": null,
        "uncertainty": {
          "lower": 42.136,
          "upper": 46.132,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-sonnet-4-5-20250929-thinking"
          },
          {
            "label": "Standard error",
            "value": "±2.00 points"
          },
          {
            "label": "Mean latency",
            "value": "74 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-sonnet-5",
        "label": "anthropic/claude-sonnet-5",
        "model": "anthropic/claude-sonnet-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 47.54,
        "costUsd": null,
        "uncertainty": {
          "lower": 45.266,
          "upper": 49.814,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-sonnet-5"
          },
          {
            "label": "Standard error",
            "value": "±2.27 points"
          },
          {
            "label": "Mean latency",
            "value": "135 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "anthropic-claude-sonnet-5-5",
        "label": "anthropic/claude-sonnet-5-5",
        "model": "anthropic/claude-sonnet-5-5",
        "provider": "Anthropic",
        "harness": null,
        "effort": "max",
        "score": 52.92,
        "costUsd": null,
        "uncertainty": {
          "lower": 50.801,
          "upper": 55.039,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "anthropic/claude-sonnet-5-5"
          },
          {
            "label": "Standard error",
            "value": "±2.12 points"
          },
          {
            "label": "Mean latency",
            "value": "239 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "cohere-command-a-plus-05-2026",
        "label": "cohere/command-a-plus-05-2026",
        "model": "cohere/command-a-plus-05-2026",
        "provider": "Cohere",
        "harness": null,
        "effort": null,
        "score": 19.718,
        "costUsd": null,
        "uncertainty": {
          "lower": 17.883,
          "upper": 21.553,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "cohere/command-a-plus-05-2026"
          },
          {
            "label": "Standard error",
            "value": "±1.83 points"
          },
          {
            "label": "Mean latency",
            "value": "104 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "deepseek-deepseek-v4-1-flash",
        "label": "deepseek/deepseek-v4.1-flash",
        "model": "deepseek/deepseek-v4.1-flash",
        "provider": "DeepSeek",
        "harness": null,
        "effort": "high",
        "score": 41.173,
        "costUsd": null,
        "uncertainty": {
          "lower": 39.131,
          "upper": 43.215,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "deepseek/deepseek-v4.1-flash"
          },
          {
            "label": "Standard error",
            "value": "±2.04 points"
          },
          {
            "label": "Mean latency",
            "value": "35 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "deepseek-deepseek-v4-flash-0731",
        "label": "deepseek/deepseek-v4-flash-0731",
        "model": "deepseek/deepseek-v4-flash-0731",
        "provider": "DeepSeek",
        "harness": null,
        "effort": "high",
        "score": 41.415,
        "costUsd": null,
        "uncertainty": {
          "lower": 39.265,
          "upper": 43.565,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "deepseek/deepseek-v4-flash-0731"
          },
          {
            "label": "Standard error",
            "value": "±2.15 points"
          },
          {
            "label": "Mean latency",
            "value": "127 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "deepseek-deepseek-v4-pro",
        "label": "deepseek/deepseek-v4-pro",
        "model": "deepseek/deepseek-v4-pro",
        "provider": "DeepSeek",
        "harness": null,
        "effort": "max",
        "score": 40.455,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.333,
          "upper": 42.577,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "deepseek/deepseek-v4-pro"
          },
          {
            "label": "Standard error",
            "value": "±2.12 points"
          },
          {
            "label": "Mean latency",
            "value": "383 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "deepseek-deepseek-v4-pro-0813",
        "label": "deepseek/deepseek-v4-pro-0813",
        "model": "deepseek/deepseek-v4-pro-0813",
        "provider": "DeepSeek",
        "harness": null,
        "effort": "max",
        "score": 42.47,
        "costUsd": null,
        "uncertainty": {
          "lower": 40.31,
          "upper": 44.629999999999995,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "deepseek/deepseek-v4-pro-0813"
          },
          {
            "label": "Standard error",
            "value": "±2.16 points"
          },
          {
            "label": "Mean latency",
            "value": "181 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "fireworks-llama4-maverick-instruct-basic",
        "label": "fireworks/llama4-maverick-instruct-basic",
        "model": "fireworks/llama4-maverick-instruct-basic",
        "provider": "Fireworks AI",
        "harness": null,
        "effort": null,
        "score": 36.514,
        "costUsd": null,
        "uncertainty": {
          "lower": 34.52,
          "upper": 38.508,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "fireworks/llama4-maverick-instruct-basic"
          },
          {
            "label": "Standard error",
            "value": "±1.99 points"
          },
          {
            "label": "Mean latency",
            "value": "21 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-2-5-flash",
        "label": "google/gemini-2.5-flash",
        "model": "google/gemini-2.5-flash",
        "provider": "Google",
        "harness": null,
        "effort": null,
        "score": 38.425,
        "costUsd": null,
        "uncertainty": {
          "lower": 36.501999999999995,
          "upper": 40.348,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-2.5-flash"
          },
          {
            "label": "Standard error",
            "value": "±1.92 points"
          },
          {
            "label": "Mean latency",
            "value": "22 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-2-5-flash-lite",
        "label": "google/gemini-2.5-flash-lite",
        "model": "google/gemini-2.5-flash-lite",
        "provider": "Google",
        "harness": null,
        "effort": null,
        "score": 27.115,
        "costUsd": null,
        "uncertainty": {
          "lower": 25.272,
          "upper": 28.958,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-2.5-flash-lite"
          },
          {
            "label": "Standard error",
            "value": "±1.84 points"
          },
          {
            "label": "Mean latency",
            "value": "6 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-2-5-flash-lite-preview-09-2025",
        "label": "google/gemini-2.5-flash-lite-preview-09-2025",
        "model": "google/gemini-2.5-flash-lite-preview-09-2025",
        "provider": "Google",
        "harness": null,
        "effort": null,
        "score": 27.079,
        "costUsd": null,
        "uncertainty": {
          "lower": 25.168,
          "upper": 28.990000000000002,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-2.5-flash-lite-preview-09-2025"
          },
          {
            "label": "Standard error",
            "value": "±1.91 points"
          },
          {
            "label": "Mean latency",
            "value": "6 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-2-5-flash-lite-preview-09-2025-thinking",
        "label": "google/gemini-2.5-flash-lite-preview-09-2025-thinking",
        "model": "google/gemini-2.5-flash-lite-preview-09-2025-thinking",
        "provider": "Google",
        "harness": null,
        "effort": null,
        "score": 34.191,
        "costUsd": null,
        "uncertainty": {
          "lower": 32.455000000000005,
          "upper": 35.927,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-2.5-flash-lite-preview-09-2025-thinking"
          },
          {
            "label": "Standard error",
            "value": "±1.74 points"
          },
          {
            "label": "Mean latency",
            "value": "10 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-2-5-flash-preview-09-2025",
        "label": "google/gemini-2.5-flash-preview-09-2025",
        "model": "google/gemini-2.5-flash-preview-09-2025",
        "provider": "Google",
        "harness": null,
        "effort": null,
        "score": 40.538,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.605999999999995,
          "upper": 42.47,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-2.5-flash-preview-09-2025"
          },
          {
            "label": "Standard error",
            "value": "±1.93 points"
          },
          {
            "label": "Mean latency",
            "value": "13 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-2-5-flash-preview-09-2025-thinking",
        "label": "google/gemini-2.5-flash-preview-09-2025-thinking",
        "model": "google/gemini-2.5-flash-preview-09-2025-thinking",
        "provider": "Google",
        "harness": null,
        "effort": null,
        "score": 40.33,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.415,
          "upper": 42.245,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-2.5-flash-preview-09-2025-thinking"
          },
          {
            "label": "Standard error",
            "value": "±1.92 points"
          },
          {
            "label": "Mean latency",
            "value": "16 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-2-5-flash-thinking",
        "label": "google/gemini-2.5-flash-thinking",
        "model": "google/gemini-2.5-flash-thinking",
        "provider": "Google",
        "harness": null,
        "effort": null,
        "score": 40.357,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.405,
          "upper": 42.309,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-2.5-flash-thinking"
          },
          {
            "label": "Standard error",
            "value": "±1.95 points"
          },
          {
            "label": "Mean latency",
            "value": "23 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-2-5-pro",
        "label": "google/gemini-2.5-pro",
        "model": "google/gemini-2.5-pro",
        "provider": "Google",
        "harness": null,
        "effort": null,
        "score": 50.59,
        "costUsd": null,
        "uncertainty": {
          "lower": 48.477000000000004,
          "upper": 52.703,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-2.5-pro"
          },
          {
            "label": "Standard error",
            "value": "±2.11 points"
          },
          {
            "label": "Mean latency",
            "value": "27 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-3-1-flash-lite-preview",
        "label": "google/gemini-3.1-flash-lite-preview",
        "model": "google/gemini-3.1-flash-lite-preview",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 47.602,
        "costUsd": null,
        "uncertainty": {
          "lower": 45.531,
          "upper": 49.672999999999995,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.1-flash-lite-preview"
          },
          {
            "label": "Standard error",
            "value": "±2.07 points"
          },
          {
            "label": "Mean latency",
            "value": "9 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-3-1-pro-preview",
        "label": "google/gemini-3.1-pro-preview",
        "model": "google/gemini-3.1-pro-preview",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 59.062,
        "costUsd": null,
        "uncertainty": {
          "lower": 57.065999999999995,
          "upper": 61.058,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.1-pro-preview"
          },
          {
            "label": "Standard error",
            "value": "±2.00 points"
          },
          {
            "label": "Mean latency",
            "value": "39 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-3-5-flash",
        "label": "google/gemini-3.5-flash",
        "model": "google/gemini-3.5-flash",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 55.825,
        "costUsd": null,
        "uncertainty": {
          "lower": 53.712,
          "upper": 57.938,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.5-flash"
          },
          {
            "label": "Standard error",
            "value": "±2.11 points"
          },
          {
            "label": "Mean latency",
            "value": "25 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-3-5-flash-lite",
        "label": "google/gemini-3.5-flash-lite",
        "model": "google/gemini-3.5-flash-lite",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 43.489,
        "costUsd": null,
        "uncertainty": {
          "lower": 41.538,
          "upper": 45.44,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.5-flash-lite"
          },
          {
            "label": "Standard error",
            "value": "±1.95 points"
          },
          {
            "label": "Mean latency",
            "value": "6 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-3-6-flash",
        "label": "google/gemini-3.6-flash",
        "model": "google/gemini-3.6-flash",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 53.153,
        "costUsd": null,
        "uncertainty": {
          "lower": 50.995999999999995,
          "upper": 55.31,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.6-flash"
          },
          {
            "label": "Standard error",
            "value": "±2.16 points"
          },
          {
            "label": "Mean latency",
            "value": "17 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-3-7-flash",
        "label": "google/gemini-3.7-flash",
        "model": "google/gemini-3.7-flash",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 53.39,
        "costUsd": null,
        "uncertainty": {
          "lower": 51.27,
          "upper": 55.51,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.7-flash"
          },
          {
            "label": "Standard error",
            "value": "±2.12 points"
          },
          {
            "label": "Mean latency",
            "value": "9 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-3-8-flash",
        "label": "google/gemini-3.8-flash",
        "model": "google/gemini-3.8-flash",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 48.135,
        "costUsd": null,
        "uncertainty": {
          "lower": 45.955,
          "upper": 50.315,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3.8-flash"
          },
          {
            "label": "Standard error",
            "value": "±2.18 points"
          },
          {
            "label": "Mean latency",
            "value": "43 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-3-flash-preview",
        "label": "google/gemini-3-flash-preview",
        "model": "google/gemini-3-flash-preview",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 55.92,
        "costUsd": null,
        "uncertainty": {
          "lower": 53.808,
          "upper": 58.032000000000004,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3-flash-preview"
          },
          {
            "label": "Standard error",
            "value": "±2.11 points"
          },
          {
            "label": "Mean latency",
            "value": "44 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "google-gemini-3-pro-preview",
        "label": "google/gemini-3-pro-preview",
        "model": "google/gemini-3-pro-preview",
        "provider": "Google",
        "harness": null,
        "effort": "high",
        "score": 52.198,
        "costUsd": null,
        "uncertainty": {
          "lower": 50.125,
          "upper": 54.271,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "google/gemini-3-pro-preview"
          },
          {
            "label": "Standard error",
            "value": "±2.07 points"
          },
          {
            "label": "Mean latency",
            "value": "56 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-0709",
        "label": "grok/grok-4-0709",
        "model": "grok/grok-4-0709",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": null,
        "score": 38.078,
        "costUsd": null,
        "uncertainty": {
          "lower": 35.872,
          "upper": 40.284000000000006,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4-0709"
          },
          {
            "label": "Standard error",
            "value": "±2.21 points"
          },
          {
            "label": "Mean latency",
            "value": "89 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-1-fast-non-reasoning",
        "label": "grok/grok-4-1-fast-non-reasoning",
        "model": "grok/grok-4-1-fast-non-reasoning",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": null,
        "score": 28.349,
        "costUsd": null,
        "uncertainty": {
          "lower": 26.428,
          "upper": 30.27,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4-1-fast-non-reasoning"
          },
          {
            "label": "Standard error",
            "value": "±1.92 points"
          },
          {
            "label": "Mean latency",
            "value": "4 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-1-fast-reasoning",
        "label": "grok/grok-4-1-fast-reasoning",
        "model": "grok/grok-4-1-fast-reasoning",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": null,
        "score": 28.08,
        "costUsd": null,
        "uncertainty": {
          "lower": 26.087999999999997,
          "upper": 30.072,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4-1-fast-reasoning"
          },
          {
            "label": "Standard error",
            "value": "±1.99 points"
          },
          {
            "label": "Mean latency",
            "value": "47 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-20-0309-reasoning",
        "label": "grok/grok-4.20-0309-reasoning",
        "model": "grok/grok-4.20-0309-reasoning",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": null,
        "score": 32.156,
        "costUsd": null,
        "uncertainty": {
          "lower": 30.032,
          "upper": 34.28,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.20-0309-reasoning"
          },
          {
            "label": "Standard error",
            "value": "±2.12 points"
          },
          {
            "label": "Mean latency",
            "value": "17 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-3",
        "label": "grok/grok-4.3",
        "model": "grok/grok-4.3",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": null,
        "score": 38.068,
        "costUsd": null,
        "uncertainty": {
          "lower": 35.986999999999995,
          "upper": 40.149,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.3"
          },
          {
            "label": "Standard error",
            "value": "±2.08 points"
          },
          {
            "label": "Mean latency",
            "value": "44 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-5",
        "label": "grok/grok-4.5",
        "model": "grok/grok-4.5",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": "high",
        "score": 43.291,
        "costUsd": null,
        "uncertainty": {
          "lower": 40.977999999999994,
          "upper": 45.604,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.5"
          },
          {
            "label": "Standard error",
            "value": "±2.31 points"
          },
          {
            "label": "Mean latency",
            "value": "60 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-6",
        "label": "grok/grok-4.6",
        "model": "grok/grok-4.6",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": "high",
        "score": 44.712,
        "costUsd": null,
        "uncertainty": {
          "lower": 42.456,
          "upper": 46.968,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.6"
          },
          {
            "label": "Standard error",
            "value": "±2.26 points"
          },
          {
            "label": "Mean latency",
            "value": "127 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-7",
        "label": "grok/grok-4.7",
        "model": "grok/grok-4.7",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": "xhigh",
        "score": 49.553,
        "costUsd": null,
        "uncertainty": {
          "lower": 47.382,
          "upper": 51.724,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4.7"
          },
          {
            "label": "Standard error",
            "value": "±2.17 points"
          },
          {
            "label": "Mean latency",
            "value": "213 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-fast-non-reasoning",
        "label": "grok/grok-4-fast-non-reasoning",
        "model": "grok/grok-4-fast-non-reasoning",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": null,
        "score": 30.036,
        "costUsd": null,
        "uncertainty": {
          "lower": 28.062,
          "upper": 32.01,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4-fast-non-reasoning"
          },
          {
            "label": "Standard error",
            "value": "±1.97 points"
          },
          {
            "label": "Mean latency",
            "value": "18 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "grok-grok-4-fast-reasoning",
        "label": "grok/grok-4-fast-reasoning",
        "model": "grok/grok-4-fast-reasoning",
        "provider": "SpaceXAI",
        "harness": null,
        "effort": null,
        "score": 37.385,
        "costUsd": null,
        "uncertainty": {
          "lower": 35.443999999999996,
          "upper": 39.326,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "grok/grok-4-fast-reasoning"
          },
          {
            "label": "Standard error",
            "value": "±1.94 points"
          },
          {
            "label": "Mean latency",
            "value": "18 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "inception-mercury-2-5",
        "label": "inception/mercury-2.5",
        "model": "inception/mercury-2.5",
        "provider": "Inception",
        "harness": null,
        "effort": "high",
        "score": 31.326,
        "costUsd": null,
        "uncertainty": {
          "lower": 29.373,
          "upper": 33.279,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "inception/mercury-2.5"
          },
          {
            "label": "Standard error",
            "value": "±1.95 points"
          },
          {
            "label": "Mean latency",
            "value": "8 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "kimi-kimi-k2-5-thinking",
        "label": "kimi/kimi-k2.5-thinking",
        "model": "kimi/kimi-k2.5-thinking",
        "provider": "Moonshot AI",
        "harness": null,
        "effort": null,
        "score": 39.316,
        "costUsd": null,
        "uncertainty": {
          "lower": 37.197,
          "upper": 41.435,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "kimi/kimi-k2.5-thinking"
          },
          {
            "label": "Standard error",
            "value": "±2.12 points"
          },
          {
            "label": "Mean latency",
            "value": "75 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "kimi-kimi-k2-6",
        "label": "kimi/kimi-k2.6",
        "model": "kimi/kimi-k2.6",
        "provider": "Moonshot AI",
        "harness": null,
        "effort": null,
        "score": 40.142,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.101000000000006,
          "upper": 42.183,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "kimi/kimi-k2.6"
          },
          {
            "label": "Standard error",
            "value": "±2.04 points"
          },
          {
            "label": "Mean latency",
            "value": "305 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "kimi-kimi-k3",
        "label": "kimi/kimi-k3",
        "model": "kimi/kimi-k3",
        "provider": "Moonshot AI",
        "harness": null,
        "effort": null,
        "score": 48.884,
        "costUsd": null,
        "uncertainty": {
          "lower": 46.691,
          "upper": 51.077,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "kimi/kimi-k3"
          },
          {
            "label": "Standard error",
            "value": "±2.19 points"
          },
          {
            "label": "Mean latency",
            "value": "116 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "meta-muse-spark",
        "label": "meta/muse_spark",
        "model": "meta/muse_spark",
        "provider": "Meta",
        "harness": null,
        "effort": null,
        "score": 51.31,
        "costUsd": null,
        "uncertainty": {
          "lower": 49.066,
          "upper": 53.554,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "meta/muse_spark"
          },
          {
            "label": "Standard error",
            "value": "±2.24 points"
          },
          {
            "label": "Mean latency",
            "value": "123 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "meta-muse-spark-1-2",
        "label": "meta/muse_spark_1_2",
        "model": "meta/muse_spark_1_2",
        "provider": "Meta",
        "harness": null,
        "effort": "xhigh",
        "score": 49.346,
        "costUsd": null,
        "uncertainty": {
          "lower": 47.159,
          "upper": 51.532999999999994,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "meta/muse_spark_1_2"
          },
          {
            "label": "Standard error",
            "value": "±2.19 points"
          },
          {
            "label": "Mean latency",
            "value": "60 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "minimax-minimax-m2-1",
        "label": "minimax/MiniMax-M2.1",
        "model": "minimax/MiniMax-M2.1",
        "provider": "MiniMax",
        "harness": null,
        "effort": null,
        "score": 34.083,
        "costUsd": null,
        "uncertainty": {
          "lower": 32.14,
          "upper": 36.025999999999996,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "minimax/MiniMax-M2.1"
          },
          {
            "label": "Standard error",
            "value": "±1.94 points"
          },
          {
            "label": "Mean latency",
            "value": "20 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "minimax-minimax-m2-7",
        "label": "minimax/MiniMax-M2.7",
        "model": "minimax/MiniMax-M2.7",
        "provider": "MiniMax",
        "harness": null,
        "effort": null,
        "score": 34.44,
        "costUsd": null,
        "uncertainty": {
          "lower": 32.455,
          "upper": 36.425,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "minimax/MiniMax-M2.7"
          },
          {
            "label": "Standard error",
            "value": "±1.99 points"
          },
          {
            "label": "Mean latency",
            "value": "30 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "minimax-minimax-m3",
        "label": "minimax/MiniMax-M3",
        "model": "minimax/MiniMax-M3",
        "provider": "MiniMax",
        "harness": null,
        "effort": null,
        "score": 46.289,
        "costUsd": null,
        "uncertainty": {
          "lower": 44.185,
          "upper": 48.393,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "minimax/MiniMax-M3"
          },
          {
            "label": "Standard error",
            "value": "±2.10 points"
          },
          {
            "label": "Mean latency",
            "value": "63 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "mistralai-mistral-medium-3-5",
        "label": "mistralai/mistral-medium-3.5",
        "model": "mistralai/mistral-medium-3.5",
        "provider": "Mistral AI",
        "harness": null,
        "effort": "high",
        "score": 33.752,
        "costUsd": null,
        "uncertainty": {
          "lower": 32.604,
          "upper": 34.900000000000006,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "mistralai/mistral-medium-3.5"
          },
          {
            "label": "Standard error",
            "value": "±1.15 points"
          },
          {
            "label": "Mean latency",
            "value": "35 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "nvidia-nemotron-3-ultra-550b-a55b",
        "label": "nvidia/nemotron-3-ultra-550b-a55b",
        "model": "nvidia/nemotron-3-ultra-550b-a55b",
        "provider": "Nvidia",
        "harness": null,
        "effort": null,
        "score": 38.621,
        "costUsd": null,
        "uncertainty": {
          "lower": 36.620000000000005,
          "upper": 40.622,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "nvidia/nemotron-3-ultra-550b-a55b"
          },
          {
            "label": "Standard error",
            "value": "±2.00 points"
          },
          {
            "label": "Mean latency",
            "value": "19 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-1-2025-11-13",
        "label": "openai/gpt-5.1-2025-11-13",
        "model": "openai/gpt-5.1-2025-11-13",
        "provider": "OpenAI",
        "harness": null,
        "effort": "high",
        "score": 52.732,
        "costUsd": null,
        "uncertainty": {
          "lower": 50.581,
          "upper": 54.882999999999996,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.1-2025-11-13"
          },
          {
            "label": "Standard error",
            "value": "±2.15 points"
          },
          {
            "label": "Mean latency",
            "value": "55 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-2-2025-12-11",
        "label": "openai/gpt-5.2-2025-12-11",
        "model": "openai/gpt-5.2-2025-12-11",
        "provider": "OpenAI",
        "harness": null,
        "effort": "xhigh",
        "score": 49.749,
        "costUsd": null,
        "uncertainty": {
          "lower": 47.487,
          "upper": 52.011,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.2-2025-12-11"
          },
          {
            "label": "Standard error",
            "value": "±2.26 points"
          },
          {
            "label": "Mean latency",
            "value": "157 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-2025-08-07",
        "label": "openai/gpt-5-2025-08-07",
        "model": "openai/gpt-5-2025-08-07",
        "provider": "OpenAI",
        "harness": null,
        "effort": "high",
        "score": 49.634,
        "costUsd": null,
        "uncertainty": {
          "lower": 47.536,
          "upper": 51.732,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5-2025-08-07"
          },
          {
            "label": "Standard error",
            "value": "±2.10 points"
          },
          {
            "label": "Mean latency",
            "value": "58 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-4-2026-03-05",
        "label": "openai/gpt-5.4-2026-03-05",
        "model": "openai/gpt-5.4-2026-03-05",
        "provider": "OpenAI",
        "harness": null,
        "effort": "xhigh",
        "score": 41.292,
        "costUsd": null,
        "uncertainty": {
          "lower": 39.144,
          "upper": 43.440000000000005,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.4-2026-03-05"
          },
          {
            "label": "Standard error",
            "value": "±2.15 points"
          },
          {
            "label": "Mean latency",
            "value": "187 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-4-nano-2026-03-17",
        "label": "openai/gpt-5.4-nano-2026-03-17",
        "model": "openai/gpt-5.4-nano-2026-03-17",
        "provider": "OpenAI",
        "harness": null,
        "effort": "high",
        "score": 41.029,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.773,
          "upper": 43.285000000000004,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.4-nano-2026-03-17"
          },
          {
            "label": "Standard error",
            "value": "±2.26 points"
          },
          {
            "label": "Mean latency",
            "value": "8 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-5",
        "label": "openai/gpt-5.5",
        "model": "openai/gpt-5.5",
        "provider": "OpenAI",
        "harness": null,
        "effort": "xhigh",
        "score": 49.1,
        "costUsd": null,
        "uncertainty": {
          "lower": 46.912,
          "upper": 51.288000000000004,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.5"
          },
          {
            "label": "Standard error",
            "value": "±2.19 points"
          },
          {
            "label": "Mean latency",
            "value": "160 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-6-luna",
        "label": "openai/gpt-5.6-luna",
        "model": "openai/gpt-5.6-luna",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 42.391,
        "costUsd": null,
        "uncertainty": {
          "lower": 40.120999999999995,
          "upper": 44.661,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.6-luna"
          },
          {
            "label": "Standard error",
            "value": "±2.27 points"
          },
          {
            "label": "Mean latency",
            "value": "81 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-6-sol",
        "label": "openai/gpt-5.6-sol",
        "model": "openai/gpt-5.6-sol",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 43.974,
        "costUsd": null,
        "uncertainty": {
          "lower": 41.715999999999994,
          "upper": 46.232,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.6-sol"
          },
          {
            "label": "Standard error",
            "value": "±2.26 points"
          },
          {
            "label": "Mean latency",
            "value": "97 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-6-terra",
        "label": "openai/gpt-5.6-terra",
        "model": "openai/gpt-5.6-terra",
        "provider": "OpenAI",
        "harness": null,
        "effort": "xhigh",
        "score": 43.414,
        "costUsd": null,
        "uncertainty": {
          "lower": 41.241,
          "upper": 45.587,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5.6-terra"
          },
          {
            "label": "Standard error",
            "value": "±2.17 points"
          },
          {
            "label": "Mean latency",
            "value": "18 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-mini-2025-08-07",
        "label": "openai/gpt-5-mini-2025-08-07",
        "model": "openai/gpt-5-mini-2025-08-07",
        "provider": "OpenAI",
        "harness": null,
        "effort": "high",
        "score": 43.045,
        "costUsd": null,
        "uncertainty": {
          "lower": 41,
          "upper": 45.09,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5-mini-2025-08-07"
          },
          {
            "label": "Standard error",
            "value": "±2.04 points"
          },
          {
            "label": "Mean latency",
            "value": "28 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-5-nano-2025-08-07",
        "label": "openai/gpt-5-nano-2025-08-07",
        "model": "openai/gpt-5-nano-2025-08-07",
        "provider": "OpenAI",
        "harness": null,
        "effort": "high",
        "score": 30.441,
        "costUsd": null,
        "uncertainty": {
          "lower": 28.493,
          "upper": 32.388999999999996,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-5-nano-2025-08-07"
          },
          {
            "label": "Standard error",
            "value": "±1.95 points"
          },
          {
            "label": "Mean latency",
            "value": "30 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-6-astra",
        "label": "openai/gpt-6-astra",
        "model": "openai/gpt-6-astra",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 48.486,
        "costUsd": null,
        "uncertainty": {
          "lower": 46.355,
          "upper": 50.617,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-6-astra"
          },
          {
            "label": "Standard error",
            "value": "±2.13 points"
          },
          {
            "label": "Mean latency",
            "value": "103 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-6-luna",
        "label": "openai/gpt-6-luna",
        "model": "openai/gpt-6-luna",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 44.685,
        "costUsd": null,
        "uncertainty": {
          "lower": 42.382000000000005,
          "upper": 46.988,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-6-luna"
          },
          {
            "label": "Standard error",
            "value": "±2.30 points"
          },
          {
            "label": "Mean latency",
            "value": "109 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-gpt-6-sol",
        "label": "openai/gpt-6-sol",
        "model": "openai/gpt-6-sol",
        "provider": "OpenAI",
        "harness": null,
        "effort": "max",
        "score": 47.072,
        "costUsd": null,
        "uncertainty": {
          "lower": 44.953,
          "upper": 49.191,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/gpt-6-sol"
          },
          {
            "label": "Standard error",
            "value": "±2.12 points"
          },
          {
            "label": "Mean latency",
            "value": "63 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-o3-2025-04-16",
        "label": "openai/o3-2025-04-16",
        "model": "openai/o3-2025-04-16",
        "provider": "OpenAI",
        "harness": null,
        "effort": "high",
        "score": 47.29,
        "costUsd": null,
        "uncertainty": {
          "lower": 45.129,
          "upper": 49.451,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/o3-2025-04-16"
          },
          {
            "label": "Standard error",
            "value": "±2.16 points"
          },
          {
            "label": "Mean latency",
            "value": "18 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "openai-o4-mini-2025-04-16",
        "label": "openai/o4-mini-2025-04-16",
        "model": "openai/o4-mini-2025-04-16",
        "provider": "OpenAI",
        "harness": null,
        "effort": "high",
        "score": 33.791,
        "costUsd": null,
        "uncertainty": {
          "lower": 31.769999999999996,
          "upper": 35.812,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "openai/o4-mini-2025-04-16"
          },
          {
            "label": "Standard error",
            "value": "±2.02 points"
          },
          {
            "label": "Mean latency",
            "value": "21 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "poolside-laguna-m-1",
        "label": "poolside/laguna-m.1",
        "model": "poolside/laguna-m.1",
        "provider": "Poolside",
        "harness": null,
        "effort": null,
        "score": 23.106,
        "costUsd": null,
        "uncertainty": {
          "lower": 21.413,
          "upper": 24.799000000000003,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "poolside/laguna-m.1"
          },
          {
            "label": "Standard error",
            "value": "±1.69 points"
          },
          {
            "label": "Mean latency",
            "value": "71 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "poolside-laguna-xs-2",
        "label": "poolside/laguna-xs.2",
        "model": "poolside/laguna-xs.2",
        "provider": "Poolside",
        "harness": null,
        "effort": null,
        "score": 21.251,
        "costUsd": null,
        "uncertainty": {
          "lower": 19.548000000000002,
          "upper": 22.954,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "poolside/laguna-xs.2"
          },
          {
            "label": "Standard error",
            "value": "±1.70 points"
          },
          {
            "label": "Mean latency",
            "value": "33 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "tencent-hy4-preview",
        "label": "tencent/hy4-preview",
        "model": "tencent/hy4-preview",
        "provider": "Tencent",
        "harness": null,
        "effort": null,
        "score": 43.247,
        "costUsd": null,
        "uncertainty": {
          "lower": 41.113,
          "upper": 45.381,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "tencent/hy4-preview"
          },
          {
            "label": "Standard error",
            "value": "±2.13 points"
          },
          {
            "label": "Mean latency",
            "value": "399 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "thinkingmachines-inkling",
        "label": "thinkingmachines/inkling",
        "model": "thinkingmachines/inkling",
        "provider": "Thinkingmachines",
        "harness": null,
        "effort": "0.99",
        "score": 41.19,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.96,
          "upper": 43.419999999999995,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "thinkingmachines/inkling"
          },
          {
            "label": "Standard error",
            "value": "±2.23 points"
          },
          {
            "label": "Mean latency",
            "value": "167 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "thinkingmachines-inkling-small",
        "label": "thinkingmachines/inkling-small",
        "model": "thinkingmachines/inkling-small",
        "provider": "Thinkingmachines",
        "harness": null,
        "effort": "0.99",
        "score": 37.893,
        "costUsd": null,
        "uncertainty": {
          "lower": 35.687,
          "upper": 40.099000000000004,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "thinkingmachines/inkling-small"
          },
          {
            "label": "Standard error",
            "value": "±2.21 points"
          },
          {
            "label": "Mean latency",
            "value": "191 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "together-meta-llama-llama-4-scout-17b-16e-instruct",
        "label": "together/meta-llama/Llama-4-Scout-17B-16E-Instruct",
        "model": "together/meta-llama/Llama-4-Scout-17B-16E-Instruct",
        "provider": "Together AI",
        "harness": null,
        "effort": null,
        "score": 23.311,
        "costUsd": null,
        "uncertainty": {
          "lower": 21.562,
          "upper": 25.06,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "together/meta-llama/Llama-4-Scout-17B-16E-Instruct"
          },
          {
            "label": "Standard error",
            "value": "±1.75 points"
          },
          {
            "label": "Mean latency",
            "value": "11 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "xiaomi-mimo-v2-5",
        "label": "xiaomi/mimo-v2.5",
        "model": "xiaomi/mimo-v2.5",
        "provider": "Xiaomi",
        "harness": null,
        "effort": null,
        "score": 31.895,
        "costUsd": null,
        "uncertainty": {
          "lower": 29.87,
          "upper": 33.92,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "xiaomi/mimo-v2.5"
          },
          {
            "label": "Standard error",
            "value": "±2.02 points"
          },
          {
            "label": "Mean latency",
            "value": "18 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "xiaomi-mimo-v2-5-pro",
        "label": "xiaomi/mimo-v2.5-pro",
        "model": "xiaomi/mimo-v2.5-pro",
        "provider": "Xiaomi",
        "harness": null,
        "effort": null,
        "score": 32.484,
        "costUsd": null,
        "uncertainty": {
          "lower": 30.577,
          "upper": 34.391000000000005,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "xiaomi/mimo-v2.5-pro"
          },
          {
            "label": "Standard error",
            "value": "±1.91 points"
          },
          {
            "label": "Mean latency",
            "value": "38 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "xiaomi-mimo-v2-6-flash",
        "label": "xiaomi/mimo-v2.6-flash",
        "model": "xiaomi/mimo-v2.6-flash",
        "provider": "Xiaomi",
        "harness": null,
        "effort": null,
        "score": 41.057,
        "costUsd": null,
        "uncertainty": {
          "lower": 39.025000000000006,
          "upper": 43.089,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "xiaomi/mimo-v2.6-flash"
          },
          {
            "label": "Standard error",
            "value": "±2.03 points"
          },
          {
            "label": "Mean latency",
            "value": "93 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "xiaomi-mimo-v2-6-pro",
        "label": "xiaomi/mimo-v2.6-pro",
        "model": "xiaomi/mimo-v2.6-pro",
        "provider": "Xiaomi",
        "harness": null,
        "effort": null,
        "score": 44.969,
        "costUsd": null,
        "uncertainty": {
          "lower": 42.872,
          "upper": 47.066,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "xiaomi/mimo-v2.6-pro"
          },
          {
            "label": "Standard error",
            "value": "±2.10 points"
          },
          {
            "label": "Mean latency",
            "value": "177 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "zai-glm-4-7",
        "label": "zai/glm-4.7",
        "model": "zai/glm-4.7",
        "provider": "Zhipu AI",
        "harness": null,
        "effort": null,
        "score": 32.772,
        "costUsd": null,
        "uncertainty": {
          "lower": 30.776,
          "upper": 34.768,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "zai/glm-4.7"
          },
          {
            "label": "Standard error",
            "value": "±2.00 points"
          },
          {
            "label": "Mean latency",
            "value": "123 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "zai-glm-5-1",
        "label": "zai/glm-5.1",
        "model": "zai/glm-5.1",
        "provider": "Zhipu AI",
        "harness": null,
        "effort": null,
        "score": 41.604,
        "costUsd": null,
        "uncertainty": {
          "lower": 39.48,
          "upper": 43.728,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "zai/glm-5.1"
          },
          {
            "label": "Standard error",
            "value": "±2.12 points"
          },
          {
            "label": "Mean latency",
            "value": "78 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "zai-glm-5-2",
        "label": "zai/glm-5.2",
        "model": "zai/glm-5.2",
        "provider": "Zhipu AI",
        "harness": null,
        "effort": null,
        "score": 40.771,
        "costUsd": null,
        "uncertainty": {
          "lower": 38.605000000000004,
          "upper": 42.937,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "zai/glm-5.2"
          },
          {
            "label": "Standard error",
            "value": "±2.17 points"
          },
          {
            "label": "Mean latency",
            "value": "95 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      },
      {
        "id": "zai-glm-5-3",
        "label": "zai/glm-5.3",
        "model": "zai/glm-5.3",
        "provider": "Zhipu AI",
        "harness": null,
        "effort": "max",
        "score": 42.864,
        "costUsd": null,
        "uncertainty": {
          "lower": 40.753,
          "upper": 44.974999999999994,
          "label": "±1 standard error"
        },
        "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
        "details": [
          {
            "label": "Publisher model id",
            "value": "zai/glm-5.3"
          },
          {
            "label": "Standard error",
            "value": "±2.11 points"
          },
          {
            "label": "Mean latency",
            "value": "169 seconds per test"
          },
          {
            "label": "Cost",
            "value": "Not presented as comparable by the publisher"
          },
          {
            "label": "Evaluation mode",
            "value": "One-shot · single response"
          }
        ]
      }
    ]
  }
}
