{
  "source": "https://clinicalbenchmarks.ai",
  "license": "CC BY 4.0",
  "citation": "Clinical Benchmarks index, 2026-08-16 snapshot. https://clinicalbenchmarks.ai.",
  "updated": "2026-08-16",
  "benchmarks": [
    {
      "slug": "healthbench-professional",
      "name": "HealthBench Professional",
      "publisher": "OpenAI",
      "released": "2026-04",
      "unit": "525 physician-authored tasks",
      "scale": "0 to 1, higher is better",
      "description": "525 tasks that physicians picked out of 15,079 real workplace AI conversations, spanning care consults, clinical documentation, and medical research, each judged on a rubric physicians wrote for it.",
      "basis": "independent-run",
      "sourceName": "healthbenchprofessional.com",
      "sourceUrl": "https://healthbenchprofessional.com",
      "ours": "https://healthbenchprofessional.com",
      "lastUpdate": "2026-08",
      "notes": "Grader GPT-5.4 at low reasoning effort with a length adjustment. Physician-written responses score 0.437 on the same rubrics.",
      "results": [
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "0.660",
          "scoreNumeric": 0.66,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "0.605",
          "scoreNumeric": 0.605,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "0.598",
          "scoreNumeric": 0.598,
          "date": "2026-08",
          "config": ""
        }
      ],
      "unitShort": "525 tasks",
      "publisherShort": "OpenAI",
      "sourceShort": "healthbenchprofessional.com",
      "fieldSize": 9,
      "category": "rubric"
    },
    {
      "slug": "healthbench-hard",
      "name": "HealthBench Hard",
      "publisher": "OpenAI",
      "released": "2025-05",
      "unit": "1,000 conversations",
      "scale": "0 to 1, higher is better",
      "description": "The bottom fifth of HealthBench: 1,000 conversations where frontier models failed most at the May 2025 release, still graded on the original physician-written rubrics.",
      "basis": "mixed",
      "sourceName": "healthbenchhard.ai",
      "sourceUrl": "https://healthbenchhard.ai",
      "ours": "https://healthbenchhard.ai",
      "lastUpdate": "2026-08",
      "notes": "Default grader GPT-4.1. Best score at the May 2025 release was o3's 0.320.",
      "results": [
        {
          "model": "Muse Spark",
          "lab": "Meta",
          "scoreDisplay": "0.428",
          "scoreNumeric": 0.428,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "0.331",
          "scoreNumeric": 0.331,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "GPT-5.6 Terra",
          "lab": "OpenAI",
          "scoreDisplay": "0.327",
          "scoreNumeric": 0.327,
          "date": "2026-08",
          "config": ""
        }
      ],
      "unitShort": "1,000 conversations",
      "publisherShort": "OpenAI",
      "sourceShort": "healthbenchhard.ai",
      "fieldSize": 9,
      "category": "rubric"
    },
    {
      "slug": "healthbench",
      "name": "HealthBench",
      "publisher": "OpenAI",
      "released": "2025-05",
      "unit": "5,000 conversations",
      "scale": "0-100 rubric-point percentage (some sites display 0-1), higher better; length-adjusted and unadjusted variants",
      "description": "5,000 realistic multi-turn health conversations graded against physician-written rubrics (48,562 criteria) covering accuracy, completeness, context awareness, communication, and instruction following. OpenAI now also reports a length-adjusted variant that penalizes verbosity.",
      "basis": "mixed",
      "sourceName": "OpenAI Deployment Safety Hub (GPT-5.6 system card + August 2026 updates); benchlm.ai and llm-stats.com mirror",
      "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "Cross-source rows are not directly comparable: Anthropic reports a raw score under its own protocol, OpenAI leads with length-adjusted numbers at maximum reasoning effort and reports production Instant settings separately, and Baichuan ran its competitors itself. The benchmark's official grader is GPT-4.1. OpenAI has said the parent set is approaching a noise ceiling for frontier models and points to HealthBench Professional for continued measurement.",
      "confidence": "verified",
      "results": [
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "67.1",
          "scoreNumeric": 67.1,
          "date": "2026-08",
          "config": "raw/unadjusted, Anthropic system-card protocol (max effort, no tools, five trials); length-adjusted 57.8; mirrored on benchlm.ai"
        },
        {
          "model": "Baichuan-M3",
          "lab": "Baichuan",
          "scoreDisplay": "65.1",
          "scoreNumeric": 65.1,
          "date": "2026-02",
          "config": "self-run in Baichuan-M3 paper (arXiv 2602.06570)"
        },
        {
          "model": "GPT-5.2-High",
          "lab": "OpenAI",
          "scoreDisplay": "63.3",
          "scoreNumeric": 63.3,
          "date": "2026-02",
          "config": "as run by Baichuan in the M3 paper, not OpenAI-reported"
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "57.0",
          "scoreNumeric": 57,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort (55.6 unadjusted), GPT-5.6 system card 2026-07-09"
        },
        {
          "model": "GPT-5.6 Terra",
          "lab": "OpenAI",
          "scoreDisplay": "57.0",
          "scoreNumeric": 57,
          "date": "2026-06",
          "config": "length-adjusted (58.7 unadjusted), max reasoning effort"
        },
        {
          "model": "GPT-5.5",
          "lab": "OpenAI",
          "scoreDisplay": "56.5",
          "scoreNumeric": 56.5,
          "date": "2026-06",
          "config": "length-adjusted (58.4 unadjusted), comparison row in GPT-5.6 system card"
        },
        {
          "model": "GPT-5.6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "55.8",
          "scoreNumeric": 55.8,
          "date": "2026-06",
          "config": "length-adjusted (55.4 unadjusted), max reasoning effort"
        },
        {
          "model": "GPT-5.6 Sol (August)",
          "lab": "OpenAI",
          "scoreDisplay": "55.0",
          "scoreNumeric": 55,
          "date": "2026-08",
          "config": "ChatGPT production/Instant deployment setting, length-adjusted (52.1 unadjusted), GPT-5.6 August Updates PDF 2026-08-06"
        }
      ],
      "unitShort": "5,000 conversations",
      "publisherShort": "OpenAI",
      "sourceShort": "OpenAI Deployment Safety Hub",
      "fieldSize": 8,
      "category": "rubric"
    },
    {
      "slug": "health-optimization-bench",
      "name": "Health Optimization Bench",
      "publisher": "Health Optimization Bench",
      "publisherShort": "healthoptimizationbench.com",
      "released": "2026-08",
      "unit": "89 released tasks (v1 evidence suite)",
      "unitShort": "89 tasks",
      "scale": "0-100 rubric credit, higher better",
      "description": "Frontier models on hard, freshness-dependent questions in preventive and optimization medicine. Every task is written against a primary source, audited by model families that did not author it, and scored blind by a panel of independent families. The v1 release set covers incretin therapeutics evidence.",
      "basis": "independent-run",
      "sourceName": "healthoptimizationbench.com",
      "sourceShort": "healthoptimizationbench.com",
      "sourceUrl": "https://healthoptimizationbench.com",
      "ours": "https://healthoptimizationbench.com",
      "fieldSize": 15,
      "lastUpdate": "2026-08",
      "notes": "Cross-family authoring with blind three-family panel grading; the authoring family never grades its own task, and 95 percent bootstrap confidence intervals accompany every score on the site.",
      "confidence": "verified",
      "category": "rubric",
      "results": [
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "83.8",
          "scoreNumeric": 83.8,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Grok 4.6",
          "lab": "xAI",
          "scoreDisplay": "81.3",
          "scoreNumeric": 81.3,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "78.3",
          "scoreNumeric": 78.3,
          "date": "2026-08",
          "config": ""
        }
      ]
    },
    {
      "slug": "mast",
      "name": "MAST (Medical AI Superintelligence Test)",
      "publisher": "ARISE AI Research Network (multi-institutional)",
      "released": "2026-08",
      "unit": "composite of 6 component benchmarks; 11 models",
      "scale": "percentage composite, higher better",
      "description": "Composite score across curated clinical benchmarks spanning diagnostic reasoning, management reasoning, safety, multimodal images, multimodal radiology, and agentic capability. Components: First Do NOHARM v2, SCT-Bench, MedAgentBench v2, PhysicianBench, ReXrank Mini, CPC-Bench.",
      "basis": "independent-run",
      "sourceName": "ARISE MAST leaderboard",
      "sourceUrl": "https://arise-ai.org/mast",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "The board is marked as a preview and was last updated August 15, 2026; component-level breakdowns are published only for First, Do NOHARM v2. Scores may move before the full release.",
      "confidence": "verified",
      "results": [
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "60.2%",
          "scoreNumeric": 60.2,
          "date": "2026-08",
          "config": "MAST in preview; 'exact scores may change'"
        },
        {
          "model": "Kimi K3",
          "lab": "Moonshot AI",
          "scoreDisplay": "60.1%",
          "scoreNumeric": 60.1,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Gemini 3.6 Flash",
          "lab": "Google",
          "scoreDisplay": "59.3%",
          "scoreNumeric": 59.3,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "58.9%",
          "scoreNumeric": 58.9,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Qwen3.5 397B A17B",
          "lab": "Alibaba",
          "scoreDisplay": "57.9%",
          "scoreNumeric": 57.9,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "57.1%",
          "scoreNumeric": 57.1,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Claude Sonnet 5",
          "lab": "Anthropic",
          "scoreDisplay": "56.6%",
          "scoreNumeric": 56.6,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Grok 4.3",
          "lab": "xAI",
          "scoreDisplay": "53.7%",
          "scoreNumeric": 53.7,
          "date": "2026-08",
          "config": ""
        }
      ],
      "unitShort": "6-benchmark composite",
      "publisherShort": "ARISE AI Research Network",
      "sourceShort": "ARISE MAST leaderboard",
      "fieldSize": 11,
      "category": "composite"
    },
    {
      "slug": "medhelm",
      "name": "MedHELM",
      "publisher": "Stanford CRFM / HAI and multi-institution collaborators",
      "released": "2025-02",
      "unit": "121 tasks / 31 datasets",
      "scale": "mean win rate 0-1, higher better",
      "description": "Holistic evaluation of LLMs on 121 clinical tasks across 5 categories and 22 subcategories (31 datasets) in a clinician-validated taxonomy; ranked by mean win rate.",
      "basis": "official-leaderboard",
      "sourceName": "MedHELM leaderboard (medhelm.org), v5.0.0",
      "sourceUrl": "https://medhelm.org/",
      "ours": null,
      "lastUpdate": "2026-05",
      "notes": "Version 5.0.0, last updated May 14, 2026, run by the Stanford-led maintainers on a roughly quarterly cadence. No Claude 5 family or GPT-5.6 rows yet. Mean win rate is relative to the evaluated cohort, so scores shift whenever the model set changes.",
      "confidence": "verified",
      "results": [
        {
          "model": "Gemini 3.1 Pro (Preview)",
          "lab": "Google",
          "scoreDisplay": "0.652",
          "scoreNumeric": 0.652,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "Gemini 3.5 Flash",
          "lab": "Google",
          "scoreDisplay": "0.642",
          "scoreNumeric": 0.642,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "Muse Spark (2026-04-08)",
          "lab": "Meta",
          "scoreDisplay": "0.621",
          "scoreNumeric": 0.621,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "GPT-5.4 mini",
          "lab": "OpenAI",
          "scoreDisplay": "0.552",
          "scoreNumeric": 0.552,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "GPT-5.4 (2026-03-05)",
          "lab": "OpenAI",
          "scoreDisplay": "0.538",
          "scoreNumeric": 0.538,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "Gemini 2.5 Pro",
          "lab": "Google",
          "scoreDisplay": "0.529",
          "scoreNumeric": 0.529,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "DeepSeek R1",
          "lab": "DeepSeek",
          "scoreDisplay": "0.485",
          "scoreNumeric": 0.485,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "Claude 4.6 Opus",
          "lab": "Anthropic",
          "scoreDisplay": "0.456",
          "scoreNumeric": 0.456,
          "date": "2026-05",
          "config": "",
          "alias": "Claude Opus 4.6"
        },
        {
          "model": "Claude 3.7 Sonnet",
          "lab": "Anthropic",
          "scoreDisplay": "0.45",
          "scoreNumeric": 0.45,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "Gemini 2.0 Flash",
          "lab": "Google",
          "scoreDisplay": "0.342",
          "scoreNumeric": 0.342,
          "date": "2026-05",
          "config": ""
        }
      ],
      "unitShort": "121 tasks",
      "publisherShort": "Stanford CRFM",
      "sourceShort": "MedHELM leaderboard",
      "fieldSize": 11,
      "category": "composite"
    },
    {
      "slug": "first-do-noharm",
      "name": "First, Do NOHARM (v2)",
      "publisher": "Stanford/Harvard-led consortium (50+ researchers incl. 29 board-certified physicians); hosted by ARISE",
      "released": "2025-12",
      "unit": "1,100 consultation cases, 10 specialties, 12,747 expert annotations on 4,249 management options",
      "scale": "percentage safety score, higher better",
      "description": "Frequency and severity of potentially harmful errors in LLM-generated medical consultation recommendations (Numerous Options Harm Assessment for Risk in Medicine); primary-care-to-specialist consults.",
      "basis": "official-leaderboard",
      "sourceName": "ARISE MAST technical leaderboard",
      "sourceUrl": "https://arise-ai.org/mast/technical",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "Paper: arXiv 2512.01241. The v1 study found potential for severe harm in up to 24.6 percent of directly applied recommendations, with errors of omission behind more than 80 percent of the severe cases.",
      "confidence": "verified",
      "results": [
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "74.6%",
          "scoreNumeric": 74.6,
          "date": "2026-08",
          "config": "v2 run on ARISE; 19 models on the board"
        },
        {
          "model": "Kimi K3",
          "lab": "Moonshot AI",
          "scoreDisplay": "74.0%",
          "scoreNumeric": 74,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "70.1%",
          "scoreNumeric": 70.1,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "GPT-5.5",
          "lab": "OpenAI",
          "scoreDisplay": "70.0%",
          "scoreNumeric": 70,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "62.6%",
          "scoreNumeric": 62.6,
          "date": "2026-08",
          "config": ""
        }
      ],
      "unitShort": "1,100 consultation cases",
      "publisherShort": "Stanford/Harvard consortium",
      "sourceShort": "ARISE MAST technical leaderboard",
      "fieldSize": 19,
      "category": "safety"
    },
    {
      "slug": "healthagentbench",
      "name": "HealthAgentBench",
      "publisher": "Microsoft Research",
      "released": "2026-07",
      "unit": "54 tasks across 7 environments; 162 trials (3 attempts per task)",
      "scale": "mean task success rate, 0-100%, higher better; cost per task also reported",
      "description": "Agentic task success in realistic terminal-based healthcare environments built from real clinical artifacts; evaluates agent harnesses (Claude Code, Codex, Copilot) end to end, not bare models.",
      "basis": "independent-run",
      "sourceName": "HealthAgentBench leaderboard (Microsoft GitHub Pages)",
      "sourceUrl": "https://microsoft.github.io/HealthAgentBench/",
      "ours": null,
      "lastUpdate": "2026-07",
      "notes": "Paper: arXiv 2606.31179. The rows are agent harnesses rather than bare models, and the board is run by Microsoft Research; Copilot, Microsoft's own harness, does not top it.",
      "confidence": "verified",
      "results": [
        {
          "model": "Claude Code (Opus 5)",
          "lab": "Anthropic",
          "scoreDisplay": "55%",
          "scoreNumeric": 55,
          "date": "2026-07",
          "config": "$3.3/task; harness+model evaluated jointly"
        },
        {
          "model": "Codex (GPT-5.6-sol)",
          "lab": "OpenAI",
          "scoreDisplay": "45%",
          "scoreNumeric": 45,
          "date": "2026-07",
          "config": "$5.2/task"
        },
        {
          "model": "Codex (GPT 5.5)",
          "lab": "OpenAI",
          "scoreDisplay": "42%",
          "scoreNumeric": 42,
          "date": "2026-07",
          "config": "$2.8/task"
        },
        {
          "model": "Copilot (Opus 4.8)",
          "lab": "Microsoft/Anthropic",
          "scoreDisplay": "36%",
          "scoreNumeric": 36,
          "date": "2026-07",
          "config": "$3.1/task"
        },
        {
          "model": "Copilot (GPT 5.5)",
          "lab": "Microsoft/OpenAI",
          "scoreDisplay": "35%",
          "scoreNumeric": 35,
          "date": "2026-07",
          "config": "$2.6/task"
        },
        {
          "model": "Claude Code (Opus 4.8)",
          "lab": "Anthropic",
          "scoreDisplay": "32%",
          "scoreNumeric": 32,
          "date": "2026-07",
          "config": "$4.0/task"
        }
      ],
      "unitShort": "54 agentic tasks",
      "publisherShort": "Microsoft Research",
      "sourceShort": "HealthAgentBench leaderboard",
      "fieldSize": 12,
      "category": "agentic"
    },
    {
      "slug": "chi-bench",
      "name": "CHI-Bench",
      "publisher": "actAVA.ai",
      "released": "2026-05",
      "unit": "75 workflows (25 per domain), 21 healthcare applications, 200+ MCP tools",
      "scale": "pass@1 with binary 0/1 reward, higher better",
      "description": "Long-horizon US healthcare operations workflows for agents (prior authorization, utilization management, care management), 60-80 step tasks across 4-6 stages, judged by deterministic unit tests plus an LLM judge for evidence grounding, consent, and cross-stage consistency.",
      "basis": "mixed",
      "sourceName": "CHI-Bench leaderboard (actAVA)",
      "sourceUrl": "https://actava.ai/benchmarks/leaderboards",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "Released May 20, 2026; updated August 12, 2026 across 45 harness configurations. The launch report led with reliability, not capability: no agent stayed above 20 percent across three identical runs. Harness choice matters as much as model choice, and the board accepts community submissions, so rows mix author-run and submitted results.",
      "confidence": "verified",
      "results": [
        {
          "model": "erius + claude-opus-5",
          "lab": "Humana (harness) / Anthropic (model)",
          "scoreDisplay": "54.7%",
          "scoreNumeric": 54.7,
          "date": "2026-08",
          "config": "community-submitted harness config validated by automated workspace judge"
        },
        {
          "model": "erius + claude-opus-4-8",
          "lab": "Humana / Anthropic",
          "scoreDisplay": "37.3%",
          "scoreNumeric": 37.3,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "claude-code + claude-opus-5",
          "lab": "Anthropic",
          "scoreDisplay": "37.3%",
          "scoreNumeric": 37.3,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "claude-code + claude-opus-4-8",
          "lab": "Anthropic",
          "scoreDisplay": "33.3%",
          "scoreNumeric": 33.3,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "claude-code + claude-opus-4-6",
          "lab": "Anthropic",
          "scoreDisplay": "28.0%",
          "scoreNumeric": 28,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "claude-code + claude-sonnet-4-6",
          "lab": "Anthropic",
          "scoreDisplay": "26.2%",
          "scoreNumeric": 26.2,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "codex + gpt-5.6-sol",
          "lab": "OpenAI",
          "scoreDisplay": "25.3%",
          "scoreNumeric": 25.3,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "openai-agents + kimi-k3",
          "lab": "Moonshot AI",
          "scoreDisplay": "25.3%",
          "scoreNumeric": 25.3,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "claude-code + claude-opus-4-7",
          "lab": "Anthropic",
          "scoreDisplay": "24.4%",
          "scoreNumeric": 24.4,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "claude-code + claude-fable-5",
          "lab": "Anthropic",
          "scoreDisplay": "24.0%",
          "scoreNumeric": 24,
          "date": "2026-08",
          "config": ""
        }
      ],
      "unitShort": "75 operations workflows",
      "publisherShort": "actAVA",
      "sourceShort": "CHI-Bench leaderboard",
      "fieldSize": 45,
      "category": "agentic"
    },
    {
      "slug": "medcode",
      "name": "MedCode (Vals AI)",
      "publisher": "Vals AI (dataset with Protege)",
      "released": "2026-02",
      "unit": "2,755 patient records",
      "scale": "percentage accuracy 0-100, higher better",
      "description": "ICD-10-CM diagnosis coding for entire hospital stays: models assign primary and secondary codes from discharge summaries plus progress/consult notes; ground truth double-annotated by certified professional coders.",
      "basis": "independent-run",
      "sourceName": "Vals AI MedCode leaderboard",
      "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "Vals AI runs every model itself; 85 models on the board, last updated August 15, 2026. Vals AI pairs this board with MedScribe, and its writeup notes that coding accuracy lags documentation quality.",
      "confidence": "verified",
      "results": [
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "63.57%",
          "scoreNumeric": 63.57,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Gemini 3.1 Pro Preview (02/26)",
          "lab": "Google",
          "scoreDisplay": "59.06%",
          "scoreNumeric": 59.06,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "56.07%",
          "scoreNumeric": 56.07,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Gemini 3 Flash (12/25)",
          "lab": "Google",
          "scoreDisplay": "55.92%",
          "scoreNumeric": 55.92,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Gemini 3.5 Flash",
          "lab": "Google",
          "scoreDisplay": "55.83%",
          "scoreNumeric": 55.83,
          "date": "2026-08",
          "config": ""
        }
      ],
      "unitShort": "2,755 patient records",
      "publisherShort": "Vals AI",
      "sourceShort": "Vals AI MedCode leaderboard",
      "fieldSize": 85,
      "category": "documentation"
    },
    {
      "slug": "medscribe",
      "name": "MedScribe (Vals AI)",
      "publisher": "Vals AI (dataset with Protege)",
      "released": "2026-02",
      "unit": "100 rubric-scored SOAP-note cases",
      "scale": "percentage accuracy 0-100, higher better",
      "description": "Clinical documentation support: quality of SOAP notes generated from clinical visits, scored against rubrics for documentation quality and compliance.",
      "basis": "independent-run",
      "sourceName": "Vals AI MedScribe leaderboard",
      "sourceUrl": "https://www.vals.ai/benchmarks/medscribe",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "Vals AI self-runs; 84 models, last updated August 15, 2026. Top scores cluster near 90, so the leaders sit close to the ceiling, and exact decimals below first place are not always displayed.",
      "confidence": "partial",
      "results": [
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "90.99%",
          "scoreNumeric": 90.99,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Muse Spark 1.2",
          "lab": "Meta",
          "scoreDisplay": "~90",
          "scoreNumeric": 90,
          "date": "2026-08",
          "config": "rank 2; exact value not displayed by the source"
        },
        {
          "model": "Muse Spark 1.1",
          "lab": "Meta",
          "scoreDisplay": "~90",
          "scoreNumeric": 90,
          "date": "2026-08",
          "config": "rank 3; exact value not displayed by the source"
        },
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "~90",
          "scoreNumeric": 90,
          "date": "2026-08",
          "config": "rank 4; exact value not displayed by the source"
        },
        {
          "model": "GPT 5.1",
          "lab": "OpenAI",
          "scoreDisplay": "88.09%",
          "scoreNumeric": 88.09,
          "date": "2026-02",
          "config": "at-release leader (Feb 2026 blog)"
        },
        {
          "model": "Claude Opus 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "86.74%",
          "scoreNumeric": 86.74,
          "date": "2026-02",
          "config": "at-release (Feb 2026 blog)"
        }
      ],
      "unitShort": "100 SOAP-note cases",
      "publisherShort": "Vals AI",
      "sourceShort": "Vals AI MedScribe leaderboard",
      "fieldSize": 84,
      "category": "documentation"
    },
    {
      "slug": "medxpertqa-mm",
      "name": "MedXpertQA (MM)",
      "publisher": "TsinghuaC3I (Tsinghua University)",
      "released": "2025-01",
      "unit": "2,000 multimodal questions (MM subset)",
      "scale": "percentage accuracy 0-100, higher better",
      "description": "Expert-level multimodal medical multiple-choice QA covering clinical images (X-ray, histology, dermatology, charts) across 17 specialties; MM subset of the 4,460-question MedXpertQA benchmark.",
      "basis": "vendor-reported",
      "sourceName": "benchlm.ai mirror of Meta's Muse Spark evaluation",
      "sourceUrl": "https://benchlm.ai/benchmarks/medxpertqamm",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "The current frontier table is vendor-reported, from Meta's Muse Spark launch evaluation, mirrored display-only on benchlm.ai. The Text subset has no comparable current table.",
      "confidence": "verified",
      "results": [
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "81.3%",
          "scoreNumeric": 81.3,
          "date": "2026-08",
          "config": "as reported in Meta's Muse Spark eval, mirrored by benchlm"
        },
        {
          "model": "Qwen3.8 Max",
          "lab": "Alibaba",
          "scoreDisplay": "80.4%",
          "scoreNumeric": 80.4,
          "date": "2026-08",
          "config": "same source"
        },
        {
          "model": "Muse Spark",
          "lab": "Meta",
          "scoreDisplay": "78.4%",
          "scoreNumeric": 78.4,
          "date": "2026-08",
          "config": "same source"
        },
        {
          "model": "GPT-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "77.1%",
          "scoreNumeric": 77.1,
          "date": "2026-08",
          "config": "same source"
        },
        {
          "model": "Qwen3.7 Plus",
          "lab": "Alibaba",
          "scoreDisplay": "71.0%",
          "scoreNumeric": 71,
          "date": "2026-08",
          "config": "same source"
        },
        {
          "model": "Grok 4.20",
          "lab": "xAI",
          "scoreDisplay": "65.8%",
          "scoreNumeric": 65.8,
          "date": "2026-08",
          "config": "same source"
        },
        {
          "model": "Claude Opus 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "64.8%",
          "scoreNumeric": 64.8,
          "date": "2026-08",
          "config": "same source"
        },
        {
          "model": "Gemma 4 12B",
          "lab": "Google",
          "scoreDisplay": "48.7%",
          "scoreNumeric": 48.7,
          "date": "2026-08",
          "config": "same source"
        }
      ],
      "unitShort": "2,000 multimodal questions",
      "publisherShort": "Tsinghua University",
      "sourceShort": "benchlm.ai mirror",
      "fieldSize": 8,
      "category": "knowledge"
    },
    {
      "slug": "artificial-analysis-healthcare",
      "name": "Artificial Analysis Healthcare & Medical Index",
      "publisher": "Artificial Analysis",
      "released": "2026-08",
      "unit": "27 of 159 models scored; composite of 4 underlying benchmarks",
      "scale": "index score, higher better",
      "description": "Weighted composite for healthcare and medical work: Medical & Health Knowledge 35%, Agentic Knowledge Work 25%, Non-Hallucination 15%, Reasoning 15%, Agentic Customer Interaction 10%, drawn from AA-Omniscience, GDPval-AA v2, Humanity's Last Exam, and tau3-Banking.",
      "basis": "independent-run",
      "sourceName": "Artificial Analysis healthcare capability page",
      "sourceUrl": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "A weighted composite of general-purpose benchmarks tilted toward health-relevant slices rather than purpose-built clinical tasks, which is why this index files it as a capability index. The page differentiates only its top scores numerically.",
      "confidence": "verified",
      "results": [
        {
          "model": "Claude Opus 5 (Adaptive Reasoning, Max Effort)",
          "lab": "Anthropic",
          "scoreDisplay": "51",
          "scoreNumeric": 51,
          "date": "2026-08",
          "config": "all underlying benchmarks run independently by Artificial Analysis"
        },
        {
          "model": "Claude Opus 5 (Adaptive Reasoning, Xhigh Effort)",
          "lab": "Anthropic",
          "scoreDisplay": "51",
          "scoreNumeric": 51,
          "date": "2026-08",
          "config": ""
        },
        {
          "model": "Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",
          "lab": "Anthropic",
          "scoreDisplay": "51",
          "scoreNumeric": 51,
          "date": "2026-08",
          "config": ""
        }
      ],
      "unitShort": "4-benchmark composite",
      "publisherShort": "Artificial Analysis",
      "sourceShort": "Artificial Analysis",
      "fieldSize": 27,
      "category": "composite"
    },
    {
      "slug": "physicianbench",
      "name": "PhysicianBench",
      "publisher": "Academic team (Ruoqi Liu, Imran Q. Mohiuddin et al., arXiv 2605.02240); also a MAST component",
      "released": "2026-05",
      "unit": "100 real-world clinical tasks, 21 specialties, 670 structured checkpoints (~27 tool calls per task)",
      "scale": "pass@1 success rate %, higher better (3 independent runs; Pass^3 also reported)",
      "description": "LLM agents on long-horizon composite physician workflows inside real EHR environments, with execution-grounded verification against actual EHR systems via standard commercial APIs.",
      "basis": "independent-run",
      "sourceName": "PhysicianBench paper (Table 2)",
      "sourceUrl": "https://arxiv.org/abs/2605.02240",
      "ours": null,
      "lastUpdate": "2026-05",
      "notes": "Scores come from the paper; there is no standalone public leaderboard, and the results also feed the ARISE MAST composite. The lower half of the table falls steeply, with several agents near 1 percent on Pass^3 consistency.",
      "confidence": "verified",
      "results": [
        {
          "model": "GPT-5.5",
          "lab": "OpenAI",
          "scoreDisplay": "46.3 ± 1.2",
          "scoreNumeric": 46.3,
          "date": "2026-05",
          "config": "pass@1; Pass^3 28.0"
        },
        {
          "model": "Claude Opus 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "31.7 ± 2.3",
          "scoreNumeric": 31.7,
          "date": "2026-05",
          "config": "Pass^3 18.0"
        },
        {
          "model": "Claude Opus 4.7",
          "lab": "Anthropic",
          "scoreDisplay": "29.3 ± 2.5",
          "scoreNumeric": 29.3,
          "date": "2026-05",
          "config": "Pass^3 18.0"
        },
        {
          "model": "GPT-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "27.7 ± 1.5",
          "scoreNumeric": 27.7,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "Claude Sonnet 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "23.0 ± 2.6",
          "scoreNumeric": 23,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "Kimi-K2.6",
          "lab": "Moonshot AI",
          "scoreDisplay": "17.0 ± 2.6",
          "scoreNumeric": 17,
          "date": "2026-05",
          "config": "open source"
        },
        {
          "model": "Qwen3.6-Plus",
          "lab": "Alibaba",
          "scoreDisplay": "13.7 ± 4.0",
          "scoreNumeric": 13.7,
          "date": "2026-05",
          "config": ""
        },
        {
          "model": "Gemini Pro 3.1",
          "lab": "Google",
          "scoreDisplay": "6.0 ± 1.0",
          "scoreNumeric": 6,
          "date": "2026-05",
          "config": "",
          "alias": "Gemini 3.1 Pro"
        }
      ],
      "unitShort": "100 clinical tasks",
      "publisherShort": "academic team",
      "sourceShort": "PhysicianBench paper",
      "fieldSize": 13,
      "category": "agentic"
    },
    {
      "slug": "ehr-complex",
      "name": "EHR-Complex",
      "publisher": "Academic team (Qiao et al., Ant Group-affiliated; arXiv 2606.23301)",
      "released": "2026-06",
      "unit": "~52,000 tasks (3,915-task test set) over 365K patients, 31 tables, 500M+ records",
      "scale": "exact-match accuracy, 0-1, higher better",
      "description": "Agentic clinical reasoning over MIMIC-IV EHR databases via SQL and Python across six clinical intents, at patient and population level with temporal evidence paths.",
      "basis": "independent-run",
      "sourceName": "EHR-Complex paper",
      "sourceUrl": "https://arxiv.org/abs/2606.23301",
      "ours": null,
      "lastUpdate": "2026-06",
      "notes": "Scores come from the paper, which reports both a headline 12-model evaluation and human-validated configurations; rows here mix the two, labeled in the config column. Consistency drops below 50 percent at Pass^4 for nearly every model.",
      "confidence": "verified",
      "results": [
        {
          "model": "GPT-5.4 (high reasoning)",
          "lab": "OpenAI",
          "scoreDisplay": "0.65",
          "scoreNumeric": 0.65,
          "date": "2026-06",
          "config": "average over 12 intent columns; run as human-validation configuration, not in the headline 12-model table"
        },
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "0.63",
          "scoreNumeric": 0.63,
          "date": "2026-06",
          "config": "validation configuration"
        },
        {
          "model": "Kimi-K2.5",
          "lab": "Moonshot AI",
          "scoreDisplay": "0.62",
          "scoreNumeric": 0.62,
          "date": "2026-06",
          "config": "headline 12-model evaluation, top open-weight"
        },
        {
          "model": "Qwen3.5-397B",
          "lab": "Alibaba",
          "scoreDisplay": "0.62",
          "scoreNumeric": 0.62,
          "date": "2026-06",
          "config": "headline evaluation"
        },
        {
          "model": "GPT-5.4 (low reasoning)",
          "lab": "OpenAI",
          "scoreDisplay": "0.58",
          "scoreNumeric": 0.58,
          "date": "2026-06",
          "config": "validation configuration"
        },
        {
          "model": "Claude Sonnet 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "0.36",
          "scoreNumeric": 0.36,
          "date": "2026-06",
          "config": "validation configuration"
        }
      ],
      "unitShort": "3,915-task test set",
      "publisherShort": "academic team",
      "sourceShort": "EHR-Complex paper",
      "fieldSize": 12,
      "category": "agentic"
    },
    {
      "slug": "whbench",
      "name": "WHBench",
      "publisher": "Independent researchers (Maurya, Govindgari, Kumar)",
      "released": "2026-04",
      "unit": "47 scenarios / 3,100 scored responses across 22 models",
      "scale": "mean normalized percentage 0-100, higher better",
      "description": "Women's health: 47 expert-crafted scenarios across 10 topics graded on a 23-criterion rubric for clinical accuracy, safety, equity, and guideline adherence; targets failure modes like outdated guidelines, unsafe omissions, dosing errors, equity blind spots.",
      "basis": "independent-run",
      "sourceName": "arXiv paper (v2 revised 2026-07-23)",
      "sourceUrl": "https://arxiv.org/abs/2604.00024",
      "ours": null,
      "lastUpdate": "2026-07",
      "notes": "An academic study with expert validation rather than a live leaderboard; the model set was frozen in March 2026, before GPT-5.6 and the Claude 5 family shipped.",
      "confidence": "verified",
      "results": [
        {
          "model": "Claude Opus 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "72.1%",
          "scoreNumeric": 72.1,
          "date": "2026-07",
          "config": "95% CI 69.6-74.4; evaluations run March 2026"
        },
        {
          "model": "Claude Sonnet 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "67.1%",
          "scoreNumeric": 67.1,
          "date": "2026-07",
          "config": "95% CI 64.5-69.6"
        },
        {
          "model": "GPT-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "66.8%",
          "scoreNumeric": 66.8,
          "date": "2026-07",
          "config": "95% CI 64.5-69.2"
        },
        {
          "model": "Gemini 3 Flash Preview",
          "lab": "Google",
          "scoreDisplay": "64.7%",
          "scoreNumeric": 64.7,
          "date": "2026-07",
          "config": ""
        },
        {
          "model": "GPT-4.1",
          "lab": "OpenAI",
          "scoreDisplay": "51.8%",
          "scoreNumeric": 51.8,
          "date": "2026-07",
          "config": ""
        },
        {
          "model": "GPT-4o",
          "lab": "OpenAI",
          "scoreDisplay": "44.6%",
          "scoreNumeric": 44.6,
          "date": "2026-07",
          "config": ""
        }
      ],
      "unitShort": "47 scenarios",
      "publisherShort": "academic team",
      "sourceShort": "WHBench paper",
      "fieldSize": 22,
      "category": "rubric"
    },
    {
      "slug": "healthadminbench",
      "name": "HealthAdminBench",
      "publisher": "Kinetic Systems (with Stanford Hospital domain experts)",
      "released": "2026-04",
      "unit": "135 tasks / 1,698 rubric-scored subtasks",
      "scale": "percentage end-to-end task success 0-100, higher better",
      "description": "End-to-end task success of computer-use LLM agents on healthcare administration workflows: prior authorizations, denial appeals, and DME ordering; success requires completing every subtask in a task.",
      "basis": "independent-run",
      "sourceName": "Kinetic Systems blog",
      "sourceUrl": "https://kineticsystems.ai/blog/healthadminbench-automating-healthcare-administration-with-computer-use-agents",
      "ours": null,
      "lastUpdate": "2026-04",
      "notes": "Run by Kinetic Systems' research team: independent of the model vendors, though published by a company selling healthcare-admin automation. No GPT-5.6 or Claude 5 rows yet, and no stated refresh cadence.",
      "confidence": "verified",
      "results": [
        {
          "model": "Claude Opus 4.6 (computer-use agent)",
          "lab": "Anthropic",
          "scoreDisplay": "36.3%",
          "scoreNumeric": 36.3,
          "date": "2026-04",
          "config": "screenshot-only, detailed prompting; subtask rate ~82%"
        },
        {
          "model": "GPT-5.4 (computer-use agent)",
          "lab": "OpenAI",
          "scoreDisplay": "26.7%",
          "scoreNumeric": 26.7,
          "date": "2026-04",
          "config": "screenshot-only, detailed prompting; subtask rate 82.8%"
        },
        {
          "model": "Kimi K2.5",
          "lab": "Moonshot AI",
          "scoreDisplay": "15.6%",
          "scoreNumeric": 15.6,
          "date": "2026-04",
          "config": "screenshot-only, detailed prompting"
        },
        {
          "model": "Qwen 3.5",
          "lab": "Alibaba",
          "scoreDisplay": "13.3%",
          "scoreNumeric": 13.3,
          "date": "2026-04",
          "config": "screenshot-only, detailed prompting"
        },
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "11.9%",
          "scoreNumeric": 11.9,
          "date": "2026-04",
          "config": "screenshot-only, detailed prompting"
        }
      ],
      "unitShort": "135 admin tasks",
      "publisherShort": "Kinetic Systems",
      "sourceShort": "Kinetic Systems",
      "fieldSize": 5,
      "category": "agentic"
    },
    {
      "slug": "openai-mental-health-evals",
      "name": "OpenAI Dynamic Mental Health Evaluations",
      "publisher": "OpenAI",
      "released": "2026-08",
      "unit": "dynamic simulated conversations (counts not disclosed)",
      "scale": "compliance rate per metric, 0 to 1, higher better; headline number is the mental-health metric",
      "description": "Multi-turn adversarial user simulations for mental health, emotional reliance, and self-harm response quality, where conversations evolve in response to model outputs rather than following fixed scripts.",
      "basis": "vendor-reported",
      "sourceName": "OpenAI GPT-5.6 August Updates (PDF)",
      "sourceUrl": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "An internal OpenAI safety evaluation covering OpenAI models only; it is not independently runnable, and OpenAI notes the error rates are not representative of average production traffic. Listed as a vendor safety eval, not a cross-vendor benchmark.",
      "confidence": "verified",
      "results": [
        {
          "model": "GPT-5.5 Instant (June Update)",
          "lab": "OpenAI",
          "scoreDisplay": "0.991",
          "scoreNumeric": 0.991,
          "date": "2026-08",
          "config": "mental health 0.991, emotional reliance 0.989, self-harm 0.967; measured at lowest reasoning deployment settings"
        },
        {
          "model": "GPT-5.6 Sol (August)",
          "lab": "OpenAI",
          "scoreDisplay": "0.981",
          "scoreNumeric": 0.981,
          "date": "2026-08",
          "config": "mental health 0.981, emotional reliance 0.961, self-harm 0.901; OpenAI flags statistically significant offline self-harm regression vs GPT-5.5 June, not reproduced online"
        },
        {
          "model": "GPT-5.6 Luna (August)",
          "lab": "OpenAI",
          "scoreDisplay": "0.977",
          "scoreNumeric": 0.977,
          "date": "2026-08",
          "config": "mental health 0.977, emotional reliance 0.965, self-harm 0.911"
        }
      ],
      "unitShort": "3 safety metrics",
      "publisherShort": "OpenAI",
      "sourceShort": "OpenAI August Updates",
      "fieldSize": 3,
      "category": "safety"
    }
  ],
  "retired": [
    {
      "name": "HealthBench Consensus",
      "reason": "near-saturated physician-consensus baseline; frontier runs stopped reporting it separately"
    },
    {
      "name": "MedQA / MultiMedQA",
      "reason": "exam-style multiple choice, saturated above 95 percent since 2025; archived by its trackers"
    },
    {
      "name": "AgentClinic",
      "reason": "no public frontier-model results since 2025"
    },
    {
      "name": "CRAFT-MD",
      "reason": "no public frontier-model results since 2025"
    },
    {
      "name": "MedAgentBench",
      "reason": "v2 lives on inside the MAST composite; the standalone board has no current frontier rows"
    },
    {
      "name": "SDBench / MAI-DxO",
      "reason": "Microsoft's 2025 sequential-diagnosis study was not re-run on current models"
    },
    {
      "name": "Open Medical-LLM Leaderboard (Hugging Face)",
      "reason": "built on saturated exam sets; no frontier submissions in 2026"
    },
    {
      "name": "MedArena",
      "reason": "clinician preference arena; ratings pool too thin on current frontier models to quote"
    },
    {
      "name": "AMIE evaluations",
      "reason": "Google DeepMind research prototypes, never opened to cross-vendor comparison"
    },
    {
      "name": "LiveClin, PrIME-LLM, MedMCP-Calc",
      "reason": "single studies with two or fewer current-frontier rows; tracked for a future qualifying update"
    }
  ]
}
