{
  "dataset": "Clinical Benchmarks",
  "kind": "benchmarks",
  "url": "https://clinicalbenchmarks.ai",
  "updated": "2026-09-30",
  "revision": 7165,
  "licence": {
    "name": "CC BY 4.0",
    "url": "https://creativecommons.org/licenses/by/4.0/",
    "attribution": "Clinical Benchmarks (clinicalbenchmarks.ai)",
    "note": "The compilation is licensed CC BY 4.0. Source documents keep their own terms."
  },
  "rule": "Each board ranks only its own rows. Scores from different benchmarks are combined only in this site's own index (/api/v1/index.json), which is labelled as such.",
  "data": [
    {
      "id": "healthbench-professional",
      "name": "HealthBench Professional",
      "shortName": null,
      "publisher": "OpenAI",
      "released": "2026-04-22",
      "unit": "525 physician-authored tasks",
      "scale": "0 to 1, higher is better",
      "scaleKind": "fraction",
      "higherIsBetter": true,
      "category": "rubric",
      "summary": "Physician-selected workplace tasks, from care consults to documentation and research, graded on physician-written rubrics.",
      "description": "525 tasks that physicians picked out of 15,079 real workplace AI conversations, spanning care consults, clinical documentation, and medical research, each judged on a rubric physicians wrote for it.",
      "officialUrl": null,
      "paperUrl": "https://arxiv.org/abs/2604.27470",
      "boardUrl": "https://healthbenchprofessional.com",
      "status": "active",
      "metrics": [
        {
          "id": "length_adjusted_rubric_score",
          "name": "Length-adjusted rubric score × 100",
          "unit": "points",
          "scaleKind": "percent",
          "description": "April 2026 paper, overall benchmark; GPT-5.4 at highest reasoning; eight samples per task.",
          "higherIsBetter": true
        }
      ],
      "facts": {
        "labs": 3,
        "task": {
          "unit": "Clinician task / conversation",
          "input": "A single- or multi-turn physician-authored conversation ending with a clinician request.",
          "output": "The next assistant response, graded with physician-written rubric items.",
          "setting": "525 selected tasks; default GPT-5.4 low-reasoning grader; primary score adjusted for final-answer length."
        },
        "scale": "0 to 1",
        "tasks": 525,
        "grader": "GPT-5.4, low reasoning effort",
        "useCases": 3,
        "countries": 50,
        "languages": 52,
        "physicians": 190,
        "pricingNote": "GPT-5.6 models charge higher rates above 272K input tokens. MAI-Thinking-1 is in public preview on Microsoft Foundry without final list pricing.",
        "specialties": 26,
        "paperVersion": "April 2026 paper v1; length-adjusted primary score",
        "candidatePool": 15079,
        "physicianBaseline": 0.437
      },
      "stats": {
        "results": 31,
        "models": 26,
        "labs": 5,
        "lastMeasured": "2026-09",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/healthbench-professional.json"
    },
    {
      "id": "medscribe",
      "name": "MedScribe (Vals AI)",
      "shortName": null,
      "publisher": "Vals AI (dataset with Protege)",
      "released": "2026-02",
      "unit": "100 rubric-scored SOAP-note cases",
      "scale": "percentage accuracy 0-100, higher better",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "documentation",
      "summary": "Quality and compliance of SOAP notes generated from clinical visits, scored against documentation rubrics.",
      "description": "Clinical documentation support: quality of SOAP notes generated from clinical visits, scored against rubrics for documentation quality and compliance.",
      "officialUrl": "https://www.vals.ai/benchmarks/medscribe",
      "paperUrl": null,
      "boardUrl": null,
      "status": "active",
      "metrics": [],
      "facts": {
        "boardRows": 105,
        "fieldSize": 105,
        "boardUpdated": "2026-09-29"
      },
      "stats": {
        "results": 105,
        "models": 94,
        "labs": 19,
        "lastMeasured": "2026-09",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/medscribe",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/medscribe.json"
    },
    {
      "id": "medcode",
      "name": "MedCode (Vals AI)",
      "shortName": null,
      "publisher": "Vals AI (dataset with Protege)",
      "released": "2026-02",
      "unit": "2,755 patient records",
      "scale": "percentage accuracy 0-100, higher better",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "documentation",
      "summary": "ICD-10-CM coding of whole hospital stays from discharge summaries and notes, checked against professional coders.",
      "description": "ICD-10-CM diagnosis coding for entire hospital stays: models assign primary and secondary codes from discharge summaries plus progress/consult notes; ground truth double-annotated by certified professional coders.",
      "officialUrl": "https://www.vals.ai/benchmarks/medcode",
      "paperUrl": null,
      "boardUrl": null,
      "status": "active",
      "metrics": [],
      "facts": {
        "boardRows": 103,
        "fieldSize": 103,
        "boardUpdated": "2026-09-29"
      },
      "stats": {
        "results": 103,
        "models": 94,
        "labs": 19,
        "lastMeasured": "2026-09",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/medcode",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/medcode.json"
    },
    {
      "id": "medpic-bench",
      "name": "MedPIC-Bench",
      "shortName": "MedPIC",
      "publisher": "Zhitian Hou, Yuhang Liu, Pengkai Wang, Zeyu Liu, Guanghao Zhu, Zheng Liu, Shuo Cai, Congkai Xie, Zhijie Sang, Kun Zeng, Hongxia Yang (The Hong Kong Polytechnic University; InfiX.ai; Sun Yat-sen University)",
      "released": "2026-08-04",
      "unit": "Overall exact-match accuracy (%)",
      "scale": "0–100",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "safety",
      "summary": "Tests whether models apply and withdraw medication-safety rules correctly as patient information changes.",
      "description": "MedPIC-Bench evaluates whether language models use patient information to apply medication-safety rules correctly. Its 467 expert-validated multiple-choice questions combine fixed guideline-following cases with counterfactual cases in which changes to patient information alter whether a rule applies. Models must select the exact correct set of options; linked-pair accuracy requires both cases to be correct. The benchmark also distinguishes activating a safety warning from withdrawing a warning whose trigger is absent.",
      "officialUrl": "https://huggingface.co/datasets/TIM0927/MedPIC-Bench",
      "paperUrl": "https://arxiv.org/html/2608.03028v1",
      "boardUrl": null,
      "status": "active",
      "metrics": [
        {
          "id": "guideline_following",
          "name": "Guideline-following accuracy (GF)",
          "unit": "%",
          "scaleKind": "percent",
          "description": "Exact option-set match accuracy on 284 fixed patient cases requiring application of a medication-safety recommendation.",
          "higherIsBetter": true
        },
        {
          "id": "counterfactual",
          "name": "Counterfactual accuracy (CF)",
          "unit": "%",
          "scaleKind": "percent",
          "description": "Exact option-set match accuracy on 183 questions with controlled changes in patient information that alter rule applicability.",
          "higherIsBetter": true
        },
        {
          "id": "pair_accuracy",
          "name": "Linked counterfactual pair accuracy",
          "unit": "%",
          "scaleKind": "percent",
          "description": "Share of the 89 linked counterfactual pairs where both questions, presented separately, are answered correctly. The pairing comes from the paper; the public dataset does not include pair identifiers.",
          "higherIsBetter": true
        },
        {
          "id": "activation",
          "name": "Risk activation accuracy",
          "unit": "%",
          "scaleKind": "percent",
          "description": "Exact option-set match accuracy on the 71 counterfactual questions where patient information activates a medication-safety rule.",
          "higherIsBetter": true
        },
        {
          "id": "deactivation",
          "name": "Risk deactivation accuracy",
          "unit": "%",
          "scaleKind": "percent",
          "description": "Exact option-set match accuracy on the 68 counterfactual questions where changed patient information removes a medication-safety rule trigger.",
          "higherIsBetter": true
        },
        {
          "id": "published_gap",
          "name": "Published GF-minus-CF accuracy gap",
          "unit": "percentage points",
          "ranked": false,
          "scaleKind": "percent",
          "description": "Guideline-following accuracy minus counterfactual accuracy, in percentage points, as printed in Table 2. It describes how much accuracy drops when patient information changes. The paper does not treat a smaller gap as better, so this board does not rank models on it."
        }
      ],
      "facts": {
        "authors": [
          "Zhitian Hou",
          "Yuhang Liu",
          "Pengkai Wang",
          "Zeyu Liu",
          "Guanghao Zhu",
          "Zheng Liu",
          "Shuo Cai",
          "Congkai Xie",
          "Zhijie Sang",
          "Kun Zeng",
          "Hongxia Yang"
        ],
        "dataset": {
          "id": "medpic-data",
          "url": "https://huggingface.co/datasets/TIM0927/MedPIC-Bench",
          "sha256": "c28c054ed788de7cbaf7592883e90ef85e530cf9464019dfd39618dcc5ea4ae6",
          "fileUrl": "https://huggingface.co/datasets/TIM0927/MedPIC-Bench/resolve/9ef6db4f13865b14fc2e6be3f94dcfaf3a0cf983/questions.json",
          "license": "CC BY 4.0",
          "revision": "9ef6db4f13865b14fc2e6be3f94dcfaf3a0cf983",
          "checkedDate": "2026-09-28",
          "lastModified": "2026-07-30T12:09:56.000Z"
        },
        "license": "CC BY 4.0",
        "coverage": {
          "benchmark_task_family": {
            "all": {
              "counterfactual": 183,
              "guideline_following": 284
            },
            "counterfactual": {
              "counterfactual": 183
            },
            "guideline_following": {
              "guideline_following": 284
            }
          },
          "taxonomy_clinical_system": {
            "all": {
              "nervous system": 162,
              "urinary system": 138,
              "skeletal system": 10,
              "digestive system": 19,
              "respiratory system": 8,
              "reproductive system": 49,
              "integumentary system": 8,
              "cardiovascular system": 62,
              "lymphatic and immune system": 11
            },
            "counterfactual": {
              "nervous system": 70,
              "urinary system": 44,
              "digestive system": 6,
              "respiratory system": 2,
              "reproductive system": 20,
              "integumentary system": 4,
              "cardiovascular system": 35,
              "lymphatic and immune system": 2
            },
            "guideline_following": {
              "nervous system": 92,
              "urinary system": 94,
              "skeletal system": 10,
              "digestive system": 13,
              "respiratory system": 6,
              "reproductive system": 29,
              "integumentary system": 4,
              "cardiovascular system": 27,
              "lymphatic and immune system": 9
            }
          },
          "taxonomy_drug_categories": {
            "all": {
              "lithium": 5,
              "opioids": 41,
              "statins": 12,
              "z-drugs": 4,
              "retinoids": 3,
              "analgesics": 2,
              "antivirals": 4,
              "anesthetics": 1,
              "antibiotics": 9,
              "antidiabetics": 33,
              "theophyllines": 5,
              "topical drugs": 1,
              "antibacterials": 15,
              "anticoagulants": 10,
              "antiepileptics": 12,
              "antihistamines": 33,
              "antipsychotics": 33,
              "gabapentinoids": 15,
              "hormonal drugs": 2,
              "prostaglandins": 2,
              "ras inhibitors": 15,
              "antiarrhythmics": 33,
              "antigout agents": 8,
              "benzodiazepines": 5,
              "immunomodulators": 4,
              "muscle relaxants": 4,
              "ophthalmic drugs": 4,
              "thyroid hormones": 9,
              "z-class hypnotics": 1,
              "dermatologic drugs": 1,
              "opioid antagonists": 1,
              "cardiovascular drugs": 2,
              "h2-receptor blockers": 7,
              "antipyretic analgesics": 22,
              "gastrointestinal drugs": 13,
              "opioid-like analgesics": 8,
              "urinary antibacterials": 4,
              "h2-receptor antagonists": 19,
              "psychiatric medications": 8,
              "calcium-channel blockers": 18,
              "alpha-1 receptor blockers": 18,
              "cholinesterase inhibitors": 17,
              "tricyclic antidepressants": 6,
              "antidepressants/analgesics": 4,
              "antihistamines/antiemetics": 1,
              "sulfonamide antibacterials": 3,
              "potassium-sparing diuretics": 4,
              "nitroimidazole antibacterials": 4,
              "nsaids/antipyretic analgesics": 50,
              "fluoroquinolone antibacterials": 4,
              "local anesthetics/topical drugs": 6,
              "thiazolidinedione antidiabetics": 4,
              "aldosterone receptor antagonists": 1,
              "antimetabolites/immunosuppressants": 2,
              "central nervous system depressants": 2,
              "factor xa inhibitors/anticoagulants": 22,
              "renin-angiotensin system inhibitors": 2,
              "sulfonamides/urinary antibacterials": 1,
              "nonselective alpha-1 receptor blockers": 2,
              "direct thrombin inhibitors/anticoagulants": 4,
              "pde3 inhibitors/peripheral vascular drugs": 21,
              "multiple drug categories (composite rules)": 12,
              "ras inhibitors/potassium-sparing diuretics": 2,
              "tricyclic antidepressants/anticholinergics": 4,
              "low-molecular-weight heparin anticoagulants": 4,
              "first-generation antihistamines/anticholinergics": 1
            },
            "counterfactual": {
              "lithium": 2,
              "opioids": 22,
              "statins": 2,
              "retinoids": 3,
              "antivirals": 2,
              "antibiotics": 2,
              "theophyllines": 2,
              "antibacterials": 6,
              "anticoagulants": 5,
              "antiepileptics": 6,
              "antipsychotics": 22,
              "gabapentinoids": 6,
              "hormonal drugs": 2,
              "prostaglandins": 2,
              "antiarrhythmics": 15,
              "antigout agents": 4,
              "immunomodulators": 4,
              "muscle relaxants": 2,
              "ophthalmic drugs": 2,
              "h2-receptor blockers": 2,
              "antipyretic analgesics": 2,
              "gastrointestinal drugs": 6,
              "opioid-like analgesics": 4,
              "urinary antibacterials": 2,
              "h2-receptor antagonists": 12,
              "psychiatric medications": 4,
              "calcium-channel blockers": 4,
              "alpha-1 receptor blockers": 8,
              "cholinesterase inhibitors": 13,
              "tricyclic antidepressants": 2,
              "antidepressants/analgesics": 2,
              "nsaids/antipyretic analgesics": 16,
              "fluoroquinolone antibacterials": 2,
              "local anesthetics/topical drugs": 4,
              "thiazolidinedione antidiabetics": 2,
              "antimetabolites/immunosuppressants": 2,
              "central nervous system depressants": 2,
              "factor xa inhibitors/anticoagulants": 8,
              "direct thrombin inhibitors/anticoagulants": 2,
              "pde3 inhibitors/peripheral vascular drugs": 9,
              "multiple drug categories (composite rules)": 3,
              "ras inhibitors/potassium-sparing diuretics": 2,
              "tricyclic antidepressants/anticholinergics": 2,
              "low-molecular-weight heparin anticoagulants": 2
            },
            "guideline_following": {
              "lithium": 3,
              "opioids": 19,
              "statins": 10,
              "z-drugs": 4,
              "analgesics": 2,
              "antivirals": 2,
              "anesthetics": 1,
              "antibiotics": 7,
              "antidiabetics": 33,
              "theophyllines": 3,
              "topical drugs": 1,
              "antibacterials": 9,
              "anticoagulants": 5,
              "antiepileptics": 6,
              "antihistamines": 33,
              "antipsychotics": 11,
              "gabapentinoids": 9,
              "ras inhibitors": 15,
              "antiarrhythmics": 18,
              "antigout agents": 4,
              "benzodiazepines": 5,
              "muscle relaxants": 2,
              "ophthalmic drugs": 2,
              "thyroid hormones": 9,
              "z-class hypnotics": 1,
              "dermatologic drugs": 1,
              "opioid antagonists": 1,
              "cardiovascular drugs": 2,
              "h2-receptor blockers": 5,
              "antipyretic analgesics": 20,
              "gastrointestinal drugs": 7,
              "opioid-like analgesics": 4,
              "urinary antibacterials": 2,
              "h2-receptor antagonists": 7,
              "psychiatric medications": 4,
              "calcium-channel blockers": 14,
              "alpha-1 receptor blockers": 10,
              "cholinesterase inhibitors": 4,
              "tricyclic antidepressants": 4,
              "antidepressants/analgesics": 2,
              "antihistamines/antiemetics": 1,
              "sulfonamide antibacterials": 3,
              "potassium-sparing diuretics": 4,
              "nitroimidazole antibacterials": 4,
              "nsaids/antipyretic analgesics": 34,
              "fluoroquinolone antibacterials": 2,
              "local anesthetics/topical drugs": 2,
              "thiazolidinedione antidiabetics": 2,
              "aldosterone receptor antagonists": 1,
              "factor xa inhibitors/anticoagulants": 14,
              "renin-angiotensin system inhibitors": 2,
              "sulfonamides/urinary antibacterials": 1,
              "nonselective alpha-1 receptor blockers": 2,
              "direct thrombin inhibitors/anticoagulants": 2,
              "pde3 inhibitors/peripheral vascular drugs": 12,
              "multiple drug categories (composite rules)": 9,
              "tricyclic antidepressants/anticholinergics": 2,
              "low-molecular-weight heparin anticoagulants": 2,
              "first-generation antihistamines/anticholinergics": 1
            }
          },
          "taxonomy_patient_info_type": {
            "all": {
              "age threshold": 42,
              "dose threshold": 118,
              "disease presence": 136,
              "drug interaction": 45,
              "pregnancy status": 20,
              "composite clinical context": 66,
              "gestational-week threshold": 29,
              "pediatric-adult contraindication": 11
            },
            "counterfactual": {
              "age threshold": 18,
              "dose threshold": 44,
              "disease presence": 59,
              "drug interaction": 18,
              "pregnancy status": 20,
              "composite clinical context": 20,
              "pediatric-adult contraindication": 4
            },
            "guideline_following": {
              "age threshold": 24,
              "dose threshold": 74,
              "disease presence": 77,
              "drug interaction": 27,
              "composite clinical context": 46,
              "gestational-week threshold": 29,
              "pediatric-adult contraindication": 7
            }
          },
          "taxonomy_population_source": {
            "all": {
              "pregnancy": 49,
              "pediatrics": 66,
              "older adults": 352
            },
            "counterfactual": {
              "pregnancy": 20,
              "pediatrics": 26,
              "older adults": 137
            },
            "guideline_following": {
              "pregnancy": 29,
              "pediatrics": 40,
              "older adults": 215
            }
          },
          "taxonomy_clinical_department": {
            "all": {
              "urology": 20,
              "neurology": 110,
              "cardiology": 60,
              "nephrology": 118,
              "pediatrics": 66,
              "psychiatry": 15,
              "pulmonology": 6,
              "pain medicine": 7,
              "gastroenterology": 6,
              "geriatric medicine": 10,
              "obstetrics and gynecology": 49
            },
            "counterfactual": {
              "neurology": 46,
              "cardiology": 35,
              "nephrology": 44,
              "pediatrics": 26,
              "psychiatry": 6,
              "pulmonology": 2,
              "pain medicine": 4,
              "obstetrics and gynecology": 20
            },
            "guideline_following": {
              "urology": 20,
              "neurology": 64,
              "cardiology": 25,
              "nephrology": 74,
              "pediatrics": 40,
              "psychiatry": 9,
              "pulmonology": 4,
              "pain medicine": 3,
              "gastroenterology": 6,
              "geriatric medicine": 10,
              "obstetrics and gynecology": 29
            }
          },
          "taxonomy_reasoning_operation": {
            "all": {
              "risk activation": 71,
              "risk deactivation": 68,
              "risk redistribution": 26,
              "interaction identification": 18,
              "static risk identification": 284
            },
            "counterfactual": {
              "risk activation": 71,
              "risk deactivation": 68,
              "risk redistribution": 26,
              "interaction identification": 18
            },
            "guideline_following": {
              "static risk identification": 284
            }
          }
        },
        "fieldSize": 28,
        "limitations": [
          "Released questions.json has no explicit pair ID, rule ID, source-citation field, raw model responses or evaluator code. Pair scores are paper-reported only.",
          "No live leaderboard: these results are the August 2026 study snapshot.",
          "Printed gaps can differ by 0.1 from subtracting the rounded guideline-following and counterfactual scores; the printed values are kept.",
          "Drug-category labels are multi-valued and can overlap; their counts do not sum to question count.",
          "Dataset questions are constructed from selected rules and disproportionately cover older adults; this is not an observed clinical cohort or patient-outcome evaluation.",
          "Risk activation/deactivation exclude the other 44 counterfactual items; do not average those two scores to reconstruct total CF.",
          "Model-group averages confound architecture, size and training. No claim that medical adaptation generally helps or harms follows from them."
        ],
        "denominators": {
          "questions": 467,
          "activation": 71,
          "deactivation": 68,
          "counterfactual": 183,
          "redistribution": 26,
          "guidelineFollowing": 284,
          "interactionIdentification": 18,
          "linkedPairsReportedInPaper": 89,
          "matchedActivationDeactivationPairsInFigure4": 67
        },
        "institutions": [
          "The Hong Kong Polytechnic University",
          "InfiX.ai",
          "Sun Yat-sen University"
        ],
        "paperVersion": "arXiv:2608.03028v1",
        "configuration": {
          "id": "paper-zero-shot",
          "setting": "Each question independently, zero-shot, common prompt requests brief rationale and option letters in an answer field. Linked cases presented separately. Exact set match scoring. Text inputs for all models. Open models via PyTorch/Transformers/SGLang on two 80GB A800 GPUs; proprietary models via APIs.",
          "maxTokens": null,
          "temperature": null,
          "uncertainty": "The paper refers to supplementary inference settings and bootstrap intervals that are not in the public paper or dataset release.",
          "measurementDate": null,
          "reasoningEffort": null,
          "exactApiModelIds": null
        },
        "headlineMetric": {
          "id": "overall",
          "name": "Overall accuracy",
          "unit": "%",
          "scaleKind": "percent",
          "description": "Exact option-set match accuracy across all 467 questions: 284 guideline-following and 183 counterfactual.",
          "higherIsBetter": true
        },
        "datasetRevision": "9ef6db4f13865b14fc2e6be3f94dcfaf3a0cf983",
        "departmentGroupResults": {
          "rows": [
            {
              "general": 31.6,
              "department": "Neurology",
              "proprietary": 40.7,
              "medicalSpecific": 26.4
            },
            {
              "general": 33,
              "department": "Cardiology",
              "proprietary": 42.9,
              "medicalSpecific": 27.9
            },
            {
              "general": 47.4,
              "department": "Pediatrics",
              "proprietary": 54.9,
              "medicalSpecific": 42.9
            },
            {
              "general": 53,
              "department": "Nephrology",
              "proprietary": 67.2,
              "medicalSpecific": 41.1
            },
            {
              "general": 81.1,
              "department": "Obstetrics & Gynecology",
              "proprietary": 96.4,
              "medicalSpecific": 80
            }
          ],
          "metric": "Mean counterfactual accuracy (%) by model group; departments with at least 20 CF questions",
          "sourceId": "medpic-paper",
          "sourceLocator": "Table 3"
        }
      },
      "stats": {
        "results": 28,
        "models": 28,
        "labs": 11,
        "lastMeasured": "2026-08",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/medpic-bench",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/medpic-bench.json"
    },
    {
      "id": "medxpertqa-mm",
      "name": "MedXpertQA (MM)",
      "shortName": null,
      "publisher": "TsinghuaC3I (Tsinghua University)",
      "released": "2025-01",
      "unit": "2,000 multimodal questions (MM subset)",
      "scale": "percentage accuracy 0-100, higher better",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "knowledge",
      "summary": "Expert-level multiple-choice questions over clinical images across 17 specialties.",
      "description": "Expert-level multimodal medical multiple-choice QA covering clinical images (X-ray, histology, dermatology, charts) across 17 specialties; MM subset of the 4,460-question MedXpertQA benchmark.",
      "officialUrl": "https://benchlm.ai/benchmarks/medxpertqamm",
      "paperUrl": "https://arxiv.org/html/2501.18362v3",
      "boardUrl": null,
      "status": "active",
      "metrics": [
        {
          "id": "accuracy",
          "name": "Accuracy",
          "unit": "%",
          "scaleKind": "percent",
          "description": "Same 2025 paper; multimodal test under its original prompt and answer-extraction protocol.",
          "higherIsBetter": true
        }
      ],
      "facts": {
        "task": {
          "unit": "One examination question",
          "input": "Medical examination question; the MM track also supplies associated images and clinical context",
          "output": "One answer choice: ten options for Text, five for MM",
          "setting": "Original zero-shot chain-of-thought evaluation; greedy decoding where supported"
        },
        "fieldSize": 5,
        "boardUpdated": "2026-04-08",
        "paperVersion": "ICML 2025 / arXiv v3"
      },
      "stats": {
        "results": 22,
        "models": 22,
        "labs": 7,
        "lastMeasured": "2026-08",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/medxpertqa-mm",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/medxpertqa-mm.json"
    },
    {
      "id": "mast",
      "name": "MAST (Medical AI Superintelligence Test)",
      "shortName": null,
      "publisher": "ARISE AI Research Network (multi-institutional)",
      "released": "2026-08",
      "unit": "composite of 6 component benchmarks; 11 models",
      "scale": "percentage composite, higher better",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "composite",
      "summary": "A composite of clinical benchmarks spanning diagnostic and management reasoning, safety, multimodal imaging, and agentic capability.",
      "description": "Composite score across curated clinical benchmarks spanning diagnostic reasoning, management reasoning, safety, multimodal images, multimodal radiology, and agentic capability. Components: First Do NOHARM v2, SCT-Bench, MedAgentBench v2, PhysicianBench, ReXrank Mini, CPC-Bench.",
      "officialUrl": "https://arise-ai.org/mast",
      "paperUrl": null,
      "boardUrl": null,
      "status": "active",
      "metrics": [],
      "facts": {
        "preview": true,
        "fieldSize": 11,
        "aggregation": "harmonic mean of Diagnostic Reasoning, Management Reasoning, Safety, Multimodal Radiology, Multimodal Images",
        "boardUpdated": "2026-08-15",
        "displayFilter": "latest large and small model from each provider",
        "previewNotice": "Preview: MAST is currently in preview. Exact scores on this benchmark may change as we undergo final validation and tuning.",
        "boardGeneratedAt": "2026-08-15 18:28 UTC",
        "modelsInDefaultView": 11
      },
      "stats": {
        "results": 8,
        "models": 8,
        "labs": 6,
        "lastMeasured": "2026-08",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/mast",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/mast.json"
    },
    {
      "id": "medhelm",
      "name": "MedHELM",
      "shortName": null,
      "publisher": "Stanford CRFM / HAI and multi-institution collaborators",
      "released": "2025-02",
      "unit": "121 tasks / 31 datasets",
      "scale": "mean win rate 0-1, higher better",
      "scaleKind": "fraction",
      "higherIsBetter": true,
      "category": "composite",
      "summary": "Holistic clinical evaluation across 121 tasks in a clinician-validated taxonomy, ranked by mean win rate.",
      "description": "Holistic evaluation of LLMs on 121 clinical tasks across 5 categories and 22 subcategories (31 datasets) in a clinician-validated taxonomy; ranked by mean win rate.",
      "officialUrl": "https://medhelm.org/",
      "paperUrl": "https://www.nature.com/articles/s41591-025-04151-2",
      "boardUrl": null,
      "status": "active",
      "metrics": [
        {
          "id": "win_rate",
          "name": "Pairwise win rate",
          "unit": "proportion",
          "scaleKind": "fraction",
          "description": "Nature Medicine 2026 final paper, Table 1; 37 benchmarks and nine historical models.",
          "higherIsBetter": true
        }
      ],
      "facts": {
        "task": {
          "unit": "A benchmark-specific evaluation instance; 37 benchmark scores feed the aggregate.",
          "input": "Task-specific records, questions, conversations or research requests.",
          "output": "Task-specific classification, calculation, generated text, code or structured answer.",
          "setting": "Nine historical models under the published evaluation pipeline."
        },
        "metric": "Mean win rate",
        "version": "v5.0.0",
        "fieldSize": 11,
        "releaseDate": "2026-05-08",
        "boardUpdated": "2026-05-14",
        "newerVersion": false,
        "paperVersion": "Final journal paper: 37-benchmark snapshot",
        "modelsOnBoard": 11
      },
      "stats": {
        "results": 10,
        "models": 10,
        "labs": 5,
        "lastMeasured": "2026-05",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/medhelm",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/medhelm.json"
    },
    {
      "id": "first-do-noharm",
      "name": "First, Do NOHARM (v2)",
      "shortName": null,
      "publisher": "Stanford/Harvard-led consortium (50+ researchers incl. 29 board-certified physicians); hosted by ARISE",
      "released": "2025-12",
      "unit": "1,100 consultation cases, 10 specialties, 12,747 expert annotations on 4,249 management options",
      "scale": "percentage safety score, higher better",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "safety",
      "summary": "How often, and how severely, model consultation recommendations contain potentially harmful errors.",
      "description": "Frequency and severity of potentially harmful errors in LLM-generated medical consultation recommendations (Numerous Options Harm Assessment for Risk in Medicine); primary-care-to-specialist consults.",
      "officialUrl": "https://arise-ai.org/mast/technical",
      "paperUrl": "https://arxiv.org/abs/2512.01241",
      "boardUrl": null,
      "status": "active",
      "metrics": [],
      "facts": {
        "metric": "F1 (Weighted)",
        "preview": true,
        "fieldSize": 19,
        "boardUpdated": "2026-08-15",
        "modelsInDataset": 43,
        "modelsInDefaultView": 19
      },
      "stats": {
        "results": 17,
        "models": 17,
        "labs": 12,
        "lastMeasured": "2026-09",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/first-do-noharm",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/first-do-noharm.json"
    },
    {
      "id": "physicianbench",
      "name": "PhysicianBench",
      "shortName": null,
      "publisher": "Academic team (Ruoqi Liu, Imran Q. Mohiuddin et al., arXiv 2605.02240); also a MAST component",
      "released": "2026-05",
      "unit": "100 real-world clinical tasks, 21 specialties, 670 structured checkpoints (~27 tool calls per task)",
      "scale": "pass@1 success rate %, higher better (3 independent runs; Pass^3 also reported)",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "agentic",
      "summary": "Agents carrying out long-horizon physician workflows inside real EHR systems, verified by execution against those systems.",
      "description": "LLM agents on long-horizon composite physician workflows inside real EHR environments, with execution-grounded verification against actual EHR systems via standard commercial APIs.",
      "officialUrl": "https://arxiv.org/abs/2605.02240",
      "paperUrl": "https://arxiv.org/abs/2605.02240",
      "boardUrl": null,
      "status": "active",
      "metrics": [],
      "facts": {
        "table": "Table 2, p. 8",
        "metric": "Pass@1 (%), mean ± sd over 3 runs",
        "fieldSize": 12,
        "paperDate": "2026-05-04",
        "paperVersion": "v1",
        "modelsInTable": 12
      },
      "stats": {
        "results": 21,
        "models": 17,
        "labs": 9,
        "lastMeasured": "2026-09",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/physicianbench",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/physicianbench.json"
    },
    {
      "id": "ehr-complex",
      "name": "EHR-Complex",
      "shortName": null,
      "publisher": "Academic team (Qiao et al., Ant Group-affiliated; arXiv 2606.23301)",
      "released": "2026-06",
      "unit": "~52,000 tasks (3,915-task test set) over 365K patients, 31 tables, 500M+ records",
      "scale": "exact-match accuracy, 0-1, higher better",
      "scaleKind": "fraction",
      "higherIsBetter": true,
      "category": "agentic",
      "summary": "Agentic clinical reasoning over MIMIC-IV records through SQL and Python, at patient and population level.",
      "description": "Agentic clinical reasoning over MIMIC-IV EHR databases via SQL and Python across six clinical intents, at patient and population level with temporal evidence paths.",
      "officialUrl": "https://arxiv.org/abs/2606.23301",
      "paperUrl": "https://arxiv.org/abs/2606.23301",
      "boardUrl": null,
      "status": "active",
      "metrics": [],
      "facts": {
        "metric": "exact-match success, Avg. over 12 intent x scope columns",
        "tables": "Table 3 (p. 6) headline; Table 10 (p. 15) strong commercial models",
        "fieldSize": 18,
        "paperDate": "2026-06-22",
        "paperVersion": "v1"
      },
      "stats": {
        "results": 18,
        "models": 17,
        "labs": 7,
        "lastMeasured": "2026-06",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/ehr-complex",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/ehr-complex.json"
    },
    {
      "id": "healthagentbench",
      "name": "HealthAgentBench",
      "shortName": null,
      "publisher": "Microsoft Research",
      "released": "2026-07",
      "unit": "54 tasks across 7 environments; 162 trials (3 attempts per task)",
      "scale": "mean task success rate, 0-100%, higher better; cost per task also reported",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "agentic",
      "summary": "Agent harnesses completing realistic terminal-based healthcare tasks built from real clinical artifacts.",
      "description": "Agentic task success in realistic terminal-based healthcare environments built from real clinical artifacts; evaluates agent harnesses (Claude Code, Codex, Copilot) end to end, not bare models.",
      "officialUrl": "https://microsoft.github.io/HealthAgentBench/",
      "paperUrl": "https://arxiv.org/abs/2606.31179",
      "boardUrl": null,
      "status": "active",
      "metrics": [],
      "facts": {
        "metric": "pooled task success rate across 3 attempts x 54 tasks (162 trials)",
        "boardRows": 12,
        "fieldSize": 12,
        "boardUpdated": "2026-07-27"
      },
      "stats": {
        "results": 12,
        "models": 10,
        "labs": 2,
        "lastMeasured": "2026-07",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/healthagentbench",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/healthagentbench.json"
    },
    {
      "id": "chi-bench",
      "name": "CHI-Bench",
      "shortName": null,
      "publisher": "actAVA.ai",
      "released": "2026-05",
      "unit": "75 workflows (25 per domain), 21 healthcare applications, 200+ MCP tools",
      "scale": "pass@1 with binary 0/1 reward, higher better",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "agentic",
      "summary": "Long-horizon healthcare operations workflows for agents: prior authorization, utilization management, and care management.",
      "description": "Long-horizon US healthcare operations workflows for agents (prior authorization, utilization management, care management), 60-80 step tasks across 4-6 stages, judged by deterministic unit tests plus an LLM judge for evidence grounding, consent, and cross-stage consistency.",
      "officialUrl": "https://actava.ai/benchmarks/leaderboards",
      "paperUrl": null,
      "boardUrl": null,
      "status": "active",
      "metrics": [],
      "facts": {
        "metric": "pass@1 accuracy, 75 tasks (25 each: prior authorization, utilization management, care management)",
        "boardRows": 45,
        "fieldSize": 45,
        "boardUpdated": "2026-08-12",
        "boardVersion": "CHI-Bench v1.0.0"
      },
      "stats": {
        "results": 43,
        "models": 25,
        "labs": 11,
        "lastMeasured": "2026-08",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/chi-bench",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/chi-bench.json"
    },
    {
      "id": "healthadminbench",
      "name": "HealthAdminBench",
      "shortName": null,
      "publisher": "Kinetic Systems (with Stanford Hospital domain experts)",
      "released": "2026-04",
      "unit": "135 tasks / 1,698 rubric-scored subtasks",
      "scale": "percentage end-to-end task success 0-100, higher better",
      "scaleKind": "percent",
      "higherIsBetter": true,
      "category": "agentic",
      "summary": "Computer-use agents completing healthcare administration workflows: prior authorizations, denial appeals, and DME ordering.",
      "description": "End-to-end task success of computer-use LLM agents on healthcare administration workflows: prior authorizations, denial appeals, and DME ordering; success requires completing every subtask in a task.",
      "officialUrl": "https://healthadminbench.stanford.edu/",
      "paperUrl": "https://arxiv.org/abs/2604.09937",
      "boardUrl": null,
      "status": "active",
      "metrics": [],
      "facts": {
        "metric": "end-to-end task success rate, n=135 tasks; screenshot-only observations, Task Description + Portal Guidance prompting",
        "fieldSize": 5,
        "boardUpdated": "2026-04-10"
      },
      "stats": {
        "results": 7,
        "models": 5,
        "labs": 5,
        "lastMeasured": "2026-04",
        "confidence": "verified"
      },
      "url": "https://clinicalbenchmarks.ai/benchmarks/healthadminbench",
      "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/healthadminbench.json"
    }
  ]
}
