{
  "dataset": "Clinical Benchmarks",
  "kind": "benchmark",
  "url": "https://clinicalbenchmarks.ai",
  "updated": "2026-09-30",
  "revision": 7165,
  "licence": {
    "name": "CC BY 4.0",
    "url": "https://creativecommons.org/licenses/by/4.0/",
    "attribution": "Clinical Benchmarks (clinicalbenchmarks.ai)",
    "note": "The compilation is licensed CC BY 4.0. Source documents keep their own terms."
  },
  "rule": "Each board ranks only its own rows. Scores from different benchmarks are combined only in this site's own index (/api/v1/index.json), which is labelled as such.",
  "data": {
    "id": "healthbench-professional",
    "name": "HealthBench Professional",
    "shortName": null,
    "publisher": "OpenAI",
    "released": "2026-04-22",
    "unit": "525 physician-authored tasks",
    "scale": "0 to 1, higher is better",
    "scaleKind": "fraction",
    "higherIsBetter": true,
    "category": "rubric",
    "summary": "Physician-selected workplace tasks, from care consults to documentation and research, graded on physician-written rubrics.",
    "description": "525 tasks that physicians picked out of 15,079 real workplace AI conversations, spanning care consults, clinical documentation, and medical research, each judged on a rubric physicians wrote for it.",
    "officialUrl": null,
    "paperUrl": "https://arxiv.org/abs/2604.27470",
    "boardUrl": "https://healthbenchprofessional.com",
    "status": "active",
    "metrics": [
      {
        "id": "length_adjusted_rubric_score",
        "name": "Length-adjusted rubric score × 100",
        "unit": "points",
        "scaleKind": "percent",
        "description": "April 2026 paper, overall benchmark; GPT-5.4 at highest reasoning; eight samples per task.",
        "higherIsBetter": true
      }
    ],
    "facts": {
      "labs": 3,
      "task": {
        "unit": "Clinician task / conversation",
        "input": "A single- or multi-turn physician-authored conversation ending with a clinician request.",
        "output": "The next assistant response, graded with physician-written rubric items.",
        "setting": "525 selected tasks; default GPT-5.4 low-reasoning grader; primary score adjusted for final-answer length."
      },
      "scale": "0 to 1",
      "tasks": 525,
      "grader": "GPT-5.4, low reasoning effort",
      "useCases": 3,
      "countries": 50,
      "languages": 52,
      "physicians": 190,
      "pricingNote": "GPT-5.6 models charge higher rates above 272K input tokens. MAI-Thinking-1 is in public preview on Microsoft Foundry without final list pricing.",
      "specialties": 26,
      "paperVersion": "April 2026 paper v1; length-adjusted primary score",
      "candidatePool": 15079,
      "physicianBaseline": 0.437
    },
    "stats": {
      "results": 31,
      "models": 26,
      "labs": 5,
      "lastMeasured": "2026-09",
      "confidence": "verified"
    },
    "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional",
    "api": "https://clinicalbenchmarks.ai/api/v1/benchmarks/healthbench-professional.json",
    "results": [
      {
        "id": 1799,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-6-astra",
        "modelKind": "model",
        "name": "GPT-6 Astra (Anthropic run)",
        "label": "GPT-6 Astra (Anthropic run)",
        "lab": "openai",
        "variant": "Anthropic public-API reproduction; max effort; no system prompt; Opus 4.8 grader; length-adjusted; raw 74.0%",
        "value": 0.703,
        "valueLabel": "0.703",
        "display": "70.3%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 1,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "third_party",
        "confidence": "verified",
        "source": 1287,
        "sourceUrl": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
        "quote": "On HealthBench Professional at max effort, length adjustment changes the ranking.\nAfter: GPT-6 Astra (70.3%) > Claude Sonnet 5.5 (69.2%) > Claude Opus 5.5 (65.6%) > Claude Fable 5.1 (62.1%).",
        "locator": "p. 138, paragraph above Figure 8.15.B, HealthBench Professional after length adjustment.",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-1799"
      },
      {
        "id": 1800,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-sonnet-5.5",
        "modelKind": "model",
        "name": "Claude Sonnet 5.5",
        "label": null,
        "lab": "anthropic",
        "variant": "Anthropic; max effort; Opus 4.8 grader; safety classifiers enabled; length-adjusted; raw 77.1%",
        "value": 0.6920000000000001,
        "valueLabel": "0.692",
        "display": "69.2%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 2,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 1287,
        "sourceUrl": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
        "quote": "On HealthBench Professional at max effort, length adjustment changes the ranking.\nAfter: GPT-6 Astra (70.3%) > Claude Sonnet 5.5 (69.2%) > Claude Opus 5.5 (65.6%) > Claude Fable 5.1 (62.1%).",
        "locator": "p. 138, Section 8.15.2 and paragraph above Figure 8.15.B.",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-1800"
      },
      {
        "id": 204,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-fable-5",
        "modelKind": "model",
        "name": "Claude Fable 5",
        "label": null,
        "lab": "anthropic",
        "variant": "length-adjusted, Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt (raw 70.3%). Measured as Claude Mythos 5; Anthropic itself prints 66.0 in the Fable 5 column of the Opus 5 card with footnote Mythos 5.",
        "value": 0.66,
        "valueLabel": "0.660",
        "display": "66.0",
        "lo": null,
        "hi": null,
        "rankOnBoard": 3,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 61,
        "sourceUrl": "https://anthropic.com/claude-fable-5-mythos-5-system-card",
        "quote": "HealthBench\nProfessional\n66.0 - 64.7 56.9 51.8 -",
        "locator": "p. 252, Table 8.1.A, row HealthBench Professional, column Mythos 5 (Fable 5 column '-'); Figure 8.18.2.A p. 298 bar label 66.0% on Claude Mythos 5",
        "corroborating": [
          {
            "source": 98,
            "role": "corroborating",
            "scoreDisplay": "66.0",
            "quote": "HealthBench Professional 59.8 57.4 66.011 60.5",
            "locator": "p. 152, Table 8.1.A, column Fable 5 (footnote 11: Mythos 5)"
          },
          {
            "source": 123,
            "role": "conflicting",
            "scoreDisplay": "63.3%",
            "quote": "HealthBench Professional 62.1% 63.3% 59.8% –",
            "locator": "p. 167, Table 8.1.A, column 'Claude Fable 5/Mythos 5'"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-204"
      },
      {
        "id": 1801,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-opus-5.5",
        "modelKind": "model",
        "name": "Claude Opus 5.5",
        "label": null,
        "lab": "anthropic",
        "variant": "Anthropic; adaptive max effort; Opus 4.8 grader; five trials; no tools or custom system prompt; safety classifiers and refusal fallback to Opus 5; length-adjusted; raw 77.1%",
        "value": 0.6559999999999999,
        "valueLabel": "0.656",
        "display": "65.6%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 4,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 1288,
        "sourceUrl": "https://www.anthropic.com/claude-opus-5-5-system-card",
        "quote": "8.15.2 HealthBench Professional results\nlength adjustment, which penalizes verbose model responses, Claude Opus 5.5 achieved a score of 65.6%.",
        "locator": "pp. 213–214, Section 8.15.2, paragraph and Figure 8.15.2.A.",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-1801"
      },
      {
        "id": 563,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-6-astra",
        "modelKind": "model",
        "name": "GPT-6 Astra",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted, max reasoning effort (69.5 unadjusted, 4,097 mean response chars); GPT-6 Astra system card Table 6, column 'gpt-6 Astra'.",
        "value": 0.647,
        "valueLabel": "0.647",
        "display": "64.7",
        "lo": null,
        "hi": null,
        "rankOnBoard": 5,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 1106,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
        "quote": "Evaluation | GPT-5.5 | GPT-5.6 Sol | GPT-5.6 Terra | GPT-5.6 Luna | GPT-6 Astra | GPT-6 Sol | GPT-6 Luna\nHealthBench Professional length-adjusted | 51.8 (57.2, 3818) | 60.5 (64.1, 3228) | 57.7 (62.4, 3618) | 55.7 (59.8, 3389) | 64.7 (68.2, 3185) | 60.8 (59.5, 1573) | 60.8 (61.2, 2119)",
        "locator": "Section 11.4.1, Table 29, HealthBench Professional length-adjusted, GPT-6 Astra column; September 22 correction.",
        "corroborating": [
          {
            "source": 1107,
            "role": "corroborating",
            "scoreDisplay": "63.4%",
            "quote": "| HealthBench Professional (length-adjusted) | 63.4% | 60.5% | 58.1% | 60.9% | 56.4% | 52.1% |",
            "locator": "Science and Health table"
          },
          {
            "source": 1106,
            "role": "mirror",
            "scoreDisplay": "63.4",
            "quote": "Astra has a length-adjusted HealthBench Professional score of 63.4 (+2.9 relative to GPT-5.6 Sol)",
            "locator": "section 6.1, HTML rendering of the same card"
          },
          {
            "source": 1104,
            "role": "conflicting",
            "scoreDisplay": "63.4",
            "quote": "Astra has a length-adjusted HealthBench Professional score of 63.4 (+2.9 relative to GPT-5.6 Sol), HealthBench score of 58.1 (+1.1), HealthBench Hard score of 36.3 (+3.2), and HealthBench Consensus score of 95.8 (+0.3).",
            "locator": "p. 19, sec. 6.1 (Table 6 cell, column 'gpt-6 Astra': 63.4 (69.5, 4097))"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-563"
      },
      {
        "id": 1870,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-fable-5",
        "modelKind": "model",
        "name": "Claude Fable 5 (September card)",
        "label": "Claude Fable 5 (September card)",
        "lab": "anthropic",
        "variant": "Measured as Claude Fable 5; Anthropic September card; length-adjusted; adaptive max effort; Opus 4.8 grader; five trials; no tools or custom system prompt; raw 68.9%",
        "value": 0.633,
        "valueLabel": "0.633",
        "display": "63.3%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 6,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 123,
        "sourceUrl": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
        "quote": "HealthBench Professional | Length-adjusted score | Claude Fable 5 | 63.3%",
        "locator": "p. 199, Figure 8.17.2.A, Claude Fable 5 length-adjusted bar (visually read printed labels).",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-1870"
      },
      {
        "id": 561,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-fable-5.1",
        "modelKind": "model",
        "name": "Claude Fable 5.1",
        "label": null,
        "lab": "anthropic",
        "variant": "length-adjusted (method published in the HealthBench Professional paper); Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over five trials, no tools or customized system prompt; Fable 5.1 run with safety classifiers active and a refusal-fallback to Claude Opus 5 (raw 74.2%).",
        "value": 0.621,
        "valueLabel": "0.621",
        "display": "62.1%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 7,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 123,
        "sourceUrl": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
        "quote": "On HealthBench Professional, Claude Fable 5.1 achieved a raw score of 74.2%, ahead of Claude Opus 5 at 73.4%, Fable 5 at 68.9%, and Claude Sonnet 5 at 62.4%. After length adjustment, which penalizes verbose model responses, Fable 5.1 achieved a score of 62.1%.",
        "locator": "p. 199, sec. 8.17.2 (same figure printed as 62.1% in Table 8.1.A, p. 167)",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-561"
      },
      {
        "id": 1803,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-6-luna",
        "modelKind": "model",
        "name": "GPT-6 Luna",
        "label": null,
        "lab": "openai",
        "variant": "OpenAI; length-adjusted; maximum reasoning effort; 60.8 (61.2, 2119) (adjusted, raw, mean response characters)",
        "value": 0.608,
        "valueLabel": "0.608",
        "display": "60.8",
        "lo": null,
        "hi": null,
        "rankOnBoard": 8,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 1106,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
        "quote": "Evaluation | GPT-5.5 | GPT-5.6 Sol | GPT-5.6 Terra | GPT-5.6 Luna | GPT-6 Astra | GPT-6 Sol | GPT-6 Luna\nHealthBench Professional length-adjusted | 51.8 (57.2, 3818) | 60.5 (64.1, 3228) | 57.7 (62.4, 3618) | 55.7 (59.8, 3389) | 64.7 (68.2, 3185) | 60.8 (59.5, 1573) | 60.8 (61.2, 2119)",
        "locator": "Section 11.4.1, Table 29, HealthBench Professional length-adjusted, GPT-6 Luna column; September 22 correction.",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-1803"
      },
      {
        "id": 1802,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-6-sol",
        "modelKind": "model",
        "name": "GPT-6 Sol",
        "label": null,
        "lab": "openai",
        "variant": "OpenAI; length-adjusted; maximum reasoning effort; 60.8 (59.5, 1573) (adjusted, raw, mean response characters)",
        "value": 0.608,
        "valueLabel": "0.608",
        "display": "60.8",
        "lo": null,
        "hi": null,
        "rankOnBoard": 8,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 1106,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
        "quote": "Evaluation | GPT-5.5 | GPT-5.6 Sol | GPT-5.6 Terra | GPT-5.6 Luna | GPT-6 Astra | GPT-6 Sol | GPT-6 Luna\nHealthBench Professional length-adjusted | 51.8 (57.2, 3818) | 60.5 (64.1, 3228) | 57.7 (62.4, 3618) | 55.7 (59.8, 3389) | 64.7 (68.2, 3185) | 60.8 (59.5, 1573) | 60.8 (61.2, 2119)",
        "locator": "Section 11.4.1, Table 29, HealthBench Professional length-adjusted, GPT-6 Sol column; September 22 correction.",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-1802"
      },
      {
        "id": 205,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.6-sol",
        "modelKind": "model",
        "name": "GPT-5.6 Sol",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted, max reasoning effort (64.1 unadjusted, 3,228 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-SOL. API max-effort setting, not the August ChatGPT production setting measured in row 492.",
        "value": 0.605,
        "valueLabel": "0.605",
        "display": "60.5",
        "lo": null,
        "hi": null,
        "rankOnBoard": 10,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 50,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
        "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
        "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-SOL",
        "corroborating": [
          {
            "source": 16,
            "role": "corroborating",
            "scoreDisplay": "60.5",
            "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-SOL"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-205"
      },
      {
        "id": 206,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-opus-5",
        "modelKind": "model",
        "name": "Claude Opus 5",
        "label": null,
        "lab": "anthropic",
        "variant": "length-adjusted, Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt (raw 73.4%).",
        "value": 0.598,
        "valueLabel": "0.598",
        "display": "59.8%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 11,
        "rowsOnBoard": 31,
        "measured": "2026-07",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 98,
        "sourceUrl": "https://www.anthropic.com/claude-opus-5-system-card",
        "quote": "Claude Opus 5 achieved a raw score of 73.4%, which is the highest amongst all Claude models, ahead of Claude Mythos 5 at 70.3%, Claude Opus 4.8 at 60.3%, and Claude Sonnet 5 at 62.4%. After length adjustment, which penalizes verbose model responses, Claude Opus 5 achieved a score of 59.8%.",
        "locator": "p. 189, section 8.15.2 HealthBench Professional results; also Table 8.1.A p. 152 ('HealthBench Professional 59.8 ...')",
        "corroborating": [
          {
            "source": 123,
            "role": "corroborating",
            "scoreDisplay": "59.8%",
            "quote": "HealthBench Professional 62.1% 63.3% 59.8% –",
            "locator": "p. 167, Table 8.1.A, column 'Claude Opus 5'; raw 73.4% also reprinted on p. 199, sec. 8.17.2"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-206"
      },
      {
        "id": 505,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "muse-spark-1.1",
        "modelKind": "model",
        "name": "Muse Spark 1.1",
        "label": null,
        "lab": "meta",
        "variant": "length-normalized, GPT-5.4 low-reasoning grader, xhigh reasoning via Meta Model API (Muse Spark 1.1 Evaluation Report Figure 44)",
        "value": 0.593,
        "valueLabel": "0.593",
        "display": "59.3",
        "lo": null,
        "hi": null,
        "rankOnBoard": 12,
        "rowsOnBoard": 31,
        "measured": "2026-07",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 105,
        "sourceUrl": "https://research.meta.ai/static/muse-spark-1-1-evaluation-report",
        "quote": "Health | HealthBench Professional | 59.3 | 54.1 | 41.6 | 55.8 | 51.8 (Figure 44 image table row; columns Muse Spark 1.1, Muse Spark, Gemini 3.1 Pro (high), Opus 4.8 (max), GPT 5.5 (xhigh))",
        "locator": "p. 101, Figure 44 'General capability benchmark results' (image), row HealthBench Professional, column Muse Spark 1.1; protocol p. 104 (printed 103): HealthBench Pro comprises 525 evaluation data points graded by rubrics. We use GPT-5.4 with low reasoning effort as the grader and report the length-normalized rubric score as done in their paper.",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-505"
      },
      {
        "id": 207,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-sonnet-5",
        "modelKind": "model",
        "name": "Claude Sonnet 5",
        "label": null,
        "lab": "anthropic",
        "variant": "length-adjusted, Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt (raw 62.4%).",
        "value": 0.578,
        "valueLabel": "0.578",
        "display": "57.8",
        "lo": null,
        "hi": null,
        "rankOnBoard": 13,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 99,
        "sourceUrl": "https://www.anthropic.com/claude-sonnet-5-system-card",
        "quote": "HealthBench\nProfessional\n57.8 44.2 51.8 -",
        "locator": "p. 115, Table 8.1.A, row HealthBench Professional, column Claude Sonnet 5; Figure 8.12.2.A p. 139",
        "corroborating": [
          {
            "source": 98,
            "role": "corroborating",
            "scoreDisplay": "57.8%",
            "quote": "Figure 8.15.2.A bar labels: Claude Sonnet 5 62.4% raw / 57.8% length-adjusted (prose: 'Claude Sonnet 5 at 62.4%')",
            "locator": "p. 189, section 8.15.2, Figure 8.15.2.A"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-207"
      },
      {
        "id": 208,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.6-terra",
        "modelKind": "model",
        "name": "GPT-5.6 Terra",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted, max reasoning effort (62.4 unadjusted, 3,618 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-TERRA.",
        "value": 0.577,
        "valueLabel": "0.577",
        "display": "57.7",
        "lo": null,
        "hi": null,
        "rankOnBoard": 14,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 50,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
        "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
        "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-TERRA",
        "corroborating": [
          {
            "source": 16,
            "role": "corroborating",
            "scoreDisplay": "57.7",
            "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-TERRA"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-208"
      },
      {
        "id": 1871,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-opus-4.8",
        "modelKind": "model",
        "name": "Claude Opus 4.8 (Opus 4.8 grader)",
        "label": "Claude Opus 4.8 (Opus 4.8 grader)",
        "lab": "anthropic",
        "variant": "Anthropic June evaluation; length-adjusted; adaptive max effort; Opus 4.8 grader; five trials; no tools or custom system prompt",
        "value": 0.574,
        "valueLabel": "0.574",
        "display": "57.4%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 15,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 99,
        "sourceUrl": "https://www.anthropic.com/claude-sonnet-5-system-card",
        "quote": "HealthBench Professional | Length-adjusted score (%) | Claude Opus 4.8 | 57.4%",
        "locator": "p. 139, Figure 8.12.2.A, Claude Opus 4.8 bar (visually read printed labels).",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-1871"
      },
      {
        "id": 1804,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "grok-4.7",
        "modelKind": "model",
        "name": "Grok 4.7",
        "label": null,
        "lab": "xai",
        "variant": "SpaceXAI evaluation; xHigh effort; HealthBench Professional",
        "value": 0.5670000000000001,
        "valueLabel": "0.567",
        "display": "56.7%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 16,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 1290,
        "sourceUrl": "https://x.ai/news/grok-4-7",
        "quote": "Grok 4.7 xHigh\nGrok 4.6 High\nGPT-5.6 Sol Max\nFable 5.1 Max\nClinical reasoningHealthBench Professional\n56.7%\n48.5%\n60.5%\n62.1%",
        "locator": "Model Improvements comparison table, Clinical reasoning / HealthBench Professional row; Grok 4.7 column; September 21, 2026.",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-1804"
      },
      {
        "id": 209,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-opus-4.8",
        "modelKind": "model",
        "name": "Claude Opus 4.8",
        "label": null,
        "lab": "anthropic",
        "variant": "length-adjusted, adaptive thinking at max effort, Claude Sonnet 4.6 grader (the grader used by the Opus 4.8 card itself); the other Claude rows on this board use the Claude Opus 4.8 grader, under which Anthropic later prints 57.4 for this model.",
        "value": 0.558,
        "valueLabel": "0.558",
        "display": "55.8%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 17,
        "rowsOnBoard": 31,
        "measured": "2026-05",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 101,
        "sourceUrl": "https://www.anthropic.com/claude-opus-4-8-system-card",
        "quote": "Claude Opus 4.8 scores 55.8%, a meaningful improvement over Claude Opus 4.7 at 51.9% and Claude Sonnet 4.6 at 41.7%.",
        "locator": "p. 228, section 8.14.1 HealthBench Professional; Figure 8.14.A p. 229",
        "corroborating": [
          {
            "source": 61,
            "role": "conflicting",
            "scoreDisplay": "56.9",
            "quote": "HealthBench\nProfessional\n66.0 - 64.7 56.9 51.8 -",
            "locator": "p. 252, Table 8.1.A, column Opus 4.8"
          },
          {
            "source": 98,
            "role": "conflicting",
            "scoreDisplay": "57.4",
            "quote": "HealthBench Professional 59.8 57.4 66.011 60.5",
            "locator": "p. 152, Table 8.1.A, column Opus 4.8"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-209"
      },
      {
        "id": 210,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.6-luna",
        "modelKind": "model",
        "name": "GPT-5.6 Luna",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted, max reasoning effort (59.8 unadjusted, 3,389 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-LUNA. API max-effort setting, not the August ChatGPT production setting measured in row 493.",
        "value": 0.557,
        "valueLabel": "0.557",
        "display": "55.7",
        "lo": null,
        "hi": null,
        "rankOnBoard": 18,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 50,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
        "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
        "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-LUNA",
        "corroborating": [
          {
            "source": 16,
            "role": "corroborating",
            "scoreDisplay": "55.7",
            "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-LUNA"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-210"
      },
      {
        "id": 506,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "muse-spark",
        "modelKind": "model",
        "name": "Muse Spark",
        "label": null,
        "lab": "meta",
        "variant": "length-normalized, GPT-5.4 low-reasoning grader; Muse Spark (1.0) column in the Muse Spark 1.1 Evaluation Report Figure 44",
        "value": 0.541,
        "valueLabel": "0.541",
        "display": "54.1",
        "lo": null,
        "hi": null,
        "rankOnBoard": 19,
        "rowsOnBoard": 31,
        "measured": "2026-07",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 105,
        "sourceUrl": "https://research.meta.ai/static/muse-spark-1-1-evaluation-report",
        "quote": "Health | HealthBench Professional | 59.3 | 54.1 | 41.6 | 55.8 | 51.8 (Figure 44 image table row; columns Muse Spark 1.1, Muse Spark, Gemini 3.1 Pro (high), Opus 4.8 (max), GPT 5.5 (xhigh))",
        "locator": "p. 101, Figure 44 (image), row HealthBench Professional, column Muse Spark; protocol p. 104 (printed 103)",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-506"
      },
      {
        "id": 492,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.6-sol",
        "modelKind": "model",
        "name": "GPT-5.6 Sol (August)",
        "label": "GPT-5.6 Sol (August)",
        "lab": "openai",
        "variant": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Sol (August) (56.6 unadjusted, 2,894 chars)",
        "value": 0.54,
        "valueLabel": "0.540",
        "display": "54.0",
        "lo": null,
        "hi": null,
        "rankOnBoard": 20,
        "rowsOnBoard": 31,
        "measured": "2026-08",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 30,
        "sourceUrl": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
        "quote": "HealthBench Professional 32.9 (33.8, 2,285) 38.4 (40.7, 2,775) 54.0 (56.6, 2,894) 44.1 (46.8, 2,920)",
        "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Sol (August)",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-492"
      },
      {
        "id": 504,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-opus-4.7",
        "modelKind": "model",
        "name": "Claude Opus 4.7",
        "label": null,
        "lab": "anthropic",
        "variant": "length-adjusted; adaptive thinking at max effort; Claude Sonnet 4.6 grader; 5 trials; comparison model in the Opus 4.8 card",
        "value": 0.519,
        "valueLabel": "0.519",
        "display": "51.9%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 21,
        "rowsOnBoard": 31,
        "measured": "2026-05",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 101,
        "sourceUrl": "https://www.anthropic.com/claude-opus-4-8-system-card",
        "quote": "Claude Opus 4.8 scores 55.8%, a meaningful improvement over Claude Opus 4.7 at 51.9% and Claude Sonnet 4.6 at 41.7%.",
        "locator": "p. 228, section 8.14.1 HealthBench Professional; Figure 8.14.A p. 229",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-504"
      },
      {
        "id": 489,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.5",
        "modelKind": "model",
        "name": "GPT-5.5",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.5 (57.2 unadjusted, 3818 chars)",
        "value": 0.518,
        "valueLabel": "0.518",
        "display": "51.8",
        "lo": null,
        "hi": null,
        "rankOnBoard": 22,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 50,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
        "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
        "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5",
        "corroborating": [
          {
            "source": 16,
            "role": "corroborating",
            "scoreDisplay": "51.8",
            "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.5"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-489"
      },
      {
        "id": 1805,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "grok-4.6",
        "modelKind": "model",
        "name": "Grok 4.6",
        "label": null,
        "lab": "xai",
        "variant": "SpaceXAI evaluation; High effort; HealthBench Professional",
        "value": 0.485,
        "valueLabel": "0.485",
        "display": "48.5%",
        "lo": null,
        "hi": null,
        "rankOnBoard": 23,
        "rowsOnBoard": 31,
        "measured": "2026-09",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 1290,
        "sourceUrl": "https://x.ai/news/grok-4-7",
        "quote": "Grok 4.7 xHigh\nGrok 4.6 High\nGPT-5.6 Sol Max\nFable 5.1 Max\nClinical reasoningHealthBench Professional\n56.7%\n48.5%\n60.5%\n62.1%",
        "locator": "Model Improvements comparison table, Clinical reasoning / HealthBench Professional row; Grok 4.6 column; September 21, 2026.",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-1805"
      },
      {
        "id": 488,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.4",
        "modelKind": "model",
        "name": "GPT-5.4",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.4 (51.9 unadjusted, 3308 chars)",
        "value": 0.481,
        "valueLabel": "0.481",
        "display": "48.1",
        "lo": null,
        "hi": null,
        "rankOnBoard": 24,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 50,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
        "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
        "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.4",
        "corroborating": [
          {
            "source": 16,
            "role": "corroborating",
            "scoreDisplay": "48.1",
            "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.4"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-488"
      },
      {
        "id": 485,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5",
        "modelKind": "model",
        "name": "GPT-5",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5 (51.0 unadjusted, 3616 chars)",
        "value": 0.462,
        "valueLabel": "0.462",
        "display": "46.2",
        "lo": null,
        "hi": null,
        "rankOnBoard": 25,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 50,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
        "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
        "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5",
        "corroborating": [
          {
            "source": 16,
            "role": "corroborating",
            "scoreDisplay": "46.2",
            "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-485"
      },
      {
        "id": 487,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.2",
        "modelKind": "model",
        "name": "GPT-5.2",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.2 (50.0 unadjusted, 3400 chars)",
        "value": 0.459,
        "valueLabel": "0.459",
        "display": "45.9",
        "lo": null,
        "hi": null,
        "rankOnBoard": 26,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 50,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
        "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
        "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.2",
        "corroborating": [
          {
            "source": 16,
            "role": "corroborating",
            "scoreDisplay": "45.9",
            "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.2"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-487"
      },
      {
        "id": 503,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "claude-sonnet-4.6",
        "modelKind": "model",
        "name": "Claude Sonnet 4.6",
        "label": null,
        "lab": "anthropic",
        "variant": "length-adjusted; Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt; comparison column in the Sonnet 5 card. Other Anthropic prints: 44.4% (Fable card Figure 8.18.2.A), 41.7% (Opus 4.8 card, Sonnet 4.6 grader)",
        "value": 0.442,
        "valueLabel": "0.442",
        "display": "44.2",
        "lo": null,
        "hi": null,
        "rankOnBoard": 27,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 99,
        "sourceUrl": "https://www.anthropic.com/claude-sonnet-5-system-card",
        "quote": "HealthBench\nProfessional\n57.8 44.2 51.8 -",
        "locator": "p. 115, Table 8.1.A, row HealthBench Professional, column Claude Sonnet 4.6",
        "corroborating": [
          {
            "source": 101,
            "role": "conflicting",
            "scoreDisplay": "41.7%",
            "quote": "Claude Opus 4.8 scores 55.8%, a meaningful improvement over Claude Opus 4.7 at 51.9% and Claude Sonnet 4.6 at 41.7%.",
            "locator": "p. 228, section 8.14.1"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-503"
      },
      {
        "id": 493,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.6-luna",
        "modelKind": "model",
        "name": "GPT-5.6 Luna (August)",
        "label": "GPT-5.6 Luna (August)",
        "lab": "openai",
        "variant": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Luna (August) (46.8 unadjusted, 2,920 chars)",
        "value": 0.441,
        "valueLabel": "0.441",
        "display": "44.1",
        "lo": null,
        "hi": null,
        "rankOnBoard": 28,
        "rowsOnBoard": 31,
        "measured": "2026-08",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 30,
        "sourceUrl": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
        "quote": "HealthBench Professional 32.9 (33.8, 2,285) 38.4 (40.7, 2,775) 54.0 (56.6, 2,894) 44.1 (46.8, 2,920)",
        "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Luna (August)",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-493"
      },
      {
        "id": 486,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.1",
        "modelKind": "model",
        "name": "GPT-5.1",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.1 (48.0 unadjusted, 4863 chars)",
        "value": 0.396,
        "valueLabel": "0.396",
        "display": "39.6",
        "lo": null,
        "hi": null,
        "rankOnBoard": 29,
        "rowsOnBoard": 31,
        "measured": "2026-06",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 50,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
        "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
        "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.1",
        "corroborating": [
          {
            "source": 16,
            "role": "corroborating",
            "scoreDisplay": "39.6",
            "quote": "HealthBench Professional length-adjusted   46.2 (51.0, 3616)   39.6 (48.0, 4863)   45.9 (50.0, 3400)   48.1 (51.9, 3308)   51.8 (57.2, 3818)   60.5 (64.1, 3228)   57.7 (62.4, 3618)   55.7 (59.8, 3389)",
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.1"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-486"
      },
      {
        "id": 211,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "gpt-5.5-instant",
        "modelKind": "model",
        "name": "GPT-5.5 Instant",
        "label": null,
        "lab": "openai",
        "variant": "length-adjusted (40.7 unadjusted, 2,775 mean response chars); GPT-5.5 Instant system card Table 5, column GPT-5.5 INSTANT.",
        "value": 0.384,
        "valueLabel": "0.384",
        "display": "38.4",
        "lo": null,
        "hi": null,
        "rankOnBoard": 30,
        "rowsOnBoard": 31,
        "measured": "2026-05",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 51,
        "sourceUrl": "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench",
        "quote": "HealthBench Professional   37.6 (40.4, 2,973)   35.7 (38.3, 2,872)   32.9 (33.8, 2,285)   38.4 (40.7, 2,775)",
        "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5 INSTANT",
        "corroborating": [
          {
            "source": 30,
            "role": "corroborating",
            "scoreDisplay": "38.4",
            "quote": "HealthBench Professional 32.9 (33.8, 2,285) 38.4 (40.7, 2,775) 54.0 (56.6, 2,894) 44.1 (46.8, 2,920)",
            "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.5 Instant"
          }
        ],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-211"
      },
      {
        "id": 212,
        "benchmark": "healthbench-professional",
        "metric": null,
        "model": "mai-thinking-1",
        "modelKind": "model",
        "name": "MAI-Thinking-1",
        "label": null,
        "lab": "microsoft",
        "variant": "length-adjusted (HealthBench Professional length penalty), standard GPT-5.4 grader and OpenAI rubrics, Microsoft AI run; printed at integer precision.",
        "value": 0.35,
        "valueLabel": "0.350",
        "display": "35",
        "lo": null,
        "hi": null,
        "rankOnBoard": 31,
        "rowsOnBoard": 31,
        "measured": "2026-08",
        "reportedBy": "vendor",
        "confidence": "verified",
        "source": 75,
        "sourceUrl": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
        "quote": "Model AIR-Bench CyberSec Instruct CyberSec Auto Long Fact Truthful QA HealthBench Prof. MedXpert QA\nMAI-Thinking-1 88 63 63 98 88 35 43\nSonnet 4.6 88 62 56 98 88 38 49",
        "locator": "p. 54, Table 12 'Post-trained model evaluation results on various public benchmarks', Health group, column HealthBench Prof.; protocol Appendix K.6 p. 106: HealthBench Professional introduces a length penalty for the primary metric, to correct for a well-observed correlation between lengthy responses and artificially increased LLM-grader scores. For all reported scores, we use the standard GPT-5.4 grader and rubrics provided by OpenAI.",
        "corroborating": [],
        "url": "https://clinicalbenchmarks.ai/benchmarks/healthbench-professional#r-212"
      }
    ]
  }
}
