{
  "dataset": "Clinical Benchmarks",
  "kind": "sources",
  "url": "https://clinicalbenchmarks.ai",
  "updated": "2026-09-30",
  "revision": 7165,
  "licence": {
    "name": "CC BY 4.0",
    "url": "https://creativecommons.org/licenses/by/4.0/",
    "attribution": "Clinical Benchmarks (clinicalbenchmarks.ai)",
    "note": "The compilation is licensed CC BY 4.0. Source documents keep their own terms."
  },
  "rule": "Each board ranks only its own rows. Scores from different benchmarks are combined only in this site's own index (/api/v1/index.json), which is labelled as such.",
  "data": [
    {
      "id": 1293,
      "title": "ARISE MAST technical leaderboard",
      "url": "https://www.arise-ai.org/mast/technical",
      "kind": "official_leaderboard",
      "publisher": "ARISE",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1293"
    },
    {
      "id": 21,
      "title": "CHI-Bench leaderboard (actAVA)",
      "url": "https://actava.ai/benchmarks/leaderboards",
      "kind": "official_leaderboard",
      "publisher": "actAVA",
      "firstParty": true,
      "publishedAt": "2026-08-12",
      "retrieved": "2026-09-07",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-21"
    },
    {
      "id": 1396,
      "title": "CHI-Bench submission manifest: claude-code + anthropic/claude-fable-5",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-22-fable-5-claude-code/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1396"
    },
    {
      "id": 1379,
      "title": "CHI-Bench submission manifest: claude-code + anthropic/claude-haiku-4-5",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-claude-code-anthropic-claude-haiku-4-5/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1379"
    },
    {
      "id": 1391,
      "title": "CHI-Bench submission manifest: claude-code + anthropic/claude-opus-4-6",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-claude-code-anthropic-claude-opus-4-6/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1391"
    },
    {
      "id": 1395,
      "title": "CHI-Bench submission manifest: claude-code + anthropic/claude-opus-4-7",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-claude-code-anthropic-claude-opus-4-7/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1395"
    },
    {
      "id": 1390,
      "title": "CHI-Bench submission manifest: claude-code + anthropic/claude-opus-4-8",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-05-28-claude-opus-4-8/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1390"
    },
    {
      "id": 1389,
      "title": "CHI-Bench submission manifest: claude-code + anthropic/claude-opus-5",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-24-opus-5-claude-code/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1389"
    },
    {
      "id": 1392,
      "title": "CHI-Bench submission manifest: claude-code + anthropic/claude-sonnet-4-6",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-claude-code-anthropic-claude-sonnet-4-6/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1392"
    },
    {
      "id": 1398,
      "title": "CHI-Bench submission manifest: claude-code + anthropic/claude-sonnet-5",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-06-sonnet-5-claude-code/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1398"
    },
    {
      "id": 1363,
      "title": "CHI-Bench submission manifest: codex + openai/gpt-5.4",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-codex-openai-gpt-5-4/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1363"
    },
    {
      "id": 1377,
      "title": "CHI-Bench submission manifest: codex + openai/gpt-5.4-mini",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-codex-openai-gpt-5-4-mini/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1377"
    },
    {
      "id": 1397,
      "title": "CHI-Bench submission manifest: codex + openai/gpt-5.5",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-codex-openai-gpt-5-5/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1397"
    },
    {
      "id": 1370,
      "title": "CHI-Bench submission manifest: codex + openai/gpt-5.6-luna",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-24-gpt-5-6-luna-codex/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1370"
    },
    {
      "id": 1393,
      "title": "CHI-Bench submission manifest: codex + openai/gpt-5.6-sol",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-24-gpt-5-6-sol-codex/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1393"
    },
    {
      "id": 1369,
      "title": "CHI-Bench submission manifest: codex + openai/gpt-5.6-terra",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-24-gpt-5-6-terra-codex/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1369"
    },
    {
      "id": 1374,
      "title": "CHI-Bench submission manifest: deepagents + openrouter/deepseek/deepseek-v4-pro",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-deepagents-openrouter-deepseek-deepseek-v4-pro/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1374"
    },
    {
      "id": 1383,
      "title": "CHI-Bench submission manifest: deepagents + openrouter/moonshotai/kimi-k2.6",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-deepagents-openrouter-moonshotai-kimi-k2-6/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1383"
    },
    {
      "id": 1376,
      "title": "CHI-Bench submission manifest: deepagents + openrouter/qwen/qwen3-6-max-preview",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-deepagents-openrouter-qwen-qwen3-6-max-preview/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1376"
    },
    {
      "id": 1384,
      "title": "CHI-Bench submission manifest: deepagents + openrouter/x-ai/grok-4.3",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-deepagents-openrouter-x-ai-grok-4-3/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1384"
    },
    {
      "id": 1373,
      "title": "CHI-Bench submission manifest: deepagents + openrouter/z-ai/glm-5.1",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-deepagents-openrouter-z-ai-glm-5-1/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1373"
    },
    {
      "id": 1388,
      "title": "CHI-Bench submission manifest: erius + anthropic/claude-opus-4-8",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-06-05-erius/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1388"
    },
    {
      "id": 1387,
      "title": "CHI-Bench submission manifest: erius + anthropic/claude-opus-5",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-26-erius-opus5/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1387"
    },
    {
      "id": 1371,
      "title": "CHI-Bench submission manifest: gemini-cli + google/gemini-3-flash-preview",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-gemini-cli-google-gemini-3-flash-preview/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1371"
    },
    {
      "id": 1356,
      "title": "CHI-Bench submission manifest: hermes + MedGuard",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-06-cuilinke-hermes-medguard/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1356"
    },
    {
      "id": 1368,
      "title": "CHI-Bench submission manifest: hermes + openrouter/deepseek/deepseek-v4-pro",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-hermes-openrouter-deepseek-deepseek-v4-pro/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1368"
    },
    {
      "id": 1365,
      "title": "CHI-Bench submission manifest: hermes + openrouter/moonshotai/kimi-k2.6",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-hermes-openrouter-moonshotai-kimi-k2-6/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1365"
    },
    {
      "id": 1362,
      "title": "CHI-Bench submission manifest: hermes + openrouter/qwen/qwen3-6-max-preview",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-hermes-openrouter-qwen-qwen3-6-max-preview/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1362"
    },
    {
      "id": 1382,
      "title": "CHI-Bench submission manifest: hermes + openrouter/x-ai/grok-4.3",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-hermes-openrouter-x-ai-grok-4-3/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1382"
    },
    {
      "id": 1358,
      "title": "CHI-Bench submission manifest: hermes + openrouter/z-ai/glm-5.1",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-hermes-openrouter-z-ai-glm-5-1/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1358"
    },
    {
      "id": 1367,
      "title": "CHI-Bench submission manifest: openai-agents + deepseek/deepseek-v4-pro",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openai-agents-deepseek-deepseek-v4-pro/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1367"
    },
    {
      "id": 1366,
      "title": "CHI-Bench submission manifest: openai-agents + moonshotai/kimi-k2.6",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openai-agents-moonshotai-kimi-k2-6/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1366"
    },
    {
      "id": 1394,
      "title": "CHI-Bench submission manifest: openai-agents + moonshotai/kimi-k3",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-24-kimi-k3-openai-agents/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1394"
    },
    {
      "id": 1386,
      "title": "CHI-Bench submission manifest: openai-agents + nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16:peft:262144",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-24-nemotron-3-ultra-openai-agents/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1386"
    },
    {
      "id": 1364,
      "title": "CHI-Bench submission manifest: openai-agents + qwen/qwen3-6-max-preview",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openai-agents-qwen-qwen3-6-max-preview/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1364"
    },
    {
      "id": 1378,
      "title": "CHI-Bench submission manifest: openai-agents + thinkingmachines/Inkling:peft:262144",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-24-inkling-256k-openai-agents/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1378"
    },
    {
      "id": 1380,
      "title": "CHI-Bench submission manifest: openai-agents + x-ai/grok-4.3",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openai-agents-x-ai-grok-4-3/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1380"
    },
    {
      "id": 1357,
      "title": "CHI-Bench submission manifest: openai-agents + z-ai/glm-5.1",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openai-agents-z-ai-glm-5-1/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1357"
    },
    {
      "id": 1359,
      "title": "CHI-Bench submission manifest: openai-agents + z-ai/glm-5.2",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/2026-07-06-glm-5-2-openai-agents/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1359"
    },
    {
      "id": 1360,
      "title": "CHI-Bench submission manifest: openclaw + anthropic/claude-opus-4-7",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openclaw-anthropic-claude-opus-4-7/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1360"
    },
    {
      "id": 1372,
      "title": "CHI-Bench submission manifest: openclaw + openrouter/deepseek/deepseek-v4-pro",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openclaw-openrouter-deepseek-deepseek-v4-pro/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1372"
    },
    {
      "id": 1375,
      "title": "CHI-Bench submission manifest: openclaw + openrouter/moonshotai/kimi-k2.6",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openclaw-openrouter-moonshotai-kimi-k2-6/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1375"
    },
    {
      "id": 1381,
      "title": "CHI-Bench submission manifest: openclaw + openrouter/qwen/qwen3-6-max-preview",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openclaw-openrouter-qwen-qwen3-6-max-preview/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1381"
    },
    {
      "id": 1385,
      "title": "CHI-Bench submission manifest: openclaw + openrouter/x-ai/grok-4.3",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openclaw-openrouter-x-ai-grok-4-3/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1385"
    },
    {
      "id": 1361,
      "title": "CHI-Bench submission manifest: openclaw + openrouter/z-ai/glm-5.1",
      "url": "https://raw.githubusercontent.com/actava-ai/leaderboard/main/benchmarks/chi-bench/submissions/chi-bench-leaderboard-2026-05-15-openclaw-openrouter-z-ai-glm-5-1/submission.json",
      "kind": "official_leaderboard",
      "publisher": "actAVA / CHI-Bench submission repository",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-30",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-1361"
    },
    {
      "id": 61,
      "title": "Claude Fable 5 and Claude Mythos 5 System Card",
      "url": "https://anthropic.com/claude-fable-5-mythos-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-06-09",
      "retrieved": "2026-09-07",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-61"
    },
    {
      "id": 123,
      "title": "Claude Fable 5.1 and Claude Mythos 5.1 System Card",
      "url": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-09-01",
      "retrieved": "2026-09-07",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-123"
    },
    {
      "id": 1288,
      "title": "Claude Opus 5.5 System Card",
      "url": "https://www.anthropic.com/claude-opus-5-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-09-22",
      "retrieved": null,
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-1288"
    },
    {
      "id": 1287,
      "title": "Claude Sonnet 5.5 System Card",
      "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-09-28",
      "retrieved": null,
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-1287"
    },
    {
      "id": 27,
      "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301)",
      "url": "https://arxiv.org/abs/2606.23301",
      "kind": "paper",
      "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
      "firstParty": false,
      "publishedAt": "2026-06-22",
      "retrieved": "2026-09-07",
      "kindLabel": "paper",
      "page": "https://clinicalbenchmarks.ai/sources#source-27"
    },
    {
      "id": 109,
      "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
      "url": "https://arxiv.org/pdf/2606.23301",
      "kind": "paper",
      "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
      "firstParty": false,
      "publishedAt": "2026-06-22",
      "retrieved": "2026-09-07",
      "kindLabel": "paper",
      "page": "https://clinicalbenchmarks.ai/sources#source-109"
    },
    {
      "id": 1223,
      "title": "Evaluating Counterfactual Sensitivity to Patient Information in Medication-Safety Reasoning",
      "url": "https://arxiv.org/html/2608.03028v1",
      "kind": "paper",
      "publisher": "Zhitian Hou, Yuhang Liu, Pengkai Wang, Zeyu Liu, Guanghao Zhu, Zheng Liu, Shuo Cai, Congkai Xie, Zhijie Sang, Kun Zeng, Hongxia Yang (The Hong Kong Polytechnic University; InfiX.ai; Sun Yat-sen University)",
      "firstParty": false,
      "publishedAt": "2026-08-04",
      "retrieved": null,
      "kindLabel": "paper",
      "page": "https://clinicalbenchmarks.ai/sources#source-1223"
    },
    {
      "id": 202,
      "title": "Gemma 4 model card",
      "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "kind": "model_card",
      "publisher": "Google",
      "firstParty": true,
      "publishedAt": "2026-04-02",
      "retrieved": "2026-09-07",
      "kindLabel": "model card",
      "page": "https://clinicalbenchmarks.ai/sources#source-202"
    },
    {
      "id": 212,
      "title": "Gemma 4 Technical Report",
      "url": "https://arxiv.org/pdf/2607.02770",
      "kind": "paper",
      "publisher": "Google DeepMind",
      "firstParty": false,
      "publishedAt": null,
      "retrieved": "2026-09-07",
      "kindLabel": "paper",
      "page": "https://clinicalbenchmarks.ai/sources#source-212"
    },
    {
      "id": 51,
      "title": "GPT-5.5 Instant System Card",
      "url": "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-05-05",
      "retrieved": "2026-09-07",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-51"
    },
    {
      "id": 30,
      "title": "GPT-5.6 - August Updates (system card addendum)",
      "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-08-06",
      "retrieved": "2026-09-07",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-30"
    },
    {
      "id": 16,
      "title": "GPT-5.6 Preview System Card",
      "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-06-26",
      "retrieved": "2026-09-07",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-16"
    },
    {
      "id": 50,
      "title": "GPT-5.6 System Card",
      "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-07-09",
      "retrieved": "2026-09-07",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-50"
    },
    {
      "id": 1104,
      "title": "GPT-6 Astra System Card",
      "url": "https://deploymentsafety.openai.com/gpt-6-astra/gpt-6-astra.pdf",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-09-03",
      "retrieved": "2026-09-08",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-1104"
    },
    {
      "id": 1106,
      "title": "GPT-6 Astra System Card - HealthBench (Deployment Safety Hub)",
      "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-09-03",
      "retrieved": "2026-09-08",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-1106"
    },
    {
      "id": 1107,
      "title": "GPT-6 Astra: A new generation of intelligence",
      "url": "https://openai.com/index/gpt-6-astra/",
      "kind": "launch_post",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-09-03",
      "retrieved": "2026-09-08",
      "kindLabel": "launch post",
      "page": "https://clinicalbenchmarks.ai/sources#source-1107"
    },
    {
      "id": 138,
      "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
      "url": "https://arxiv.org/pdf/2604.09937",
      "kind": "paper",
      "publisher": "Bedi, Welch, Steinberg et al. (Stanford University / Kinetic Systems / Stanford Health Care)",
      "firstParty": false,
      "publishedAt": "2026-04-10",
      "retrieved": "2026-09-07",
      "kindLabel": "paper",
      "page": "https://clinicalbenchmarks.ai/sources#source-138"
    },
    {
      "id": 64,
      "title": "HealthAgentBench detailed results",
      "url": "https://microsoft.github.io/HealthAgentBench/results",
      "kind": "official_leaderboard",
      "publisher": "Microsoft Research (HealthAgentBench)",
      "firstParty": true,
      "publishedAt": "2026-07-27",
      "retrieved": "2026-09-07",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-64"
    },
    {
      "id": 20,
      "title": "HealthAgentBench leaderboard",
      "url": "https://microsoft.github.io/HealthAgentBench/",
      "kind": "official_leaderboard",
      "publisher": "Microsoft Research (HealthAgentBench)",
      "firstParty": true,
      "publishedAt": "2026-07-27",
      "retrieved": "2026-09-07",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-20"
    },
    {
      "id": 46,
      "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
      "url": "https://arxiv.org/pdf/2606.31179",
      "kind": "paper",
      "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
      "firstParty": false,
      "publishedAt": "2026-06-30",
      "retrieved": "2026-09-07",
      "kindLabel": "paper",
      "page": "https://clinicalbenchmarks.ai/sources#source-46"
    },
    {
      "id": 1290,
      "title": "Introducing Grok 4.7",
      "url": "https://x.ai/news/grok-4-7",
      "kind": "launch_post",
      "publisher": "SpaceXAI",
      "firstParty": true,
      "publishedAt": "2026-09-21",
      "retrieved": null,
      "kindLabel": "launch post",
      "page": "https://clinicalbenchmarks.ai/sources#source-1290"
    },
    {
      "id": 29,
      "title": "Introducing HealthAdminBench: AI Agents Can Diagnose Rare Diseases, But Can They Handle Your Insurance? (Kinetic Systems blog)",
      "url": "https://kineticsystems.ai/blog/healthadminbench-automating-healthcare-administration-with-computer-use-agents",
      "kind": "blog",
      "publisher": "Kinetic Systems",
      "firstParty": false,
      "publishedAt": "2026-04-14",
      "retrieved": "2026-09-07",
      "kindLabel": "blog post",
      "page": "https://clinicalbenchmarks.ai/sources#source-29"
    },
    {
      "id": 71,
      "title": "Introducing Muse Spark: Scaling Towards Personal Superintelligence",
      "url": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
      "kind": "launch_post",
      "publisher": "Meta",
      "firstParty": true,
      "publishedAt": "2026-04-08",
      "retrieved": "2026-09-07",
      "kindLabel": "launch post",
      "page": "https://clinicalbenchmarks.ai/sources#source-71"
    },
    {
      "id": 75,
      "title": "MAI-Thinking-1: Building a Hill-Climbing Machine",
      "url": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
      "kind": "model_card",
      "publisher": "Microsoft AI",
      "firstParty": true,
      "publishedAt": "2026-08-12",
      "retrieved": "2026-09-07",
      "kindLabel": "model card",
      "page": "https://clinicalbenchmarks.ai/sources#source-75"
    },
    {
      "id": 19,
      "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
      "url": "https://arise-ai.org/mast/technical",
      "kind": "official_leaderboard",
      "publisher": "ARISE AI Research Network",
      "firstParty": true,
      "publishedAt": "2026-08-15",
      "retrieved": "2026-09-07",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-19"
    },
    {
      "id": 17,
      "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
      "url": "https://arise-ai.org/mast",
      "kind": "official_leaderboard",
      "publisher": "ARISE AI Research Network",
      "firstParty": true,
      "publishedAt": "2026-08-15",
      "retrieved": "2026-09-07",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-17"
    },
    {
      "id": 18,
      "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
      "url": "https://medhelm.org/",
      "kind": "official_leaderboard",
      "publisher": "Stanford CRFM (MedHELM)",
      "firstParty": true,
      "publishedAt": "2026-05-14",
      "retrieved": "2026-09-07",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-18"
    },
    {
      "id": 83,
      "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
      "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
      "kind": "official_leaderboard",
      "publisher": "Stanford CRFM (MedHELM)",
      "firstParty": true,
      "publishedAt": "2026-05-08",
      "retrieved": "2026-09-07",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-83"
    },
    {
      "id": 24,
      "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
      "url": "https://benchlm.ai/benchmarks/medxpertqamm",
      "kind": "mirror",
      "publisher": "benchlm.ai",
      "firstParty": false,
      "publishedAt": null,
      "retrieved": "2026-09-07",
      "kindLabel": "mirror",
      "page": "https://clinicalbenchmarks.ai/sources#source-24"
    },
    {
      "id": 105,
      "title": "Muse Spark 1.1 Evaluation Report",
      "url": "https://research.meta.ai/static/muse-spark-1-1-evaluation-report",
      "kind": "model_card",
      "publisher": "Meta",
      "firstParty": true,
      "publishedAt": "2026-07-09",
      "retrieved": "2026-09-07",
      "kindLabel": "model card",
      "page": "https://clinicalbenchmarks.ai/sources#source-105"
    },
    {
      "id": 74,
      "title": "Muse Spark Eval Methodology",
      "url": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
      "kind": "model_card",
      "publisher": "Meta",
      "firstParty": true,
      "publishedAt": "2026-04-08",
      "retrieved": "2026-09-07",
      "kindLabel": "model card",
      "page": "https://clinicalbenchmarks.ai/sources#source-74"
    },
    {
      "id": 26,
      "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
      "url": "https://arxiv.org/abs/2605.02240",
      "kind": "paper",
      "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
      "firstParty": false,
      "publishedAt": "2026-05-04",
      "retrieved": "2026-09-07",
      "kindLabel": "paper",
      "page": "https://clinicalbenchmarks.ai/sources#source-26"
    },
    {
      "id": 94,
      "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
      "url": "https://arxiv.org/pdf/2605.02240",
      "kind": "paper",
      "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
      "firstParty": false,
      "publishedAt": "2026-05-04",
      "retrieved": "2026-09-07",
      "kindLabel": "paper",
      "page": "https://clinicalbenchmarks.ai/sources#source-94"
    },
    {
      "id": 318,
      "title": "Qwen/Qwen3.5-397B-A17B model card",
      "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
      "kind": "model_card",
      "publisher": "Alibaba / Qwen",
      "firstParty": true,
      "publishedAt": "2026-02-16",
      "retrieved": "2026-09-07",
      "kindLabel": "model card",
      "page": "https://clinicalbenchmarks.ai/sources#source-318"
    },
    {
      "id": 205,
      "title": "Qwen3.7-Plus: Multimodal Agent Intelligence",
      "url": "https://qwen.ai/blog?id=qwen3.7-plus",
      "kind": "launch_post",
      "publisher": "Alibaba",
      "firstParty": true,
      "publishedAt": "2026-05-31",
      "retrieved": "2026-09-07",
      "kindLabel": "launch post",
      "page": "https://clinicalbenchmarks.ai/sources#source-205"
    },
    {
      "id": 152,
      "title": "Qwen3.8-Max: A New Bar for Coding and Cowork",
      "url": "https://qwen.ai/blog?id=qwen3.8",
      "kind": "launch_post",
      "publisher": "Alibaba",
      "firstParty": true,
      "publishedAt": "2026-08-02",
      "retrieved": "2026-09-07",
      "kindLabel": "launch post",
      "page": "https://clinicalbenchmarks.ai/sources#source-152"
    },
    {
      "id": 101,
      "title": "System Card: Claude Opus 4.8",
      "url": "https://www.anthropic.com/claude-opus-4-8-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-05-28",
      "retrieved": "2026-09-07",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-101"
    },
    {
      "id": 98,
      "title": "System Card: Claude Opus 5",
      "url": "https://www.anthropic.com/claude-opus-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-07-24",
      "retrieved": "2026-09-07",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-98"
    },
    {
      "id": 99,
      "title": "System Card: Claude Sonnet 5",
      "url": "https://www.anthropic.com/claude-sonnet-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-06-30",
      "retrieved": "2026-09-07",
      "kindLabel": "system card",
      "page": "https://clinicalbenchmarks.ai/sources#source-99"
    },
    {
      "id": 22,
      "title": "Vals AI MedCode leaderboard",
      "url": "https://www.vals.ai/benchmarks/medcode",
      "kind": "official_leaderboard",
      "publisher": "Vals AI",
      "firstParty": true,
      "publishedAt": "2026-09-29",
      "retrieved": "2026-09-08",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-22"
    },
    {
      "id": 23,
      "title": "Vals AI MedScribe leaderboard",
      "url": "https://www.vals.ai/benchmarks/medscribe",
      "kind": "official_leaderboard",
      "publisher": "Vals AI",
      "firstParty": true,
      "publishedAt": "2026-09-03",
      "retrieved": "2026-09-08",
      "kindLabel": "official leaderboard",
      "page": "https://clinicalbenchmarks.ai/sources#source-23"
    }
  ]
}
