{
  "source": "https://healthevals.com",
  "license": "CC BY 4.0",
  "citation": "Health Evals index, 2026-09-28 snapshot. https://healthevals.com.",
  "updated": "2026-09-28",
  "benchmarks": [
    {
      "slug": "healthbench-professional",
      "name": "HealthBench Professional",
      "publisher": "OpenAI",
      "released": "2026-04-22",
      "unit": "525 physician-authored tasks",
      "scale": "0 to 1, higher is better",
      "description": "525 tasks that physicians picked out of 15,079 real workplace AI conversations, spanning care consults, clinical documentation, and medical research, each judged on a rubric physicians wrote for it.",
      "basis": "mixed",
      "sourceName": "Published model evaluation reports",
      "sourceUrl": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
      "ours": "https://healthbenchprofessional.com",
      "lastUpdate": "2026-09",
      "notes": "All displayed values use a 0–1 scale (source percentages divided by 100). Most are length-adjusted, but grader models, safeguards and evaluator protocols differ. Anthropic’s Astra reproduction (Opus 4.8 grader) is a separate row from OpenAI’s own run. Grok’s release table does not fully document its grading protocol. A score-source check is not an independent rerun.",
      "results": [
        {
          "model": "GPT-6 Astra (Anthropic run)",
          "lab": "OpenAI",
          "scoreDisplay": "0.703",
          "scoreNumeric": 0.703,
          "date": "2026-09",
          "config": "Anthropic reproduction of GPT-6 Astra through the public API; max effort; no system prompt; Claude Opus 4.8 grader; length-adjusted 70.3%, raw 74.0%. Different grader/protocol from OpenAI’s own 64.7% report.",
          "modelSlug": "gpt-6-astra",
          "scorePrinted": "70.3",
          "variant": "Anthropic reproduction of GPT-6 Astra through the public API; max effort; no system prompt; Claude Opus 4.8 grader; length-adjusted 70.3%, raw 74.0%. Different grader/protocol from OpenAI’s own 64.7% report.",
          "measured": "2026-09",
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10011,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 8.15, pp. 137–138; text above Figure 8.15.B: max-effort Astra 70.3 adjusted and 74.0 raw"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "0.692",
          "scoreNumeric": 0.6920000000000001,
          "date": "2026-09",
          "config": "Anthropic evaluation; length-adjusted; adaptive thinking at max effort; Claude Opus 4.8 grader; no tools or custom system prompt; safety classifiers enabled. Raw score 77.1%. The paper reports raw and adjusted scores separately; the max-effort HealthBench chart label is 65.4%.",
          "modelSlug": "claude-sonnet-5.5",
          "scorePrinted": "69.2",
          "variant": "Anthropic evaluation; length-adjusted; adaptive thinking at max effort; Claude Opus 4.8 grader; no tools or custom system prompt; safety classifiers enabled. Raw score 77.1%. The paper reports raw and adjusted scores separately; the max-effort HealthBench chart label is 65.4%.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10009,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 8.15.2; pp. 137–139, Figure 8.15.B (max effort)"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "0.656",
          "scoreNumeric": 0.6559999999999999,
          "date": "2026-09",
          "config": "Anthropic evaluation; length-adjusted; adaptive thinking at max effort; Claude Opus 4.8 grader; no tools or custom system prompt; safety classifiers enabled. Raw score 77.1%. Five-trial average; refusal fallback to Claude Opus 5.",
          "modelSlug": "claude-opus-5.5",
          "scorePrinted": "65.6",
          "variant": "Anthropic evaluation; length-adjusted; adaptive thinking at max effort; Claude Opus 4.8 grader; no tools or custom system prompt; safety classifiers enabled. Raw score 77.1%. Five-trial average; refusal fallback to Claude Opus 5.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10007,
          "source": {
            "id": 10510,
            "title": "Claude Opus 5.5 System Card",
            "url": "https://www.anthropic.com/claude-opus-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 8.15.2; pp. 213–214, Figures 8.15.1.A and 8.15.2.A"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Astra",
          "lab": "OpenAI",
          "scoreDisplay": "0.647",
          "scoreNumeric": 0.647,
          "date": "2026-09",
          "config": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 68.2%, mean answer 3,185 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "modelSlug": "gpt-6-astra",
          "scorePrinted": "64.7",
          "variant": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 68.2%, mean answer 3,185 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 563,
          "source": {
            "id": 10501,
            "title": "GPT-6 Astra System Card — September 22 revision",
            "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-09-03",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 11.4.1, Table 29; HealthBench Professional length-adjusted row, GPT-6 Astra column; value 64.7 (68.2, 3185)"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "0.633",
          "scoreNumeric": 0.633,
          "date": "2026-09",
          "config": "Anthropic evaluation of Claude Fable 5, length-adjusted; adaptive max effort, Opus 4.8 grader, five trials, no tools or custom system prompt; raw 68.9%. Explicit Fable 5 figure in the September 1 card; earlier catalog value came from a Mythos 5 column and is not used for Fable 5.",
          "modelSlug": "claude-fable-5",
          "scorePrinted": "63.3",
          "variant": "Anthropic evaluation of Claude Fable 5, length-adjusted; adaptive max effort, Opus 4.8 grader, five trials, no tools or custom system prompt; raw 68.9%. Explicit Fable 5 figure in the September 1 card; earlier catalog value came from a Mythos 5 column and is not used for Fable 5.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 204,
          "source": {
            "id": 10515,
            "title": "Claude Fable 5.1 and Claude Mythos 5.1 System Card",
            "url": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-01",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 199, Figure 8.17.2.A; Claude Fable 5, adjusted bar 63.3%; also p. 167 Table 8.1.A"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5.1",
          "lab": "Anthropic",
          "scoreDisplay": "0.621",
          "scoreNumeric": 0.621,
          "date": "2026-09",
          "config": "length-adjusted (method published in the HealthBench Professional paper); Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over five trials, no tools or customized system prompt; Fable 5.1 run with safety classifiers active and a refusal-fallback to Claude Opus 5 (raw 74.2%).",
          "modelSlug": "claude-fable-5.1",
          "scorePrinted": "62.1%",
          "variant": "length-adjusted (method published in the HealthBench Professional paper); Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over five trials, no tools or customized system prompt; Fable 5.1 run with safety classifiers active and a refusal-fallback to Claude Opus 5 (raw 74.2%).",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 561,
          "source": {
            "id": 10515,
            "title": "Claude Fable 5.1 and Claude Mythos 5.1 System Card",
            "url": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-01",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 199, sec. 8.17.2 (same figure printed as 62.1% in Table 8.1.A, p. 167)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "0.608",
          "scoreNumeric": 0.608,
          "date": "2026-09",
          "config": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 59.5%, mean answer 1,573 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "modelSlug": "gpt-6-sol",
          "scorePrinted": "60.8",
          "variant": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 59.5%, mean answer 1,573 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10001,
          "source": {
            "id": 10501,
            "title": "GPT-6 Astra System Card — September 22 revision",
            "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-09-03",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 11.4.1, Table 29; HealthBench Professional length-adjusted row, GPT-6 Sol column; value 60.8 (59.5, 1573)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "0.608",
          "scoreNumeric": 0.608,
          "date": "2026-09",
          "config": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 61.2%, mean answer 2,119 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "modelSlug": "gpt-6-luna",
          "scorePrinted": "60.8",
          "variant": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 61.2%, mean answer 2,119 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10002,
          "source": {
            "id": 10501,
            "title": "GPT-6 Astra System Card — September 22 revision",
            "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-09-03",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 11.4.1, Table 29; HealthBench Professional length-adjusted row, GPT-6 Luna column; value 60.8 (61.2, 2119)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "0.605",
          "scoreNumeric": 0.605,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort (64.1 unadjusted, 3,228 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-SOL. API max-effort setting, not the August ChatGPT production setting measured in row 492.",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "60.5",
          "variant": "length-adjusted, max reasoning effort (64.1 unadjusted, 3,228 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-SOL. API max-effort setting, not the August ChatGPT production setting measured in row 492.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 205,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-SOL"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "60.5",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-SOL"
            }
          ]
        },
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "0.598",
          "scoreNumeric": 0.598,
          "date": "2026-07",
          "config": "length-adjusted, Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt (raw 73.4%).",
          "modelSlug": "claude-opus-5",
          "scorePrinted": "59.8%",
          "variant": "length-adjusted, Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt (raw 73.4%).",
          "measured": "2026-07",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 206,
          "source": {
            "id": 98,
            "title": "System Card: Claude Opus 5",
            "url": "https://www.anthropic.com/claude-opus-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-07-24",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 189, section 8.15.2 HealthBench Professional results; also Table 8.1.A p. 152 ('HealthBench Professional 59.8 ...')"
          },
          "corroborating": [
            {
              "id": 10515,
              "title": "Claude Fable 5.1 and Claude Mythos 5.1 System Card",
              "url": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
              "kind": "system_card",
              "publisher": "Anthropic",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "59.8%",
              "quote": null,
              "locator": "p. 167, Table 8.1.A, column 'Claude Opus 5'; raw 73.4% also reprinted on p. 199, sec. 8.17.2"
            }
          ]
        },
        {
          "model": "Muse Spark 1.1",
          "lab": "Meta",
          "scoreDisplay": "0.593",
          "scoreNumeric": 0.593,
          "date": "2026-07",
          "config": "length-normalized, GPT-5.4 low-reasoning grader, xhigh reasoning via Meta Model API (Muse Spark 1.1 Evaluation Report Figure 44)",
          "modelSlug": "muse-spark-1.1",
          "scorePrinted": "59.3",
          "variant": "length-normalized, GPT-5.4 low-reasoning grader, xhigh reasoning via Meta Model API (Muse Spark 1.1 Evaluation Report Figure 44)",
          "measured": "2026-07",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 505,
          "source": {
            "id": 105,
            "title": "Muse Spark 1.1 Evaluation Report",
            "url": "https://research.meta.ai/static/muse-spark-1-1-evaluation-report",
            "kind": "model_card",
            "publisher": "Meta",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 101, Figure 44 'General capability benchmark results' (image), row HealthBench Professional, column Muse Spark 1.1; protocol p. 104 (printed 103): HealthBench Pro comprises 525 evaluation data points graded by rubrics. We use GPT-5.4 with low reasoning effort as the grader and report the length-normalized rubric score as done in their paper."
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5",
          "lab": "Anthropic",
          "scoreDisplay": "0.578",
          "scoreNumeric": 0.578,
          "date": "2026-06",
          "config": "length-adjusted, Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt (raw 62.4%).",
          "modelSlug": "claude-sonnet-5",
          "scorePrinted": "57.8",
          "variant": "length-adjusted, Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt (raw 62.4%).",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 207,
          "source": {
            "id": 99,
            "title": "System Card: Claude Sonnet 5",
            "url": "https://www.anthropic.com/claude-sonnet-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-06-30",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 115, Table 8.1.A, row HealthBench Professional, column Claude Sonnet 5; Figure 8.12.2.A p. 139"
          },
          "corroborating": [
            {
              "id": 98,
              "title": "System Card: Claude Opus 5",
              "url": "https://www.anthropic.com/claude-opus-5-system-card",
              "kind": "system_card",
              "publisher": "Anthropic",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "57.8%",
              "quote": null,
              "locator": "p. 189, section 8.15.2, Figure 8.15.2.A"
            }
          ]
        },
        {
          "model": "GPT-5.6 Terra",
          "lab": "OpenAI",
          "scoreDisplay": "0.577",
          "scoreNumeric": 0.577,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort (62.4 unadjusted, 3,618 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-TERRA.",
          "modelSlug": "gpt-5.6-terra",
          "scorePrinted": "57.7",
          "variant": "length-adjusted, max reasoning effort (62.4 unadjusted, 3,618 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-TERRA.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 208,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-TERRA"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "57.7",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-TERRA"
            }
          ]
        },
        {
          "model": "Claude Opus 4.8",
          "lab": "Anthropic",
          "scoreDisplay": "0.574",
          "scoreNumeric": 0.574,
          "date": "2026-06",
          "config": "Length-adjusted; Anthropic June evaluation, adaptive max effort, Claude Opus 4.8 grader, five-trial average, no tools or custom system prompt. Earlier May report was 55.8% using Claude Sonnet 4.6 as grader; the grader changed.",
          "modelSlug": "claude-opus-4.8",
          "scorePrinted": "57.4",
          "variant": "Length-adjusted; Anthropic June evaluation, adaptive max effort, Claude Opus 4.8 grader, five-trial average, no tools or custom system prompt. Earlier May report was 55.8% using Claude Sonnet 4.6 as grader; the grader changed.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 209,
          "source": {
            "id": 99,
            "title": "Claude Sonnet 5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-06-30",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 139, Figure 8.12.2.A; Opus 4.8 bar 57.4%"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.7",
          "lab": "xAI",
          "scoreDisplay": "0.567",
          "scoreNumeric": 0.5670000000000001,
          "date": "2026-09",
          "config": "SpaceXAI vendor report at xhigh effort. HealthBench Professional release-table score; the page does not specify grader or explicitly label length adjustment, so exact protocol comparability is unconfirmed.",
          "modelSlug": "grok-4.7",
          "scorePrinted": "56.7",
          "variant": "SpaceXAI vendor report at xhigh effort. HealthBench Professional release-table score; the page does not specify grader or explicitly label length adjustment, so exact protocol comparability is unconfirmed.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "partial",
          "resultId": 10012,
          "source": {
            "id": 10518,
            "title": "Introducing Grok 4.7",
            "url": "https://x.ai/news/grok-4-7",
            "kind": "launch_post",
            "publisher": "SpaceXAI",
            "firstParty": true,
            "publishedAt": "2026-09-21",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Model Improvements table, Clinical reasoning / HealthBench Professional row, Grok 4.7 xhigh column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "0.557",
          "scoreNumeric": 0.557,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort (59.8 unadjusted, 3,389 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-LUNA. API max-effort setting, not the August ChatGPT production setting measured in row 493.",
          "modelSlug": "gpt-5.6-luna",
          "scorePrinted": "55.7",
          "variant": "length-adjusted, max reasoning effort (59.8 unadjusted, 3,389 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-LUNA. API max-effort setting, not the August ChatGPT production setting measured in row 493.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 210,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-LUNA"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "55.7",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-LUNA"
            }
          ]
        },
        {
          "model": "Muse Spark",
          "lab": "Meta",
          "scoreDisplay": "0.541",
          "scoreNumeric": 0.541,
          "date": "2026-07",
          "config": "length-normalized, GPT-5.4 low-reasoning grader; Muse Spark (1.0) column in the Muse Spark 1.1 Evaluation Report Figure 44",
          "modelSlug": "muse-spark",
          "scorePrinted": "54.1",
          "variant": "length-normalized, GPT-5.4 low-reasoning grader; Muse Spark (1.0) column in the Muse Spark 1.1 Evaluation Report Figure 44",
          "measured": "2026-07",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 506,
          "source": {
            "id": 105,
            "title": "Muse Spark 1.1 Evaluation Report",
            "url": "https://research.meta.ai/static/muse-spark-1-1-evaluation-report",
            "kind": "model_card",
            "publisher": "Meta",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 101, Figure 44 (image), row HealthBench Professional, column Muse Spark; protocol p. 104 (printed 103)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Sol (August)",
          "lab": "OpenAI",
          "scoreDisplay": "0.540",
          "scoreNumeric": 0.54,
          "date": "2026-08",
          "config": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Sol (August) (56.6 unadjusted, 2,894 chars)",
          "alias": "GPT-5.6 Sol",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "54.0",
          "variant": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Sol (August) (56.6 unadjusted, 2,894 chars)",
          "measured": "2026-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 492,
          "source": {
            "id": 30,
            "title": "GPT-5.6 - August Updates (system card addendum)",
            "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-08-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Sol (August)"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.7",
          "lab": "Anthropic",
          "scoreDisplay": "0.519",
          "scoreNumeric": 0.519,
          "date": "2026-05",
          "config": "length-adjusted; adaptive thinking at max effort; Claude Sonnet 4.6 grader; 5 trials; comparison model in the Opus 4.8 card",
          "modelSlug": "claude-opus-4.7",
          "scorePrinted": "51.9%",
          "variant": "length-adjusted; adaptive thinking at max effort; Claude Sonnet 4.6 grader; 5 trials; comparison model in the Opus 4.8 card",
          "measured": "2026-05",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 504,
          "source": {
            "id": 101,
            "title": "System Card: Claude Opus 4.8",
            "url": "https://www.anthropic.com/claude-opus-4-8-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-05-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 228, section 8.14.1 HealthBench Professional; Figure 8.14.A p. 229"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.5",
          "lab": "OpenAI",
          "scoreDisplay": "0.518",
          "scoreNumeric": 0.518,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.5 (57.2 unadjusted, 3818 chars)",
          "modelSlug": "gpt-5.5",
          "scorePrinted": "51.8",
          "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.5 (57.2 unadjusted, 3818 chars)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 489,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "51.8",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.5"
            }
          ]
        },
        {
          "model": "Grok 4.6",
          "lab": "xAI",
          "scoreDisplay": "0.485",
          "scoreNumeric": 0.485,
          "date": "2026-09",
          "config": "SpaceXAI vendor report at high effort. HealthBench Professional release-table score; the page does not specify grader or explicitly label length adjustment, so exact protocol comparability is unconfirmed.",
          "modelSlug": "grok-4.6",
          "scorePrinted": "48.5",
          "variant": "SpaceXAI vendor report at high effort. HealthBench Professional release-table score; the page does not specify grader or explicitly label length adjustment, so exact protocol comparability is unconfirmed.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "partial",
          "resultId": 10013,
          "source": {
            "id": 10518,
            "title": "Introducing Grok 4.7",
            "url": "https://x.ai/news/grok-4-7",
            "kind": "launch_post",
            "publisher": "SpaceXAI",
            "firstParty": true,
            "publishedAt": "2026-09-21",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Model Improvements table, Clinical reasoning / HealthBench Professional row, Grok 4.6 high column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "0.481",
          "scoreNumeric": 0.481,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.4 (51.9 unadjusted, 3308 chars)",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "48.1",
          "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.4 (51.9 unadjusted, 3308 chars)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 488,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.4"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "48.1",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.4"
            }
          ]
        },
        {
          "model": "GPT-5",
          "lab": "OpenAI",
          "scoreDisplay": "0.462",
          "scoreNumeric": 0.462,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5 (51.0 unadjusted, 3616 chars)",
          "modelSlug": "gpt-5",
          "scorePrinted": "46.2",
          "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5 (51.0 unadjusted, 3616 chars)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 485,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "46.2",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5"
            }
          ]
        },
        {
          "model": "GPT-5.2",
          "lab": "OpenAI",
          "scoreDisplay": "0.459",
          "scoreNumeric": 0.459,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.2 (50.0 unadjusted, 3400 chars)",
          "modelSlug": "gpt-5.2",
          "scorePrinted": "45.9",
          "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.2 (50.0 unadjusted, 3400 chars)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 487,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.2"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "45.9",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.2"
            }
          ]
        },
        {
          "model": "Claude Sonnet 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "0.442",
          "scoreNumeric": 0.442,
          "date": "2026-06",
          "config": "length-adjusted; Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt; comparison column in the Sonnet 5 card. Other Anthropic prints: 44.4% (Fable card Figure 8.18.2.A), 41.7% (Opus 4.8 card, Sonnet 4.6 grader)",
          "modelSlug": "claude-sonnet-4.6",
          "scorePrinted": "44.2",
          "variant": "length-adjusted; Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt; comparison column in the Sonnet 5 card. Other Anthropic prints: 44.4% (Fable card Figure 8.18.2.A), 41.7% (Opus 4.8 card, Sonnet 4.6 grader)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 503,
          "source": {
            "id": 99,
            "title": "System Card: Claude Sonnet 5",
            "url": "https://www.anthropic.com/claude-sonnet-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-06-30",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 115, Table 8.1.A, row HealthBench Professional, column Claude Sonnet 4.6"
          },
          "corroborating": [
            {
              "id": 101,
              "title": "System Card: Claude Opus 4.8",
              "url": "https://www.anthropic.com/claude-opus-4-8-system-card",
              "kind": "system_card",
              "publisher": "Anthropic",
              "firstParty": true,
              "role": "conflicting",
              "scoreDisplay": "41.7%",
              "quote": null,
              "locator": "p. 228, section 8.14.1"
            }
          ]
        },
        {
          "model": "GPT-5.6 Luna (August)",
          "lab": "OpenAI",
          "scoreDisplay": "0.441",
          "scoreNumeric": 0.441,
          "date": "2026-08",
          "config": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Luna (August) (46.8 unadjusted, 2,920 chars)",
          "alias": "GPT-5.6 Luna",
          "modelSlug": "gpt-5.6-luna",
          "scorePrinted": "44.1",
          "variant": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Luna (August) (46.8 unadjusted, 2,920 chars)",
          "measured": "2026-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 493,
          "source": {
            "id": 30,
            "title": "GPT-5.6 - August Updates (system card addendum)",
            "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-08-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Luna (August)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.1",
          "lab": "OpenAI",
          "scoreDisplay": "0.396",
          "scoreNumeric": 0.396,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.1 (48.0 unadjusted, 4863 chars)",
          "modelSlug": "gpt-5.1",
          "scorePrinted": "39.6",
          "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.1 (48.0 unadjusted, 4863 chars)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 486,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.1"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "39.6",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.1"
            }
          ]
        },
        {
          "model": "GPT-5.5 Instant",
          "lab": "OpenAI",
          "scoreDisplay": "0.384",
          "scoreNumeric": 0.384,
          "date": "2026-05",
          "config": "length-adjusted (40.7 unadjusted, 2,775 mean response chars); GPT-5.5 Instant system card Table 5, column GPT-5.5 INSTANT.",
          "modelSlug": "gpt-5.5-instant",
          "scorePrinted": "38.4",
          "variant": "length-adjusted (40.7 unadjusted, 2,775 mean response chars); GPT-5.5 Instant system card Table 5, column GPT-5.5 INSTANT.",
          "measured": "2026-05",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 211,
          "source": {
            "id": 51,
            "title": "GPT-5.5 Instant System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-05-05",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5 INSTANT"
          },
          "corroborating": [
            {
              "id": 30,
              "title": "GPT-5.6 - August Updates (system card addendum)",
              "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "38.4",
              "quote": null,
              "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.5 Instant"
            }
          ]
        },
        {
          "model": "MAI-Thinking-1",
          "lab": "Microsoft",
          "scoreDisplay": "0.350",
          "scoreNumeric": 0.35,
          "date": "2026-08",
          "config": "length-adjusted (HealthBench Professional length penalty), standard GPT-5.4 grader and OpenAI rubrics, Microsoft AI run; printed at integer precision.",
          "modelSlug": "mai-thinking-1",
          "scorePrinted": "35",
          "variant": "length-adjusted (HealthBench Professional length penalty), standard GPT-5.4 grader and OpenAI rubrics, Microsoft AI run; printed at integer precision.",
          "measured": "2026-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 212,
          "source": {
            "id": 75,
            "title": "MAI-Thinking-1: Building a Hill-Climbing Machine",
            "url": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
            "kind": "model_card",
            "publisher": "Microsoft AI",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 54, Table 12 'Post-trained model evaluation results on various public benchmarks', Health group, column HealthBench Prof.; protocol Appendix K.6 p. 106: HealthBench Professional introduces a length penalty for the primary metric, to correct for a well-observed correlation between lengthy responses and artificially increased LLM-grader scores. For all reported scores, we use the standard GPT-5.4 grader and rubrics provided by OpenAI."
          },
          "corroborating": []
        }
      ],
      "unitShort": "525 tasks",
      "publisherShort": "OpenAI",
      "sourceShort": "Vendor reports and independent runs",
      "fieldSize": 29,
      "category": "rubric",
      "confidence": "partial",
      "officialUrl": "https://arxiv.org/abs/2604.27470",
      "paperUrl": "https://arxiv.org/abs/2604.27470",
      "scaleKind": "fraction",
      "summary": "525 clinician tasks with physician-written rubrics; results retain evaluator, grader and length-adjustment details.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "partial",
        "summary": "Updated through the September 28 Sonnet 5.5 card, September 22 OpenAI correction and GPT-6 appendix. All source numbers checked. Evaluators use different graders; the Anthropic Astra reproduction is separate. Grok scores are numeric-source verified but protocol details are incomplete.",
        "urls": [
          "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
          "https://www.anthropic.com/claude-sonnet-5-5-system-card",
          "https://www.anthropic.com/claude-opus-5-5-system-card",
          "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
          "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
          "https://www.anthropic.com/claude-opus-5-system-card",
          "https://research.meta.ai/static/muse-spark-1-1-evaluation-report",
          "https://www.anthropic.com/claude-sonnet-5-system-card",
          "https://x.ai/news/grok-4-7",
          "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
          "https://www.anthropic.com/claude-opus-4-8-system-card",
          "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench",
          "https://microsoft.ai/pdf/mai-thinking-1.pdf"
        ]
      }
    },
    {
      "slug": "healthbench-hard",
      "name": "HealthBench Hard",
      "publisher": "OpenAI",
      "released": "2025-05-12",
      "unit": "1,000 conversations",
      "scale": "0 to 1, higher is better",
      "description": "The bottom fifth of HealthBench: 1,000 conversations where frontier models failed most at the May 2025 release, still graded on the original physician-written rubrics.",
      "basis": "mixed",
      "sourceName": "healthbenchhard.ai",
      "sourceUrl": "https://healthbenchhard.ai",
      "ours": "https://healthbenchhard.ai",
      "lastUpdate": "2026-09",
      "notes": "The 0–1 display divides source percentages by 100. OpenAI’s newer rows are length-adjusted; Meta, Baichuan, gpt-oss and GPT-5.3 launch rows report raw scores. These protocols are not directly interchangeable. The September 22 Astra correction and GPT-6 Sol/Luna appendix are included.",
      "results": [
        {
          "model": "Baichuan-M3",
          "lab": "Baichuan",
          "scoreDisplay": "0.444",
          "scoreNumeric": 0.444,
          "date": "2026-02",
          "config": "Raw/unadjusted HealthBench Hard score as evaluated by Baichuan in its M3 report; distinct from OpenAI’s length-adjusted protocol.",
          "modelSlug": "baichuan-m3",
          "scorePrinted": "44.4",
          "variant": "Raw/unadjusted HealthBench Hard score as evaluated by Baichuan in its M3 report; distinct from OpenAI’s length-adjusted protocol.",
          "measured": "2026-02",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10018,
          "source": {
            "id": 10524,
            "title": "Baichuan-M3 Technical Report",
            "url": "https://arxiv.org/pdf/2602.06570",
            "kind": "paper",
            "publisher": "Baichuan",
            "firstParty": true,
            "publishedAt": "2026-02-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 23 (PDF page index 22), section 4.2.1 and Figure 7; HealthBench Hard"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark",
          "lab": "Meta",
          "scoreDisplay": "0.428",
          "scoreNumeric": 0.428,
          "date": "2026-04",
          "config": "raw score (no length adjustment), GPT-4.1 grader via the OpenAI simple-evals implementation, Muse Spark Thinking; Meta run, launch-post benchmark table.",
          "modelSlug": "muse-spark",
          "scorePrinted": "42.8",
          "variant": "raw score (no length adjustment), GPT-4.1 grader via the OpenAI simple-evals implementation, Muse Spark Thinking; Meta run, launch-post benchmark table.",
          "measured": "2026-04",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 195,
          "source": {
            "id": 71,
            "title": "Introducing Muse Spark: Scaling Towards Personal Superintelligence",
            "url": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
            "kind": "launch_post",
            "publisher": "Meta",
            "firstParty": true,
            "publishedAt": "2026-04-08",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Launch-post benchmark table image, HEALTH section, row HealthBench Hard, column Muse Spark Thinking; identical table in the Eval Methodology PDF p. 5; protocol p. 2: HealthBench Hard: This is a subset of OpenAI's HealthBench benchmark, containing 1000 prompts. We used the same implementation as in the OpenAI’s official simple-evals repo, with GPT-4.1-genai as the LLM-as-judge model."
          },
          "corroborating": [
            {
              "id": 74,
              "title": "Muse Spark Eval Methodology",
              "url": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
              "kind": "model_card",
              "publisher": "Meta",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "42.8",
              "quote": null,
              "locator": "p. 5 results table image, row HealthBench Hard; protocol p. 2"
            }
          ]
        },
        {
          "model": "GPT-5.2-High (Baichuan run)",
          "lab": "OpenAI",
          "scoreDisplay": "0.420",
          "scoreNumeric": 0.42,
          "date": "2026-02",
          "config": "Raw/unadjusted HealthBench Hard score as evaluated by Baichuan in its M3 report; distinct from OpenAI’s length-adjusted protocol.",
          "modelSlug": "gpt-5.2",
          "scorePrinted": "42",
          "variant": "Raw/unadjusted HealthBench Hard score as evaluated by Baichuan in its M3 report; distinct from OpenAI’s length-adjusted protocol.",
          "measured": "2026-02",
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10019,
          "source": {
            "id": 10524,
            "title": "Baichuan-M3 Technical Report",
            "url": "https://arxiv.org/pdf/2602.06570",
            "kind": "paper",
            "publisher": "Baichuan",
            "firstParty": true,
            "publishedAt": "2026-02-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 23 (PDF page index 22), section 4.2.1 and Figure 7; HealthBench Hard"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Astra",
          "lab": "OpenAI",
          "scoreDisplay": "0.366",
          "scoreNumeric": 0.366,
          "date": "2026-09",
          "config": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 34.2%, mean answer 1,697 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "modelSlug": "gpt-6-astra",
          "scorePrinted": "36.6",
          "variant": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 34.2%, mean answer 1,697 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 565,
          "source": {
            "id": 10501,
            "title": "GPT-6 Astra System Card — September 22 revision",
            "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-09-03",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 11.4.1, Table 29; HealthBench Hard length-adjusted row, GPT-6 Astra column; value 36.6 (34.2, 1697)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5",
          "lab": "OpenAI",
          "scoreDisplay": "0.347",
          "scoreNumeric": 0.347,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort (41.6 unadjusted, 2,880 mean response chars); GPT-5.6 system card Table 6, column GPT-5. The GPT-5 launch system card printed 46.2% raw for gpt-5-thinking.",
          "modelSlug": "gpt-5",
          "scorePrinted": "34.7",
          "variant": "length-adjusted, max reasoning effort (41.6 unadjusted, 2,880 mean response chars); GPT-5.6 system card Table 6, column GPT-5. The GPT-5 launch system card printed 46.2% raw for gpt-5-thinking.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 203,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "34.7",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5"
            },
            {
              "id": 55,
              "title": "GPT-5.5 System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-5/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "34.7",
              "quote": null,
              "locator": "Section 5 Health, Table 7, column GPT-5"
            },
            {
              "id": 79,
              "title": "GPT-5 System Card",
              "url": "https://cdn.openai.com/gpt-5-system-card.pdf",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "conflicting",
              "scoreDisplay": "46.2",
              "quote": null,
              "locator": "p. 18, section 3.10 Health, Figure 6 (HealthBench Hard, raw score %)"
            }
          ]
        },
        {
          "model": "GPT-5.2",
          "lab": "OpenAI",
          "scoreDisplay": "0.343",
          "scoreNumeric": 0.343,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.2 (38.9 unadjusted, 2585 chars)",
          "modelSlug": "gpt-5.2",
          "scorePrinted": "34.3",
          "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.2 (38.9 unadjusted, 2585 chars)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 482,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.2"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "34.3",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.2"
            }
          ]
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "0.331",
          "scoreNumeric": 0.331,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort (31.1 unadjusted, 1,751 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-SOL. API max-effort setting, not the August ChatGPT production setting measured in row 490.",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "33.1",
          "variant": "length-adjusted, max reasoning effort (31.1 unadjusted, 1,751 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-SOL. API max-effort setting, not the August ChatGPT production setting measured in row 490.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 196,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-SOL"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "33.1",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-SOL"
            }
          ]
        },
        {
          "model": "GPT-5.6 Terra",
          "lab": "OpenAI",
          "scoreDisplay": "0.327",
          "scoreNumeric": 0.327,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort (34.3 unadjusted, 2,199 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-TERRA.",
          "modelSlug": "gpt-5.6-terra",
          "scorePrinted": "32.7",
          "variant": "length-adjusted, max reasoning effort (34.3 unadjusted, 2,199 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-TERRA.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 197,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-TERRA"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "32.7",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-TERRA"
            }
          ]
        },
        {
          "model": "GPT-5.6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "0.320",
          "scoreNumeric": 0.32,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort (31.4 unadjusted, 1,923 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-LUNA. API max-effort setting, not the August ChatGPT production setting measured in row 491.",
          "modelSlug": "gpt-5.6-luna",
          "scorePrinted": "32.0",
          "variant": "length-adjusted, max reasoning effort (31.4 unadjusted, 1,923 mean response chars); GPT-5.6 system card Table 6, column GPT-5.6-LUNA. API max-effort setting, not the August ChatGPT production setting measured in row 491.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 198,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-LUNA"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "32.0",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-LUNA"
            }
          ]
        },
        {
          "model": "GPT-5.5",
          "lab": "OpenAI",
          "scoreDisplay": "0.315",
          "scoreNumeric": 0.315,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.5 (33.8 unadjusted, 2289 chars)",
          "modelSlug": "gpt-5.5",
          "scorePrinted": "31.5",
          "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.5 (33.8 unadjusted, 2289 chars)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 484,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "31.5",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.5"
            }
          ]
        },
        {
          "model": "GPT-5.6 Sol (August)",
          "lab": "OpenAI",
          "scoreDisplay": "0.314",
          "scoreNumeric": 0.314,
          "date": "2026-08",
          "config": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Sol (August) (27.1 unadjusted, 1,450 chars)",
          "alias": "GPT-5.6 Sol",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "31.4",
          "variant": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Sol (August) (27.1 unadjusted, 1,450 chars)",
          "measured": "2026-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 490,
          "source": {
            "id": 30,
            "title": "GPT-5.6 - August Updates (system card addendum)",
            "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-08-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Sol (August)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "0.314",
          "scoreNumeric": 0.314,
          "date": "2026-09",
          "config": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 25.4%, mean answer 1,241 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "modelSlug": "gpt-6-luna",
          "scorePrinted": "31.4",
          "variant": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 25.4%, mean answer 1,241 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10006,
          "source": {
            "id": 10501,
            "title": "GPT-6 Astra System Card — September 22 revision",
            "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-09-03",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 11.4.1, Table 29; HealthBench Hard length-adjusted row, GPT-6 Luna column; value 31.4 (25.4, 1241)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "0.301",
          "scoreNumeric": 0.301,
          "date": "2026-09",
          "config": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 22.1%, mean answer 974 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "modelSlug": "gpt-6-sol",
          "scorePrinted": "30.1",
          "variant": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 22.1%, mean answer 974 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10005,
          "source": {
            "id": 10501,
            "title": "GPT-6 Astra System Card — September 22 revision",
            "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-09-03",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 11.4.1, Table 29; HealthBench Hard length-adjusted row, GPT-6 Sol column; value 30.1 (22.1, 974)"
          },
          "corroborating": []
        },
        {
          "model": "GPT OSS 120B",
          "lab": "OpenAI",
          "scoreDisplay": "0.300",
          "scoreNumeric": 0.3,
          "date": "2025-08",
          "config": "raw score (%), reasoning level high; gpt-oss model card Table 3. Not length-adjusted, unlike the GPT-5.x rows on this board.",
          "modelSlug": "gpt-oss-120b",
          "scorePrinted": "30.0",
          "variant": "raw score (%), reasoning level high; gpt-oss model card Table 3. Not length-adjusted, unlike the GPT-5.x rows on this board.",
          "measured": "2025-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 199,
          "source": {
            "id": 63,
            "title": "gpt-oss-120b & gpt-oss-20b Model Card",
            "url": "https://deploymentsafety.openai.com/gpt-oss/healthbench",
            "kind": "model_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2025-08-05",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 2, Table 3: Evaluations across multiple benchmarks and reasoning levels, row HealthBench Hard, column gpt-oss-120b high"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "0.291",
          "scoreNumeric": 0.291,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.4 (30.3 unadjusted, 2161 chars)",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "29.1",
          "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.4 (30.3 unadjusted, 2161 chars)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 483,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.4"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "29.1",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.4"
            }
          ]
        },
        {
          "model": "GPT-5.6 Luna (August)",
          "lab": "OpenAI",
          "scoreDisplay": "0.287",
          "scoreNumeric": 0.287,
          "date": "2026-08",
          "config": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Luna (August) (24.9 unadjusted, 1,523 chars)",
          "alias": "GPT-5.6 Luna",
          "modelSlug": "gpt-5.6-luna",
          "scorePrinted": "28.7",
          "variant": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Luna (August) (24.9 unadjusted, 1,523 chars)",
          "measured": "2026-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 491,
          "source": {
            "id": 30,
            "title": "GPT-5.6 - August Updates (system card addendum)",
            "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-08-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Luna (August)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.3 Chat",
          "lab": "OpenAI",
          "scoreDisplay": "0.259",
          "scoreNumeric": 0.259,
          "date": "2026-03",
          "config": "raw score (no length adjustment), GPT-5.3 Instant system card Table 3, column GPT-5.3-INSTANT; OpenAI later cards print 20.2 length-adjusted (17.8 unadjusted) for the re-run model.",
          "modelSlug": "gpt-5.3-chat",
          "scorePrinted": "25.9%",
          "variant": "raw score (no length adjustment), GPT-5.3 Instant system card Table 3, column GPT-5.3-INSTANT; OpenAI later cards print 20.2 length-adjusted (17.8 unadjusted) for the re-run model.",
          "measured": "2026-03",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 200,
          "source": {
            "id": 54,
            "title": "GPT-5.3 Instant System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-3-instant/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-03-02",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 4.1 HealthBench, Table 3: HealthBench, row Hard, column GPT-5.3-INSTANT"
          },
          "corroborating": [
            {
              "id": 51,
              "title": "GPT-5.5 Instant System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "conflicting",
              "scoreDisplay": "20.2",
              "quote": null,
              "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.3 INSTANT"
            },
            {
              "id": 30,
              "title": "GPT-5.6 - August Updates (system card addendum)",
              "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "conflicting",
              "scoreDisplay": "20.2",
              "quote": null,
              "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.3 Instant"
            }
          ]
        },
        {
          "model": "GPT-5.1",
          "lab": "OpenAI",
          "scoreDisplay": "0.254",
          "scoreNumeric": 0.254,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.1 (41.4 unadjusted, 4049 chars)",
          "modelSlug": "gpt-5.1",
          "scorePrinted": "25.4",
          "variant": "length-adjusted, max reasoning effort, GPT-5.6 system card Table 6 column GPT-5.1 (41.4 unadjusted, 4049 chars)",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 481,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.1"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "25.4",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.1"
            }
          ]
        },
        {
          "model": "GPT-5.5 Instant",
          "lab": "OpenAI",
          "scoreDisplay": "0.229",
          "scoreNumeric": 0.229,
          "date": "2026-05",
          "config": "length-adjusted (21.3 unadjusted, 1,794 mean response chars); GPT-5.5 Instant system card Table 5, column GPT-5.5 INSTANT.",
          "modelSlug": "gpt-5.5-instant",
          "scorePrinted": "22.9",
          "variant": "length-adjusted (21.3 unadjusted, 1,794 mean response chars); GPT-5.5 Instant system card Table 5, column GPT-5.5 INSTANT.",
          "measured": "2026-05",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 201,
          "source": {
            "id": 51,
            "title": "GPT-5.5 Instant System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-05-05",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5 INSTANT"
          },
          "corroborating": [
            {
              "id": 30,
              "title": "GPT-5.6 - August Updates (system card addendum)",
              "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "22.9",
              "quote": null,
              "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.5 Instant"
            }
          ]
        },
        {
          "model": "GPT OSS 20B",
          "lab": "OpenAI",
          "scoreDisplay": "0.108",
          "scoreNumeric": 0.108,
          "date": "2025-08",
          "config": "raw score (%), reasoning level high; gpt-oss model card Table 3. Not length-adjusted, unlike the GPT-5.x rows on this board.",
          "modelSlug": "gpt-oss-20b",
          "scorePrinted": "10.8",
          "variant": "raw score (%), reasoning level high; gpt-oss model card Table 3. Not length-adjusted, unlike the GPT-5.x rows on this board.",
          "measured": "2025-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 202,
          "source": {
            "id": 63,
            "title": "gpt-oss-120b & gpt-oss-20b Model Card",
            "url": "https://deploymentsafety.openai.com/gpt-oss/healthbench",
            "kind": "model_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2025-08-05",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 2, Table 3: Evaluations across multiple benchmarks and reasoning levels, row HealthBench Hard, column gpt-oss-20b high"
          },
          "corroborating": []
        }
      ],
      "unitShort": "1,000 conversations",
      "publisherShort": "OpenAI",
      "sourceShort": "healthbenchhard.ai",
      "fieldSize": 20,
      "category": "rubric",
      "confidence": "verified",
      "officialUrl": null,
      "paperUrl": "https://arxiv.org/abs/2505.08775",
      "scaleKind": "fraction",
      "summary": "The 1,000 HealthBench conversations frontier models failed most, still graded on the original physician rubrics.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "Checked primary model reports, applied September 22 Astra correction, added Sol/Luna and Baichuan-report results. Historical raw results and current adjusted results are explicitly labeled and are not a matched comparison.",
        "urls": [
          "https://healthbenchhard.ai",
          "https://arxiv.org/pdf/2602.06570",
          "https://ai.meta.com/blog/introducing-muse-spark-msl/",
          "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
          "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
          "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
          "https://deploymentsafety.openai.com/gpt-oss/healthbench",
          "https://deploymentsafety.openai.com/gpt-5-3-instant/healthbench",
          "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench"
        ]
      }
    },
    {
      "slug": "healthbench",
      "name": "HealthBench",
      "publisher": "OpenAI",
      "released": "2025-05",
      "unit": "5,000 conversations",
      "scale": "0-100 rubric-point percentage (some sites display 0-1), higher better; length-adjusted and unadjusted variants",
      "description": "5,000 realistic multi-turn health conversations graded against physician-written rubrics (48,562 criteria) covering accuracy, completeness, context awareness, communication, and instruction following. OpenAI now also reports a length-adjusted variant that penalizes verbosity.",
      "basis": "mixed",
      "sourceName": "Published model evaluation reports",
      "sourceUrl": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
      "ours": null,
      "lastUpdate": "2026-09",
      "notes": "Rows mix documented length-adjusted and raw scores and are not a controlled cross-model comparison. Raw Baichuan, gpt-oss and GPT-5.3 launch results remain labeled in their row settings. New Claude rows use Anthropic’s Opus 4.8 grader; OpenAI reports its own evaluations. The September 22 Astra correction is applied. Use HealthBench Professional for a newer clinician task set.",
      "results": [
        {
          "model": "Claude Sonnet 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "65.4",
          "scoreNumeric": 65.4,
          "date": "2026-09",
          "config": "Anthropic evaluation; length-adjusted; adaptive thinking at max effort; Claude Opus 4.8 grader; no tools or custom system prompt; safety classifiers enabled. Raw score 69.4%. The paper reports raw and adjusted scores separately; the max-effort HealthBench chart label is 65.4%.",
          "modelSlug": "claude-sonnet-5.5",
          "scorePrinted": "65.4",
          "variant": "Anthropic evaluation; length-adjusted; adaptive thinking at max effort; Claude Opus 4.8 grader; no tools or custom system prompt; safety classifiers enabled. Raw score 69.4%. The paper reports raw and adjusted scores separately; the max-effort HealthBench chart label is 65.4%.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10010,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 8.15.1; pp. 137–139, Figure 8.15.B (max effort)"
          },
          "corroborating": []
        },
        {
          "model": "Baichuan-M3",
          "lab": "Baichuan",
          "scoreDisplay": "65.1",
          "scoreNumeric": 65.1,
          "date": "2026-02",
          "config": "self-run in Baichuan-M3 paper (arXiv 2602.06570)",
          "modelSlug": "baichuan-m3",
          "scorePrinted": "65.1",
          "variant": "self-run in Baichuan-M3 paper (arXiv 2602.06570)",
          "measured": "2026-02",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 99,
          "source": {
            "id": 10524,
            "title": "Baichuan-M3 Technical Report",
            "url": "https://arxiv.org/pdf/2602.06570",
            "kind": "paper",
            "publisher": "Baichuan",
            "firstParty": true,
            "publishedAt": "2026-02-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 23, section 4.2.1 HealthBench-Main; Table on p. 25 (Model / HealthBench Score)"
          },
          "corroborating": [
            {
              "id": 10524,
              "title": "Baichuan-M3 Technical Report",
              "url": "https://arxiv.org/pdf/2602.06570",
              "kind": "paper",
              "publisher": "Baichuan",
              "firstParty": false,
              "role": "corroborating",
              "scoreDisplay": "65.1",
              "quote": null,
              "locator": "p. 23, section 4.2.1 HealthBench-Main (Figure 7); also Table 2, p. 25 (Baichuan-M3-235B, HealthBench Score 65.1)"
            }
          ]
        },
        {
          "model": "GPT-5.2-High",
          "lab": "OpenAI",
          "scoreDisplay": "63.3",
          "scoreNumeric": 63.3,
          "date": "2026-02",
          "config": "raw score as run by Baichuan in the M3 technical report, not an OpenAI-reported number; OpenAI own GPT-5.2 figure is 56.8 length-adjusted (60.7 unadjusted) in the GPT-5.6 system card.",
          "alias": "GPT-5.2",
          "modelSlug": "gpt-5.2",
          "scorePrinted": "63.3",
          "variant": "raw score as run by Baichuan in the M3 technical report, not an OpenAI-reported number; OpenAI own GPT-5.2 figure is 56.8 length-adjusted (60.7 unadjusted) in the GPT-5.6 system card.",
          "measured": "2026-02",
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 100,
          "source": {
            "id": 10524,
            "title": "Baichuan-M3 Technical Report",
            "url": "https://arxiv.org/pdf/2602.06570",
            "kind": "paper",
            "publisher": "Baichuan",
            "firstParty": true,
            "publishedAt": "2026-02-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 25, HealthBench-Hallu table (Model / HealthBench Score column); also p. 23 prose"
          },
          "corroborating": [
            {
              "id": 10524,
              "title": "Baichuan-M3 Technical Report",
              "url": "https://arxiv.org/pdf/2602.06570",
              "kind": "paper",
              "publisher": "Baichuan",
              "firstParty": false,
              "role": "corroborating",
              "scoreDisplay": "63.3",
              "quote": null,
              "locator": "p. 23, section 4.2.1 HealthBench-Main (Figure 7); also Table 2, p. 25 (GPT-5.2-High, HealthBench Score 63.3)"
            },
            {
              "id": 50,
              "title": "GPT-5.6 System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "conflicting",
              "scoreDisplay": "56.8",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.2"
            }
          ]
        },
        {
          "model": "Claude Opus 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "60.6",
          "scoreNumeric": 60.6,
          "date": "2026-09",
          "config": "Anthropic evaluation; length-adjusted; adaptive thinking at max effort; Claude Opus 4.8 grader; no tools or custom system prompt; safety classifiers enabled. Raw score 68.1%. Five-trial average; refusal fallback to Claude Opus 5.",
          "modelSlug": "claude-opus-5.5",
          "scorePrinted": "60.6",
          "variant": "Anthropic evaluation; length-adjusted; adaptive thinking at max effort; Claude Opus 4.8 grader; no tools or custom system prompt; safety classifiers enabled. Raw score 68.1%. Five-trial average; refusal fallback to Claude Opus 5.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10008,
          "source": {
            "id": 10510,
            "title": "Claude Opus 5.5 System Card",
            "url": "https://www.anthropic.com/claude-opus-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 8.15.1; pp. 213–214, Figures 8.15.1.A and 8.15.2.A"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "60.4",
          "scoreNumeric": 60.4,
          "date": "2026-09",
          "config": "Anthropic evaluation of Claude Fable 5, length-adjusted; adaptive max effort, Opus 4.8 grader, five trials, no tools or custom system prompt; raw 61.2%. Explicit Fable 5 figure in the September 1 card; earlier catalog value came from a Mythos 5 column and is not used for Fable 5.",
          "modelSlug": "claude-fable-5",
          "scorePrinted": "60.4",
          "variant": "Anthropic evaluation of Claude Fable 5, length-adjusted; adaptive max effort, Opus 4.8 grader, five trials, no tools or custom system prompt; raw 61.2%. Explicit Fable 5 figure in the September 1 card; earlier catalog value came from a Mythos 5 column and is not used for Fable 5.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 500,
          "source": {
            "id": 10515,
            "title": "Claude Fable 5.1 and Claude Mythos 5.1 System Card",
            "url": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-01",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 198, Figure 8.17.1.A; Claude Fable 5, adjusted bar 60.4%"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5.1",
          "lab": "Anthropic",
          "scoreDisplay": "60%",
          "scoreNumeric": 60,
          "date": "2026-09",
          "config": "length-adjusted (method published in OpenAI's GPT-5.5 System Card); Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over five trials, no tools or customized system prompt; Fable 5.1 run with safety classifiers active and a refusal-fallback to Claude Opus 5 (raw 66.7%).",
          "modelSlug": "claude-fable-5.1",
          "scorePrinted": "60%",
          "variant": "length-adjusted (method published in OpenAI's GPT-5.5 System Card); Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over five trials, no tools or customized system prompt; Fable 5.1 run with safety classifiers active and a refusal-fallback to Claude Opus 5 (raw 66.7%).",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 562,
          "source": {
            "id": 10515,
            "title": "Claude Fable 5.1 and Claude Mythos 5.1 System Card",
            "url": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-01",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 198, sec. 8.17.1 / Figure 8.17.1.A"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.8",
          "lab": "Anthropic",
          "scoreDisplay": "59.3",
          "scoreNumeric": 59.3,
          "date": "2026-06",
          "config": "length-adjusted; Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt; comparison column in the Fable/Mythos 5 card; raw 58.8% per Opus 5 card; not in the Opus 4.8 card itself",
          "modelSlug": "claude-opus-4.8",
          "scorePrinted": "59.3",
          "variant": "length-adjusted; Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt; comparison column in the Fable/Mythos 5 card; raw 58.8% per Opus 5 card; not in the Opus 4.8 card itself",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 502,
          "source": {
            "id": 61,
            "title": "Claude Fable 5 and Claude Mythos 5 System Card",
            "url": "https://anthropic.com/claude-fable-5-mythos-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-06-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 252, Table 8.1.A, row HealthBench, column Opus 4.8"
          },
          "corroborating": [
            {
              "id": 98,
              "title": "System Card: Claude Opus 5",
              "url": "https://www.anthropic.com/claude-opus-5-system-card",
              "kind": "system_card",
              "publisher": "Anthropic",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "59.3%",
              "quote": null,
              "locator": "p. 188, section 8.15.1, Figure 8.15.1.A"
            }
          ]
        },
        {
          "model": "Claude Sonnet 5",
          "lab": "Anthropic",
          "scoreDisplay": "58.7%",
          "scoreNumeric": 58.7,
          "date": "2026-06",
          "config": "length-adjusted; Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt; figure-only in the Sonnet 5 card; raw 59.2% per Opus 5 card",
          "modelSlug": "claude-sonnet-5",
          "scorePrinted": "58.7%",
          "variant": "length-adjusted; Anthropic protocol: adaptive thinking at max effort, Claude Opus 4.8 grader, averaged over 5 trials, no tools or custom system prompt; figure-only in the Sonnet 5 card; raw 59.2% per Opus 5 card",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 501,
          "source": {
            "id": 99,
            "title": "System Card: Claude Sonnet 5",
            "url": "https://www.anthropic.com/claude-sonnet-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-06-30",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 138, section 8.12.1 HealthBench results, Figure 8.12.1.A bar label (no prose or table number)"
          },
          "corroborating": [
            {
              "id": 98,
              "title": "System Card: Claude Opus 5",
              "url": "https://www.anthropic.com/claude-opus-5-system-card",
              "kind": "system_card",
              "publisher": "Anthropic",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "58.7%",
              "quote": null,
              "locator": "p. 188, section 8.15.1, Figure 8.15.1.A"
            }
          ]
        },
        {
          "model": "GPT-6 Astra",
          "lab": "OpenAI",
          "scoreDisplay": "58.3",
          "scoreNumeric": 58.3,
          "date": "2026-09",
          "config": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 56.9%, mean answer 1,760 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "modelSlug": "gpt-6-astra",
          "scorePrinted": "58.3",
          "variant": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 56.9%, mean answer 1,760 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 564,
          "source": {
            "id": 10501,
            "title": "GPT-6 Astra System Card — September 22 revision",
            "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-09-03",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 11.4.1, Table 29; HealthBench length-adjusted row, GPT-6 Astra column; value 58.3 (56.9, 1760)"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "57.8",
          "scoreNumeric": 57.8,
          "date": "2026-07",
          "config": "Anthropic evaluation; length-adjusted 57.8%, raw 67.1%; adaptive max effort, Opus 4.8 grader, five-trial average, no tools or customized system prompt. Previously this catalog displayed the raw value.",
          "modelSlug": "claude-opus-5",
          "scorePrinted": "57.8",
          "variant": "Anthropic evaluation; length-adjusted 57.8%, raw 67.1%; adaptive max effort, Opus 4.8 grader, five-trial average, no tools or customized system prompt. Previously this catalog displayed the raw value.",
          "measured": "2026-07",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 98,
          "source": {
            "id": 98,
            "title": "System Card: Claude Opus 5",
            "url": "https://www.anthropic.com/claude-opus-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-07-24",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 188, section 8.15.1 and Figure 8.15.1.A; adjusted 57.8%, raw 67.1%"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5",
          "lab": "OpenAI",
          "scoreDisplay": "57.7",
          "scoreNumeric": 57.7,
          "date": "2026-06",
          "config": "OpenAI length-adjusted score, maximum reasoning effort; raw 63.1%, mean answer 2,904 characters; GPT-5.6 card Table 6.",
          "modelSlug": "gpt-5",
          "scorePrinted": "57.7",
          "variant": "OpenAI length-adjusted score, maximum reasoning effort; raw 63.1%, mean answer 2,904 characters; GPT-5.6 card Table 6.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10014,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1, Table 6, HealthBench length-adjusted row, GPT-5 column"
          },
          "corroborating": []
        },
        {
          "model": "GPT OSS 120B",
          "lab": "OpenAI",
          "scoreDisplay": "57.6",
          "scoreNumeric": 57.6,
          "date": "2025-08",
          "config": "reasoning level high, raw score (%), gpt-oss model card Table 3 (low 53.0, medium 55.9)",
          "modelSlug": "gpt-oss-120b",
          "scorePrinted": "57.6",
          "variant": "reasoning level high, raw score (%), gpt-oss model card Table 3 (low 53.0, medium 55.9)",
          "measured": "2025-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 497,
          "source": {
            "id": 63,
            "title": "gpt-oss-120b & gpt-oss-20b Model Card",
            "url": "https://deploymentsafety.openai.com/gpt-oss/healthbench",
            "kind": "model_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2025-08-05",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 2, Table 3: Evaluations across multiple benchmarks and reasoning levels, row HealthBench, column gpt-oss-120b high"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "57.0",
          "scoreNumeric": 57,
          "date": "2026-06",
          "config": "length-adjusted, max reasoning effort (55.6 unadjusted), GPT-5.6 system card 2026-07-09",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "57.0",
          "variant": "length-adjusted, max reasoning effort (55.6 unadjusted), GPT-5.6 system card 2026-07-09",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 101,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-SOL"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "57.0",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-SOL"
            }
          ]
        },
        {
          "model": "GPT-5.6 Terra",
          "lab": "OpenAI",
          "scoreDisplay": "57.0",
          "scoreNumeric": 57,
          "date": "2026-06",
          "config": "length-adjusted (58.7 unadjusted), max reasoning effort",
          "modelSlug": "gpt-5.6-terra",
          "scorePrinted": "57.0",
          "variant": "length-adjusted (58.7 unadjusted), max reasoning effort",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 102,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-TERRA"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "57.0",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-TERRA"
            }
          ]
        },
        {
          "model": "GPT-5.2",
          "lab": "OpenAI",
          "scoreDisplay": "56.8",
          "scoreNumeric": 56.8,
          "date": "2026-06",
          "config": "OpenAI length-adjusted score, maximum reasoning effort; raw 60.7%, mean answer 2,645 characters; GPT-5.6 card Table 6.",
          "modelSlug": "gpt-5.2",
          "scorePrinted": "56.8",
          "variant": "OpenAI length-adjusted score, maximum reasoning effort; raw 60.7%, mean answer 2,645 characters; GPT-5.6 card Table 6.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10016,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1, Table 6, HealthBench length-adjusted row, GPT-5.2 column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.5",
          "lab": "OpenAI",
          "scoreDisplay": "56.5",
          "scoreNumeric": 56.5,
          "date": "2026-04",
          "config": "length-adjusted (58.4 unadjusted), comparison row in GPT-5.6 system card",
          "modelSlug": "gpt-5.5",
          "scorePrinted": "56.5",
          "variant": "length-adjusted (58.4 unadjusted), comparison row in GPT-5.6 system card",
          "measured": "2026-04",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 103,
          "source": {
            "id": 55,
            "title": "GPT-5.5 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-5/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-04-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5 Health, Table 7 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "56.5",
              "quote": null,
              "locator": null
            },
            {
              "id": 50,
              "title": "GPT-5.6 System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "56.5",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5"
            }
          ]
        },
        {
          "model": "GPT-5.6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "55.8",
          "scoreNumeric": 55.8,
          "date": "2026-06",
          "config": "length-adjusted (55.4 unadjusted), max reasoning effort",
          "modelSlug": "gpt-5.6-luna",
          "scorePrinted": "55.8",
          "variant": "length-adjusted (55.4 unadjusted), max reasoning effort",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 104,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-LUNA"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "55.8",
              "quote": null,
              "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-LUNA"
            }
          ]
        },
        {
          "model": "GPT-5.6 Sol (August)",
          "lab": "OpenAI",
          "scoreDisplay": "55.0",
          "scoreNumeric": 55,
          "date": "2026-08",
          "config": "ChatGPT production/Instant deployment setting, length-adjusted (52.1 unadjusted), GPT-5.6 August Updates PDF 2026-08-06",
          "alias": "GPT-5.6 Sol",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "55.0",
          "variant": "ChatGPT production/Instant deployment setting, length-adjusted (52.1 unadjusted), GPT-5.6 August Updates PDF 2026-08-06",
          "measured": "2026-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 105,
          "source": {
            "id": 30,
            "title": "GPT-5.6 - August Updates (system card addendum)",
            "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-08-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Sol (August)"
          },
          "corroborating": [
            {
              "id": 16,
              "title": "GPT-5.6 Preview System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "55.0",
              "quote": null,
              "locator": null
            }
          ]
        },
        {
          "model": "GPT-6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "54.5",
          "scoreNumeric": 54.5,
          "date": "2026-09",
          "config": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 50%, mean answer 1,255 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "modelSlug": "gpt-6-luna",
          "scorePrinted": "54.5",
          "variant": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 50%, mean answer 1,255 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10004,
          "source": {
            "id": 10501,
            "title": "GPT-6 Astra System Card — September 22 revision",
            "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-09-03",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 11.4.1, Table 29; HealthBench length-adjusted row, GPT-6 Luna column; value 54.5 (50, 1255)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.3 Chat",
          "lab": "OpenAI",
          "scoreDisplay": "54.1%",
          "scoreNumeric": 54.1,
          "date": "2026-03",
          "config": "raw score (no length adjustment), GPT-5.3 Instant system card Table 3 column GPT-5.3-INSTANT; later OpenAI cards print 49.6 length-adjusted (47.9 unadjusted)",
          "modelSlug": "gpt-5.3-chat",
          "scorePrinted": "54.1%",
          "variant": "raw score (no length adjustment), GPT-5.3 Instant system card Table 3 column GPT-5.3-INSTANT; later OpenAI cards print 49.6 length-adjusted (47.9 unadjusted)",
          "measured": "2026-03",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 496,
          "source": {
            "id": 54,
            "title": "GPT-5.3 Instant System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-3-instant/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-03-02",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 4.1 HealthBench, Table 3: HealthBench, row HealthBench, column GPT-5.3-INSTANT"
          },
          "corroborating": [
            {
              "id": 51,
              "title": "GPT-5.5 Instant System Card",
              "url": "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "conflicting",
              "scoreDisplay": "49.6",
              "quote": null,
              "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.3 INSTANT"
            }
          ]
        },
        {
          "model": "GPT-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "54.0",
          "scoreNumeric": 54,
          "date": "2026-06",
          "config": "OpenAI length-adjusted score, maximum reasoning effort; raw 55.7%, mean answer 2,275 characters; GPT-5.6 card Table 6.",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "54",
          "variant": "OpenAI length-adjusted score, maximum reasoning effort; raw 55.7%, mean answer 2,275 characters; GPT-5.6 card Table 6.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10017,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1, Table 6, HealthBench length-adjusted row, GPT-5.4 column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Luna (August)",
          "lab": "OpenAI",
          "scoreDisplay": "53.3",
          "scoreNumeric": 53.3,
          "date": "2026-08",
          "config": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Luna (August) (50.7 unadjusted, 1,567 chars)",
          "alias": "GPT-5.6 Luna",
          "modelSlug": "gpt-5.6-luna",
          "scorePrinted": "53.3",
          "variant": "ChatGPT production (Instant) deployment setting, length-adjusted, GPT-5.6 August Updates PDF column GPT-5.6 Luna (August) (50.7 unadjusted, 1,567 chars)",
          "measured": "2026-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 494,
          "source": {
            "id": 30,
            "title": "GPT-5.6 - August Updates (system card addendum)",
            "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-08-06",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Luna (August)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "53.2",
          "scoreNumeric": 53.2,
          "date": "2026-09",
          "config": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 47.1%, mean answer 977 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "modelSlug": "gpt-6-sol",
          "scorePrinted": "53.2",
          "variant": "OpenAI evaluation; length-adjusted at maximum reasoning effort; raw 47.1%, mean answer 977 characters. September 22, 2026 system-card revision; the Astra values correct a prior evaluation misconfiguration.",
          "measured": "2026-09",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10003,
          "source": {
            "id": 10501,
            "title": "GPT-6 Astra System Card — September 22 revision",
            "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-09-03",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 11.4.1, Table 29; HealthBench length-adjusted row, GPT-6 Sol column; value 53.2 (47.1, 977)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.5 Instant",
          "lab": "OpenAI",
          "scoreDisplay": "51.4",
          "scoreNumeric": 51.4,
          "date": "2026-05",
          "config": "length-adjusted, GPT-5.5 Instant system card Table 5 column GPT-5.5 INSTANT (50.9 unadjusted, 1,922 chars); same number in GPT-5.6 August Updates p. 11",
          "modelSlug": "gpt-5.5-instant",
          "scorePrinted": "51.4",
          "variant": "length-adjusted, GPT-5.5 Instant system card Table 5 column GPT-5.5 INSTANT (50.9 unadjusted, 1,922 chars); same number in GPT-5.6 August Updates p. 11",
          "measured": "2026-05",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 495,
          "source": {
            "id": 51,
            "title": "GPT-5.5 Instant System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-05-05",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5 INSTANT"
          },
          "corroborating": [
            {
              "id": 30,
              "title": "GPT-5.6 - August Updates (system card addendum)",
              "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
              "kind": "system_card",
              "publisher": "OpenAI",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "51.4",
              "quote": null,
              "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.5 Instant"
            }
          ]
        },
        {
          "model": "GPT-5.1",
          "lab": "OpenAI",
          "scoreDisplay": "50.9",
          "scoreNumeric": 50.9,
          "date": "2026-06",
          "config": "OpenAI length-adjusted score, maximum reasoning effort; raw 64.2%, mean answer 4,222 characters; GPT-5.6 card Table 6.",
          "modelSlug": "gpt-5.1",
          "scorePrinted": "50.9",
          "variant": "OpenAI length-adjusted score, maximum reasoning effort; raw 64.2%, mean answer 4,222 characters; GPT-5.6 card Table 6.",
          "measured": "2026-06",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 10015,
          "source": {
            "id": 50,
            "title": "GPT-5.6 System Card",
            "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
            "kind": "system_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2026-07-09",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 5.1, Table 6, HealthBench length-adjusted row, GPT-5.1 column"
          },
          "corroborating": []
        },
        {
          "model": "GPT OSS 20B",
          "lab": "OpenAI",
          "scoreDisplay": "42.5",
          "scoreNumeric": 42.5,
          "date": "2025-08",
          "config": "reasoning level high, raw score (%), gpt-oss model card Table 3 (low 40.4, medium 41.8)",
          "modelSlug": "gpt-oss-20b",
          "scorePrinted": "42.5",
          "variant": "reasoning level high, raw score (%), gpt-oss model card Table 3 (low 40.4, medium 41.8)",
          "measured": "2025-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 498,
          "source": {
            "id": 63,
            "title": "gpt-oss-120b & gpt-oss-20b Model Card",
            "url": "https://deploymentsafety.openai.com/gpt-oss/healthbench",
            "kind": "model_card",
            "publisher": "OpenAI",
            "firstParty": true,
            "publishedAt": "2025-08-05",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Section 2, Table 3: Evaluations across multiple benchmarks and reasoning levels, row HealthBench, column gpt-oss-20b high"
          },
          "corroborating": []
        }
      ],
      "unitShort": "5,000 conversations",
      "publisherShort": "OpenAI",
      "sourceShort": "OpenAI Deployment Safety Hub",
      "fieldSize": 26,
      "category": "rubric",
      "confidence": "verified",
      "officialUrl": "https://openai.com/index/healthbench/",
      "paperUrl": null,
      "scaleKind": "percent",
      "summary": "Realistic multi-turn health conversations graded on physician-written rubrics for accuracy, completeness, and communication.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "Updated through September 28 Sonnet 5.5 and September 22 OpenAI correction. Fixed Fable/Mythos alias and Opus 5 raw/adjusted display. Raw older results remain labeled; evaluator protocols differ.",
        "urls": [
          "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
          "https://www.anthropic.com/claude-sonnet-5-5-system-card",
          "https://arxiv.org/pdf/2602.06570",
          "https://www.anthropic.com/claude-opus-5-5-system-card",
          "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
          "https://anthropic.com/claude-fable-5-mythos-5-system-card",
          "https://www.anthropic.com/claude-sonnet-5-system-card",
          "https://www.anthropic.com/claude-opus-5-system-card",
          "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
          "https://deploymentsafety.openai.com/gpt-oss/healthbench",
          "https://deploymentsafety.openai.com/gpt-5-5/healthbench",
          "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
          "https://deploymentsafety.openai.com/gpt-5-3-instant/healthbench",
          "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench"
        ]
      }
    },
    {
      "slug": "health-optimization-bench",
      "name": "Health Optimization Bench",
      "publisher": "Arcophos",
      "released": "2026-08",
      "unit": "257 tasks across eight subject suites",
      "scale": "0-100 rubric credit, higher better",
      "description": "Questions in eight areas of preventive and optimization medicine, grounded in primary evidence and scored against task-specific rubrics. The current main ranking covers 257 released tasks, separate from the 89-task incretin therapeutics evidence suite.",
      "basis": "independent-run",
      "sourceName": "healthoptimizationbench.com",
      "sourceUrl": "https://healthoptimizationbench.com/sources",
      "ours": "https://healthoptimizationbench.com",
      "lastUpdate": "2026-09",
      "notes": "The main ranking moved from the 89-task incretin suite to 257 subject-suite tasks. This page shows only the 257-task September 10 snapshot; the two task sets must not be pooled. Three grader families vote, with split decisions escalated to a fourth, and the authoring family excluded. Fable 5.1 safeguards declined 97 tasks; those receive no credit.",
      "results": [
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "70.9",
          "scoreNumeric": 70.9,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 68.0–73.8. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "claude-fable-5",
          "scorePrinted": "70.9",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 68.0–73.8. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10020,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Claude Fable 5; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.709, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "69.3",
          "scoreNumeric": 69.3,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 66.2–72.4. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "claude-opus-5",
          "scorePrinted": "69.3",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 66.2–72.4. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10021,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Claude Opus 5; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.693, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.6",
          "lab": "xAI",
          "scoreDisplay": "66.8",
          "scoreNumeric": 66.8,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 63.8–69.8. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "grok-4.6",
          "scorePrinted": "66.8",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 63.8–69.8. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10022,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Grok 4.6; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.668, n=257"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Sol (max)",
          "lab": "OpenAI",
          "scoreDisplay": "66.6",
          "scoreNumeric": 66.6,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 63.4–69.6. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking. Maximum reasoning effort.",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "66.6",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 63.4–69.6. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking. Maximum reasoning effort.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10023,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; GPT-5.6 Sol (max); tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.666, n=257"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Sol (high)",
          "lab": "OpenAI",
          "scoreDisplay": "64.6",
          "scoreNumeric": 64.6,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 61.5–67.6. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking. High reasoning effort.",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "64.6",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 61.5–67.6. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking. High reasoning effort.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10024,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; GPT-5.6 Sol (high); tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.646, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K3",
          "lab": "Moonshot AI",
          "scoreDisplay": "59.9",
          "scoreNumeric": 59.9,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 56.5–63.3. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "kimi-k3",
          "scorePrinted": "59.9",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 56.5–63.3. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10025,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Kimi K3; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.599, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark",
          "lab": "Meta",
          "scoreDisplay": "57.2",
          "scoreNumeric": 57.2,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 53.9–60.5. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "muse-spark",
          "scorePrinted": "57.2",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 53.9–60.5. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10026,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Muse Spark; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.572, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5.1",
          "lab": "Anthropic",
          "scoreDisplay": "47.3",
          "scoreNumeric": 47.3,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 42.2–52.2. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking. Vendor safeguards declined 97/257 tasks, scored with no credit; mean over answered tasks is 75.9.",
          "modelSlug": "claude-fable-5.1",
          "scorePrinted": "47.3",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 42.2–52.2. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking. Vendor safeguards declined 97/257 tasks, scored with no credit; mean over answered tasks is 75.9.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10027,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Claude Fable 5.1; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.473, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.6",
          "lab": "Google",
          "scoreDisplay": "39.7",
          "scoreNumeric": 39.7,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 36.2–43.1. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "gemini-3.6",
          "scorePrinted": "39.7",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 36.2–43.1. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10028,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Gemini 3.6; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.397, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Inkling",
          "lab": "Thinking Machines",
          "scoreDisplay": "35.6",
          "scoreNumeric": 35.6,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 32.3–39.0. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "inkling",
          "scorePrinted": "35.6",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 32.3–39.0. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10029,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Inkling; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.356, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5",
          "lab": "Anthropic",
          "scoreDisplay": "34.6",
          "scoreNumeric": 34.6,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 31.6–37.8. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "claude-sonnet-5",
          "scorePrinted": "34.6",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 31.6–37.8. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10030,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Claude Sonnet 5; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.346, n=257"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.2",
          "lab": "Zhipu",
          "scoreDisplay": "20.7",
          "scoreNumeric": 20.7,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 18.1–23.4. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "glm-5.2",
          "scorePrinted": "20.7",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 18.1–23.4. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10031,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; GLM 5.2; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.207, n=257"
          },
          "corroborating": []
        },
        {
          "model": "MiniMax M3",
          "lab": "MiniMax",
          "scoreDisplay": "18.3",
          "scoreNumeric": 18.3,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 15.7–20.9. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "minimax-m3",
          "scorePrinted": "18.3",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 15.7–20.9. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10032,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; MiniMax M3; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.183, n=257"
          },
          "corroborating": []
        },
        {
          "model": "MAI Thinking",
          "lab": "Microsoft AI",
          "scoreDisplay": "17.5",
          "scoreNumeric": 17.5,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 15.1–20.0. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "mai-thinking-1",
          "scorePrinted": "17.5",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 15.1–20.0. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10033,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; MAI Thinking; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.175, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Mistral Medium 3.5",
          "lab": "Mistral",
          "scoreDisplay": "9.2",
          "scoreNumeric": 9.2,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 7.4–11.1. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "mistral-medium-3.5",
          "scorePrinted": "9.2",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 7.4–11.1. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10034,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Mistral Medium 3.5; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.092, n=257"
          },
          "corroborating": []
        },
        {
          "model": "Nemotron 3.5 Lightning",
          "lab": "NVIDIA",
          "scoreDisplay": "4.9",
          "scoreNumeric": 4.9,
          "date": "2026-09",
          "config": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 3.6–6.2. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "modelSlug": "nemotron-3.5-lightning",
          "scorePrinted": "4.9",
          "variant": "Subject suites release set: 257 tasks across eight subjects; one answer per task, no tools; blind cross-family grading. 95% bootstrap CI 3.6–6.2. Harness snapshot September 10, 2026; not the separate 89-task incretin ranking.",
          "measured": "2026-09",
          "reportedBy": "arcophos",
          "reportedByLabel": "Arcophos run",
          "confidence": "verified",
          "resultId": 10035,
          "source": {
            "id": 10526,
            "title": "Health Optimization Bench — subject suites results",
            "url": "https://healthoptimizationbench.com/sources",
            "kind": "arcophos_run",
            "publisher": "Arcophos",
            "firstParty": true,
            "publishedAt": "2026-09-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Subject suites table; Nemotron 3.5 Lightning; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.049, n=257"
          },
          "corroborating": []
        }
      ],
      "unitShort": "257 tasks",
      "publisherShort": "Arcophos",
      "sourceShort": "healthoptimizationbench.com",
      "fieldSize": 16,
      "category": "rubric",
      "confidence": "verified",
      "officialUrl": "https://healthoptimizationbench.com",
      "paperUrl": null,
      "scaleKind": "percent",
      "summary": "257 evidence-grounded tasks across eight preventive medicine subjects, graded blind by independent model families.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "Replaced the prior 89-task incretin ranking with the official main ranking for 257 subject-suite tasks, snapshot September 10. All sixteen scores and confidence intervals checked against the source table.",
        "urls": [
          "https://healthoptimizationbench.com/sources"
        ]
      }
    },
    {
      "slug": "mast",
      "name": "MAST (Medical AI Superintelligence Test)",
      "publisher": "ARISE AI Research Network (multi-institutional)",
      "released": "2026-08",
      "unit": "composite of 6 component benchmarks; 11 models",
      "scale": "percentage composite, higher better",
      "description": "Composite score across curated clinical benchmarks spanning diagnostic reasoning, management reasoning, safety, multimodal images, multimodal radiology, and agentic capability. Components: First Do NOHARM v2, SCT-Bench, MedAgentBench v2, PhysicianBench, ReXrank Mini, CPC-Bench.",
      "basis": "independent-run",
      "sourceName": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
      "sourceUrl": "https://arise-ai.org/mast",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "The board is marked as a preview and was last updated August 15, 2026; component-level breakdowns are published only for First, Do NOHARM v2. Scores may move before the full release. The official overview still states August 15, 2026. Eight displayed general-ranking rows are indexed here, from eleven models on that view. MAST is in preview and scores may change.",
      "results": [
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "60.2%",
          "scoreNumeric": 60.2,
          "date": "2026-08",
          "config": "MAST in preview; 'exact scores may change'",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "60.2%",
          "variant": "MAST in preview; 'exact scores may change'",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 106,
          "source": {
            "id": 17,
            "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
            "url": "https://arise-ai.org/mast",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K3",
          "lab": "Moonshot AI",
          "scoreDisplay": "60.1%",
          "scoreNumeric": 60.1,
          "date": "2026-08",
          "config": "",
          "modelSlug": "kimi-k3",
          "scorePrinted": "60.1%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 107,
          "source": {
            "id": 17,
            "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
            "url": "https://arise-ai.org/mast",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.6 Flash",
          "lab": "Google",
          "scoreDisplay": "59.3%",
          "scoreNumeric": 59.3,
          "date": "2026-08",
          "config": "",
          "modelSlug": "gemini-3.6-flash",
          "scorePrinted": "59.3%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 108,
          "source": {
            "id": 17,
            "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
            "url": "https://arise-ai.org/mast",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "58.9%",
          "scoreNumeric": 58.9,
          "date": "2026-08",
          "config": "",
          "modelSlug": "gemini-3.1-pro",
          "scorePrinted": "58.9%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 109,
          "source": {
            "id": 17,
            "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
            "url": "https://arise-ai.org/mast",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3.5 397B A17B",
          "lab": "Alibaba",
          "scoreDisplay": "57.9%",
          "scoreNumeric": 57.9,
          "date": "2026-08",
          "config": "",
          "modelSlug": "qwen3.5-397b-a17b",
          "scorePrinted": "57.9%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 110,
          "source": {
            "id": 17,
            "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
            "url": "https://arise-ai.org/mast",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "57.1%",
          "scoreNumeric": 57.1,
          "date": "2026-08",
          "config": "",
          "modelSlug": "claude-opus-5",
          "scorePrinted": "57.1%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 111,
          "source": {
            "id": 17,
            "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
            "url": "https://arise-ai.org/mast",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5",
          "lab": "Anthropic",
          "scoreDisplay": "56.6%",
          "scoreNumeric": 56.6,
          "date": "2026-08",
          "config": "",
          "modelSlug": "claude-sonnet-5",
          "scorePrinted": "56.6%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 112,
          "source": {
            "id": 17,
            "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
            "url": "https://arise-ai.org/mast",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.3",
          "lab": "xAI",
          "scoreDisplay": "53.7%",
          "scoreNumeric": 53.7,
          "date": "2026-08",
          "config": "",
          "modelSlug": "grok-4.3",
          "scorePrinted": "53.7%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 113,
          "source": {
            "id": 17,
            "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
            "url": "https://arise-ai.org/mast",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')"
          },
          "corroborating": []
        }
      ],
      "unitShort": "6-benchmark composite",
      "publisherShort": "ARISE AI Research Network",
      "sourceShort": "ARISE MAST leaderboard",
      "fieldSize": 11,
      "category": "composite",
      "confidence": "verified",
      "officialUrl": "https://arise-ai.org/mast",
      "paperUrl": null,
      "scaleKind": "percent",
      "summary": "A composite of clinical benchmarks spanning diagnostic and management reasoning, safety, multimodal imaging, and agentic capability.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "Official source still reports August 15, 2026; all eight displayed general-ranking scores match. This is a fresh source check of the existing snapshot, not a new model evaluation.",
        "urls": [
          "https://arise-ai.org/mast"
        ]
      }
    },
    {
      "slug": "medhelm",
      "name": "MedHELM",
      "publisher": "Stanford CRFM / HAI and multi-institution collaborators",
      "released": "2025-02",
      "unit": "121 tasks / 31 datasets",
      "scale": "mean win rate 0-1, higher better",
      "description": "Holistic evaluation of LLMs on 121 clinical tasks across 5 categories and 22 subcategories (31 datasets) in a clinician-validated taxonomy; ranked by mean win rate.",
      "basis": "official-leaderboard",
      "sourceName": "MedHELM leaderboard (medhelm.org), v5.0.0",
      "sourceUrl": "https://medhelm.org/",
      "ours": null,
      "lastUpdate": "2026-05",
      "notes": "Version 5.0.0, last updated May 14, 2026, run by the Stanford-led maintainers on a roughly quarterly cadence. No Claude 5 family or GPT-5.6 rows yet. Mean win rate is relative to the evaluated cohort, so scores shift whenever the model set changes. The current official home-page ranking is v5.0.0, updated May 14, 2026; ten of eleven models are displayed. September 28 is a source check, not a new evaluation.",
      "results": [
        {
          "model": "Gemini 3.1 Pro (Preview)",
          "lab": "Google",
          "scoreDisplay": "0.652",
          "scoreNumeric": 0.652,
          "date": "2026-05",
          "config": "",
          "alias": "Gemini 3.1 Pro",
          "modelSlug": "gemini-3.1-pro",
          "scorePrinted": "0.652",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 114,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.6520833333333333",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        },
        {
          "model": "Gemini 3.5 Flash",
          "lab": "Google",
          "scoreDisplay": "0.642",
          "scoreNumeric": 0.642,
          "date": "2026-05",
          "config": "",
          "modelSlug": "gemini-3.5-flash",
          "scorePrinted": "0.642",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 115,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.6416666666666667",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        },
        {
          "model": "Muse Spark (2026-04-08)",
          "lab": "Meta",
          "scoreDisplay": "0.621",
          "scoreNumeric": 0.621,
          "date": "2026-05",
          "config": "",
          "alias": "Muse Spark",
          "modelSlug": "muse-spark",
          "scorePrinted": "0.621",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 116,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.6208333333333333",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        },
        {
          "model": "GPT-5.4 mini",
          "lab": "OpenAI",
          "scoreDisplay": "0.552",
          "scoreNumeric": 0.552,
          "date": "2026-05",
          "config": "",
          "modelSlug": "gpt-5.4-mini",
          "scorePrinted": "0.552",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 117,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.5520833333333334",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        },
        {
          "model": "GPT-5.4 (2026-03-05)",
          "lab": "OpenAI",
          "scoreDisplay": "0.538",
          "scoreNumeric": 0.538,
          "date": "2026-05",
          "config": "",
          "alias": "GPT-5.4",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "0.538",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 118,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.5375",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        },
        {
          "model": "Gemini 2.5 Pro",
          "lab": "Google",
          "scoreDisplay": "0.529",
          "scoreNumeric": 0.529,
          "date": "2026-05",
          "config": "",
          "modelSlug": "gemini-2.5-pro",
          "scorePrinted": "0.529",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 119,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.5291666666666667",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        },
        {
          "model": "DeepSeek R1",
          "lab": "DeepSeek",
          "scoreDisplay": "0.485",
          "scoreNumeric": 0.485,
          "date": "2026-05",
          "config": "",
          "modelSlug": "deepseek-r1",
          "scorePrinted": "0.485",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 120,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.48541666666666666",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        },
        {
          "model": "Claude 4.6 Opus",
          "lab": "Anthropic",
          "scoreDisplay": "0.456",
          "scoreNumeric": 0.456,
          "date": "2026-05",
          "config": "",
          "alias": "Claude Opus 4.6",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "0.456",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 121,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.45625",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        },
        {
          "model": "Claude 3.7 Sonnet",
          "lab": "Anthropic",
          "scoreDisplay": "0.45",
          "scoreNumeric": 0.45,
          "date": "2026-05",
          "config": "",
          "modelSlug": "claude-3.7-sonnet",
          "scorePrinted": "0.45",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 122,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.45",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        },
        {
          "model": "Gemini 2.0 Flash",
          "lab": "Google",
          "scoreDisplay": "0.342",
          "scoreNumeric": 0.342,
          "date": "2026-05",
          "config": "",
          "modelSlug": "gemini-2.0-flash",
          "scorePrinted": "0.342",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 123,
          "source": {
            "id": 18,
            "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
            "url": "https://medhelm.org/",
            "kind": "official_leaderboard",
            "publisher": "Stanford CRFM (MedHELM)",
            "firstParty": true,
            "publishedAt": "2026-05-14",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')"
          },
          "corroborating": [
            {
              "id": 83,
              "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
              "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
              "kind": "official_leaderboard",
              "publisher": "Stanford CRFM (MedHELM)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.3416666666666667",
              "quote": null,
              "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'"
            }
          ]
        }
      ],
      "unitShort": "121 tasks",
      "publisherShort": "Stanford CRFM",
      "sourceShort": "MedHELM leaderboard",
      "fieldSize": 11,
      "category": "composite",
      "confidence": "verified",
      "officialUrl": "https://medhelm.org/",
      "paperUrl": null,
      "scaleKind": "fraction",
      "summary": "Holistic clinical evaluation across 121 tasks in a clinician-validated taxonomy, ranked by mean win rate.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "Official v5.0.0 home-page scores match all ten indexed models and still state May 14, 2026. No later result is implied by the September source check.",
        "urls": [
          "https://medhelm.org/"
        ]
      }
    },
    {
      "slug": "first-do-noharm",
      "name": "First, Do NOHARM (v2)",
      "publisher": "Stanford/Harvard-led consortium (50+ researchers incl. 29 board-certified physicians); hosted by ARISE",
      "released": "2025-12",
      "unit": "1,100 consultation cases, 10 specialties, 12,747 expert annotations on 4,249 management options",
      "scale": "percentage safety score, higher better",
      "description": "Frequency and severity of potentially harmful errors in LLM-generated medical consultation recommendations (Numerous Options Harm Assessment for Risk in Medicine); primary-care-to-specialist consults.",
      "basis": "official-leaderboard",
      "sourceName": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
      "sourceUrl": "https://arise-ai.org/mast/technical",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "Paper: arXiv 2512.01241. The v1 study found potential for severe harm in up to 24.6 percent of directly applied recommendations, with errors of omission behind more than 80 percent of the severe cases. Official technical leaderboard checked September 28, 2026; retained August dates on existing rows; newly indexed rows have no claimed measurement date. Includes labeled RAG clinical systems and base models, which have different tool access. The board is in preview. Seventeen of nineteen overall rows are indexed.",
      "results": [
        {
          "model": "LiSA 2.5",
          "lab": "AMBOSS",
          "scoreDisplay": "86.2",
          "scoreNumeric": 86.2,
          "date": "2026-09",
          "config": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. RAG clinical system, not a base-model-only result. Observed September 28, 2026; the individual run date is not published.",
          "modelSlug": null,
          "scorePrinted": "86.2",
          "variant": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. RAG clinical system, not a base-model-only result. Observed September 28, 2026; the individual run date is not published.",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark-owner run",
          "confidence": "verified",
          "resultId": 10036,
          "source": {
            "id": 10542,
            "title": "ARISE MAST technical leaderboard",
            "url": "https://www.arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "First Do NOHARM v2 overall leaderboard; LiSA 2.5; score 86.2%"
          },
          "corroborating": []
        },
        {
          "model": "Doximity Ask 6.1",
          "lab": "Doximity",
          "scoreDisplay": "84.5",
          "scoreNumeric": 84.5,
          "date": "2026-09",
          "config": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. RAG clinical system, not a base-model-only result. Observed September 28, 2026; the individual run date is not published.",
          "modelSlug": null,
          "scorePrinted": "84.5",
          "variant": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. RAG clinical system, not a base-model-only result. Observed September 28, 2026; the individual run date is not published.",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark-owner run",
          "confidence": "verified",
          "resultId": 10037,
          "source": {
            "id": 10542,
            "title": "ARISE MAST technical leaderboard",
            "url": "https://www.arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "First Do NOHARM v2 overall leaderboard; Doximity Ask 6.1; score 84.5%"
          },
          "corroborating": []
        },
        {
          "model": "OpenEvidence",
          "lab": "OpenEvidence",
          "scoreDisplay": "80.0",
          "scoreNumeric": 80,
          "date": "2026-09",
          "config": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. RAG clinical system, not a base-model-only result. Observed September 28, 2026; the individual run date is not published.",
          "modelSlug": null,
          "scorePrinted": "80",
          "variant": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. RAG clinical system, not a base-model-only result. Observed September 28, 2026; the individual run date is not published.",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark-owner run",
          "confidence": "verified",
          "resultId": 10038,
          "source": {
            "id": 10542,
            "title": "ARISE MAST technical leaderboard",
            "url": "https://www.arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "First Do NOHARM v2 overall leaderboard; OpenEvidence; score 80%"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark 1.1",
          "lab": "Meta",
          "scoreDisplay": "79.7%",
          "scoreNumeric": 79.7,
          "date": "2026-08",
          "config": "",
          "modelSlug": "muse-spark-1.1",
          "scorePrinted": "79.7%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 288,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)"
          },
          "corroborating": []
        },
        {
          "model": "Glass 5.6 Max",
          "lab": "Glass Health",
          "scoreDisplay": "79.7",
          "scoreNumeric": 79.7,
          "date": "2026-09",
          "config": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. RAG clinical system, not a base-model-only result. Observed September 28, 2026; the individual run date is not published.",
          "modelSlug": null,
          "scorePrinted": "79.7",
          "variant": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. RAG clinical system, not a base-model-only result. Observed September 28, 2026; the individual run date is not published.",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark-owner run",
          "confidence": "verified",
          "resultId": 10039,
          "source": {
            "id": 10542,
            "title": "ARISE MAST technical leaderboard",
            "url": "https://www.arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "First Do NOHARM v2 overall leaderboard; Glass 5.6 Max; score 79.7%"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "74.6%",
          "scoreNumeric": 74.6,
          "date": "2026-08",
          "config": "v2 run on ARISE; 19 models on the board",
          "modelSlug": "claude-opus-5",
          "scorePrinted": "74.6%",
          "variant": "v2 run on ARISE; 19 models on the board",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 124,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K3",
          "lab": "Moonshot AI",
          "scoreDisplay": "74.0%",
          "scoreNumeric": 74,
          "date": "2026-08",
          "config": "",
          "modelSlug": "kimi-k3",
          "scorePrinted": "74.0%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 125,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "70.1%",
          "scoreNumeric": 70.1,
          "date": "2026-08",
          "config": "",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "70.1%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 126,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.5",
          "lab": "OpenAI",
          "scoreDisplay": "70.0%",
          "scoreNumeric": 70,
          "date": "2026-08",
          "config": "",
          "modelSlug": "gpt-5.5",
          "scorePrinted": "70.0%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 127,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5",
          "lab": "OpenAI",
          "scoreDisplay": "68.6%",
          "scoreNumeric": 68.6,
          "date": "2026-08",
          "config": "from the Model Leaderboard SAFETY column (NOHARM v2 F1 weighted, shown with CI); not in the Latest Flagships ranking",
          "modelSlug": "gpt-5",
          "scorePrinted": "68.6%",
          "variant": "from the Model Leaderboard SAFETY column (NOHARM v2 F1 weighted, shown with CI); not in the Latest Flagships ranking",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 294,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, Model Leaderboard (Top 10 shown), row 10, SAFETY column"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "65.0%",
          "scoreNumeric": 65,
          "date": "2026-08",
          "config": "",
          "modelSlug": "claude-fable-5",
          "scorePrinted": "65.0%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 289,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "62.6%",
          "scoreNumeric": 62.6,
          "date": "2026-08",
          "config": "",
          "modelSlug": "gemini-3.1-pro",
          "scorePrinted": "62.6%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 128,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Pro",
          "lab": "Google",
          "scoreDisplay": "61.9%",
          "scoreNumeric": 61.9,
          "date": "2026-08",
          "config": "",
          "modelSlug": "gemini-2.5-pro",
          "scorePrinted": "61.9%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 290,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3.5 397B A17B",
          "lab": "Alibaba",
          "scoreDisplay": "61.1%",
          "scoreNumeric": 61.1,
          "date": "2026-08",
          "config": "",
          "modelSlug": "qwen3.5-397b-a17b",
          "scorePrinted": "61.1%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 291,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K2.6",
          "lab": "Moonshot AI",
          "scoreDisplay": "59.1%",
          "scoreNumeric": 59.1,
          "date": "2026-08",
          "config": "",
          "modelSlug": "kimi-k2.6",
          "scorePrinted": "59.1%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 292,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.1",
          "lab": "Z.ai",
          "scoreDisplay": "57.9",
          "scoreNumeric": 57.9,
          "date": "2026-09",
          "config": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. Open-weight model as listed by the evaluator. Observed September 28, 2026; the individual run date is not published.",
          "modelSlug": "glm-5.1",
          "scorePrinted": "57.9",
          "variant": "First Do NOHARM v2 overall safety score on ARISE’s technical leaderboard; preview benchmark. Open-weight model as listed by the evaluator. Observed September 28, 2026; the individual run date is not published.",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark-owner run",
          "confidence": "verified",
          "resultId": 10040,
          "source": {
            "id": 10542,
            "title": "ARISE MAST technical leaderboard",
            "url": "https://www.arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "First Do NOHARM v2 overall leaderboard; GLM 5.1; score 57.9%"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek R1",
          "lab": "DeepSeek",
          "scoreDisplay": "55.8%",
          "scoreNumeric": 55.8,
          "date": "2026-08",
          "config": "",
          "modelSlug": "deepseek-r1",
          "scorePrinted": "55.8%",
          "variant": "",
          "measured": "2026-08",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 293,
          "source": {
            "id": 19,
            "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
            "url": "https://arise-ai.org/mast/technical",
            "kind": "official_leaderboard",
            "publisher": "ARISE AI Research Network",
            "firstParty": true,
            "publishedAt": "2026-08-15",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)"
          },
          "corroborating": []
        }
      ],
      "unitShort": "1,100 consultation cases",
      "publisherShort": "Stanford/Harvard consortium",
      "sourceShort": "ARISE MAST technical leaderboard",
      "fieldSize": 19,
      "category": "safety",
      "confidence": "verified",
      "officialUrl": "https://arise-ai.org/mast/technical",
      "paperUrl": "https://arxiv.org/abs/2512.01241",
      "scaleKind": "percent",
      "summary": "How often, and how severely, model consultation recommendations contain potentially harmful errors.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "Checked the official v2 technical board and added five omitted published systems/models. Seventeen of nineteen rows are indexed, with RAG systems labeled separately in settings. Board remains in preview. Newly indexed scores show a September observation date with measurement date left unknown.",
        "urls": [
          "https://arise-ai.org/mast/technical",
          "https://www.arise-ai.org/mast/technical"
        ]
      }
    },
    {
      "slug": "healthagentbench",
      "name": "HealthAgentBench",
      "publisher": "Microsoft Research",
      "released": "2026-07",
      "unit": "54 tasks across 7 environments; 162 trials (3 attempts per task)",
      "scale": "mean task success rate, 0-100%, higher better; cost per task also reported",
      "description": "Agentic task success in realistic terminal-based healthcare environments built from real clinical artifacts; evaluates agent harnesses (Claude Code, Codex, Copilot) end to end, not bare models.",
      "basis": "independent-run",
      "sourceName": "HealthAgentBench leaderboard",
      "sourceUrl": "https://microsoft.github.io/HealthAgentBench/",
      "ours": null,
      "lastUpdate": "2026-07",
      "notes": "Paper: arXiv 2606.31179. The rows are agent harnesses rather than bare models, and the board is run by Microsoft Research; Copilot, Microsoft's own harness, does not top it.",
      "results": [
        {
          "model": "Claude Code (Opus 5)",
          "lab": "Anthropic",
          "scoreDisplay": "55%",
          "scoreNumeric": 55,
          "date": "2026-07",
          "config": "$3.3/task; harness+model evaluated jointly",
          "modelSlug": null,
          "scorePrinted": "55%",
          "variant": "$3.3/task; harness+model evaluated jointly",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 129,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 1, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "55%",
              "quote": null,
              "locator": "Detailed results table, rank 1"
            }
          ]
        },
        {
          "model": "Codex (GPT-5.6-sol)",
          "lab": "OpenAI",
          "scoreDisplay": "45%",
          "scoreNumeric": 45,
          "date": "2026-07",
          "config": "$5.2/task",
          "modelSlug": null,
          "scorePrinted": "45%",
          "variant": "$5.2/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 130,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 2, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "45%",
              "quote": null,
              "locator": "Detailed results table, rank 2"
            }
          ]
        },
        {
          "model": "Codex (GPT 5.5)",
          "lab": "OpenAI",
          "scoreDisplay": "42%",
          "scoreNumeric": 42,
          "date": "2026-07",
          "config": "$2.8/task",
          "modelSlug": null,
          "scorePrinted": "42%",
          "variant": "$2.8/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 131,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 3, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "42%",
              "quote": null,
              "locator": "Detailed results table, rank 3"
            },
            {
              "id": 46,
              "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
              "url": "https://arxiv.org/pdf/2606.31179",
              "kind": "paper",
              "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "42%",
              "quote": null,
              "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)"
            }
          ]
        },
        {
          "model": "Copilot (Opus 4.8)",
          "lab": "Microsoft/Anthropic",
          "scoreDisplay": "36%",
          "scoreNumeric": 36,
          "date": "2026-07",
          "config": "$3.1/task",
          "modelSlug": null,
          "scorePrinted": "36%",
          "variant": "$3.1/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 132,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 4, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "36%",
              "quote": null,
              "locator": "Detailed results table, rank 4"
            },
            {
              "id": 46,
              "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
              "url": "https://arxiv.org/pdf/2606.31179",
              "kind": "paper",
              "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "36%",
              "quote": null,
              "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)"
            }
          ]
        },
        {
          "model": "Copilot (GPT 5.5)",
          "lab": "Microsoft/OpenAI",
          "scoreDisplay": "35%",
          "scoreNumeric": 35,
          "date": "2026-07",
          "config": "$2.6/task",
          "modelSlug": null,
          "scorePrinted": "35%",
          "variant": "$2.6/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 133,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 5, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "35%",
              "quote": null,
              "locator": "Detailed results table, rank 5"
            },
            {
              "id": 46,
              "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
              "url": "https://arxiv.org/pdf/2606.31179",
              "kind": "paper",
              "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "35%",
              "quote": null,
              "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)"
            }
          ]
        },
        {
          "model": "Claude Code (Opus 4.8)",
          "lab": "Anthropic",
          "scoreDisplay": "32%",
          "scoreNumeric": 32,
          "date": "2026-07",
          "config": "$4.0/task",
          "modelSlug": null,
          "scorePrinted": "32%",
          "variant": "$4.0/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 134,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 6, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "32%",
              "quote": null,
              "locator": "Detailed results table, rank 6"
            },
            {
              "id": 46,
              "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
              "url": "https://arxiv.org/pdf/2606.31179",
              "kind": "paper",
              "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "32%",
              "quote": null,
              "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)"
            }
          ]
        },
        {
          "model": "Codex (GPT 5.4)",
          "lab": "OpenAI",
          "scoreDisplay": "28%",
          "scoreNumeric": 28,
          "date": "2026-07",
          "config": "$1.3/task",
          "modelSlug": null,
          "scorePrinted": "28%",
          "variant": "$1.3/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 321,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 7, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "28%",
              "quote": null,
              "locator": "Detailed results table, rank 7"
            },
            {
              "id": 46,
              "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
              "url": "https://arxiv.org/pdf/2606.31179",
              "kind": "paper",
              "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "28%",
              "quote": null,
              "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)"
            }
          ]
        },
        {
          "model": "Claude Code (Opus 4.7)",
          "lab": "Anthropic",
          "scoreDisplay": "27%",
          "scoreNumeric": 27,
          "date": "2026-07",
          "config": "$4.8/task",
          "modelSlug": null,
          "scorePrinted": "27%",
          "variant": "$4.8/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 322,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 8, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "27%",
              "quote": null,
              "locator": "Detailed results table, rank 8"
            },
            {
              "id": 46,
              "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
              "url": "https://arxiv.org/pdf/2606.31179",
              "kind": "paper",
              "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "27%",
              "quote": null,
              "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)"
            }
          ]
        },
        {
          "model": "Codex (GPT 5.3)",
          "lab": "OpenAI",
          "scoreDisplay": "22%",
          "scoreNumeric": 22,
          "date": "2026-07",
          "config": "$1.0/task; harness and model evaluated jointly",
          "modelSlug": null,
          "scorePrinted": "22%",
          "variant": "$1.0/task; harness and model evaluated jointly",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20001,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Homepage leaderboard, rank 9, success rate and cost/task"
          },
          "corroborating": []
        },
        {
          "model": "Claude Code (Opus 4.6)",
          "lab": "Anthropic",
          "scoreDisplay": "19%",
          "scoreNumeric": 19,
          "date": "2026-07",
          "config": "$4.1/task",
          "modelSlug": null,
          "scorePrinted": "19%",
          "variant": "$4.1/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 323,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 10, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "19%",
              "quote": null,
              "locator": "Detailed results table, rank 10"
            },
            {
              "id": 46,
              "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
              "url": "https://arxiv.org/pdf/2606.31179",
              "kind": "paper",
              "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "19%",
              "quote": null,
              "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)"
            }
          ]
        },
        {
          "model": "Claude Code (Sonnet 4.6)",
          "lab": "Anthropic",
          "scoreDisplay": "17%",
          "scoreNumeric": 17,
          "date": "2026-07",
          "config": "$2.9/task",
          "modelSlug": null,
          "scorePrinted": "17%",
          "variant": "$2.9/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 324,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 11, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "17%",
              "quote": null,
              "locator": "Detailed results table, rank 11"
            },
            {
              "id": 46,
              "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
              "url": "https://arxiv.org/pdf/2606.31179",
              "kind": "paper",
              "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "17%",
              "quote": null,
              "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)"
            }
          ]
        },
        {
          "model": "Codex (GPT 5.4 Mini)",
          "lab": "OpenAI",
          "scoreDisplay": "16%",
          "scoreNumeric": 16,
          "date": "2026-07",
          "config": "$0.6/task",
          "modelSlug": null,
          "scorePrinted": "16%",
          "variant": "$0.6/task",
          "measured": "2026-07",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 325,
          "source": {
            "id": 20,
            "title": "HealthAgentBench leaderboard",
            "url": "https://microsoft.github.io/HealthAgentBench/",
            "kind": "official_leaderboard",
            "publisher": "Microsoft Research (HealthAgentBench)",
            "firstParty": true,
            "publishedAt": "2026-07-27",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Leaderboard table (homepage), rank 12, Success Rate column"
          },
          "corroborating": [
            {
              "id": 64,
              "title": "HealthAgentBench detailed results",
              "url": "https://microsoft.github.io/HealthAgentBench/results",
              "kind": "official_leaderboard",
              "publisher": "Microsoft Research (HealthAgentBench)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "16%",
              "quote": null,
              "locator": "Detailed results table, rank 12"
            },
            {
              "id": 46,
              "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
              "url": "https://arxiv.org/pdf/2606.31179",
              "kind": "paper",
              "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "16%",
              "quote": null,
              "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)"
            }
          ]
        }
      ],
      "unitShort": "54 agentic tasks",
      "publisherShort": "Microsoft Research",
      "sourceShort": "HealthAgentBench leaderboard",
      "fieldSize": 12,
      "category": "agentic",
      "confidence": "verified",
      "officialUrl": "https://microsoft.github.io/HealthAgentBench/",
      "paperUrl": "https://arxiv.org/abs/2606.31179",
      "scaleKind": "percent",
      "summary": "Agent harnesses completing realistic terminal-based healthcare tasks built from real clinical artifacts.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "All 12 official harness results checked; added the omitted Codex GPT 5.3 row (22%). Current leader remains Claude Code (Opus 5), 55%. Source does not publish a new run date.",
        "urls": [
          "https://microsoft.github.io/HealthAgentBench/"
        ]
      }
    },
    {
      "slug": "chi-bench",
      "name": "CHI-Bench",
      "publisher": "actAVA.ai",
      "released": "2026-05",
      "unit": "75 workflows (25 per domain), 21 healthcare applications, 200+ MCP tools",
      "scale": "pass@1 with binary 0/1 reward, higher better",
      "description": "Long-horizon US healthcare operations workflows for agents (prior authorization, utilization management, care management), 60-80 step tasks across 4-6 stages, judged by deterministic unit tests plus an LLM judge for evidence grounding, consent, and cross-stage consistency.",
      "basis": "mixed",
      "sourceName": "CHI-Bench leaderboard (actAVA)",
      "sourceUrl": "https://actava.ai/benchmarks/leaderboards",
      "ours": null,
      "lastUpdate": "2026-08-12",
      "notes": "Official board updated 2026-08-12: 45 submitted harness configurations, 44 with all-domain accuracy. The PA-only MedArise submission has no all-domain score and is excluded here. Row dates are submission dates; run dates are not published. Community submissions and author-run baselines share the automated workspace judge.",
      "results": [
        {
          "model": "erius + claude-opus-5",
          "lab": "Humana (harness) / Anthropic (model)",
          "scoreDisplay": "54.7%",
          "scoreNumeric": 54.7,
          "date": "2026-07-26",
          "config": "All Domains pass@1; PA 72.0%, UM 36.0%, CM 56.0%; submitted 2026-07-26; run date not published",
          "modelSlug": null,
          "scorePrinted": "54.7%",
          "variant": "All Domains pass@1; PA 72.0%, UM 36.0%, CM 56.0%; submitted 2026-07-26; run date not published",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 135,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 1, Accuracy column; submission date 2026-07-26"
          },
          "corroborating": []
        },
        {
          "model": "erius + claude-opus-4-8",
          "lab": "Humana (harness) / Anthropic (model)",
          "scoreDisplay": "37.3%",
          "scoreNumeric": 37.3,
          "date": "2026-06-05",
          "config": "All Domains pass@1; PA 40.0%, UM 16.0%, CM 56.0%; submitted 2026-06-05; run date not published",
          "modelSlug": null,
          "scorePrinted": "37.3%",
          "variant": "All Domains pass@1; PA 40.0%, UM 16.0%, CM 56.0%; submitted 2026-06-05; run date not published",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 136,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 2, Accuracy column; submission date 2026-06-05"
          },
          "corroborating": []
        },
        {
          "model": "claude-code + claude-opus-5",
          "lab": "Anthropic",
          "scoreDisplay": "37.3%",
          "scoreNumeric": 37.3,
          "date": "2026-07-24",
          "config": "All Domains pass@1; PA 20.0%, UM 32.0%, CM 60.0%; submitted 2026-07-24; run date not published",
          "modelSlug": null,
          "scorePrinted": "37.3%",
          "variant": "All Domains pass@1; PA 20.0%, UM 32.0%, CM 60.0%; submitted 2026-07-24; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 137,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 3, Accuracy column; submission date 2026-07-24"
          },
          "corroborating": []
        },
        {
          "model": "claude-code + claude-opus-4-8",
          "lab": "Anthropic",
          "scoreDisplay": "33.3%",
          "scoreNumeric": 33.3,
          "date": "2026-05-28",
          "config": "All Domains pass@1; PA 32.0%, UM 28.0%, CM 40.0%; submitted 2026-05-28; run date not published",
          "modelSlug": null,
          "scorePrinted": "33.3%",
          "variant": "All Domains pass@1; PA 32.0%, UM 28.0%, CM 40.0%; submitted 2026-05-28; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 138,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 4, Accuracy column; submission date 2026-05-28"
          },
          "corroborating": []
        },
        {
          "model": "claude-code + claude-opus-4-6",
          "lab": "Anthropic",
          "scoreDisplay": "28.0%",
          "scoreNumeric": 28,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 20.0%, UM 36.0%, CM 28.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "28.0%",
          "variant": "All Domains pass@1; PA 20.0%, UM 36.0%, CM 28.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 139,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 5, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "claude-code + claude-sonnet-4-6",
          "lab": "Anthropic",
          "scoreDisplay": "26.2%",
          "scoreNumeric": 26.2,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 24.0%, UM 34.7%, CM 20.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "26.2%",
          "variant": "All Domains pass@1; PA 24.0%, UM 34.7%, CM 20.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 140,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 6, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "codex + gpt-5.6-sol",
          "lab": "OpenAI",
          "scoreDisplay": "25.3%",
          "scoreNumeric": 25.3,
          "date": "2026-07-24",
          "config": "All Domains pass@1; PA 36.0%, UM 28.0%, CM 12.0%; submitted 2026-07-24; run date not published",
          "modelSlug": null,
          "scorePrinted": "25.3%",
          "variant": "All Domains pass@1; PA 36.0%, UM 28.0%, CM 12.0%; submitted 2026-07-24; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 141,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 7, Accuracy column; submission date 2026-07-24"
          },
          "corroborating": []
        },
        {
          "model": "openai-agents + kimi-k3",
          "lab": "Moonshot AI",
          "scoreDisplay": "25.3%",
          "scoreNumeric": 25.3,
          "date": "2026-07-24",
          "config": "All Domains pass@1; PA 28.0%, UM 32.0%, CM 16.0%; submitted 2026-07-24; run date not published",
          "modelSlug": null,
          "scorePrinted": "25.3%",
          "variant": "All Domains pass@1; PA 28.0%, UM 32.0%, CM 16.0%; submitted 2026-07-24; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 142,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 8, Accuracy column; submission date 2026-07-24"
          },
          "corroborating": []
        },
        {
          "model": "claude-code + claude-opus-4-7",
          "lab": "Anthropic",
          "scoreDisplay": "24.4%",
          "scoreNumeric": 24.4,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 24.0%, UM 17.3%, CM 32.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "24.4%",
          "variant": "All Domains pass@1; PA 24.0%, UM 17.3%, CM 32.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 143,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 9, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "claude-code + claude-fable-5",
          "lab": "Anthropic",
          "scoreDisplay": "24.0%",
          "scoreNumeric": 24,
          "date": "2026-07-22",
          "config": "All Domains pass@1; PA 24.0%, UM 24.0%, CM 24.0%; submitted 2026-07-22; run date not published",
          "modelSlug": null,
          "scorePrinted": "24.0%",
          "variant": "All Domains pass@1; PA 24.0%, UM 24.0%, CM 24.0%; submitted 2026-07-22; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 144,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 10, Accuracy column; submission date 2026-07-22"
          },
          "corroborating": []
        },
        {
          "model": "hermes + MedGuard",
          "lab": "MedGuard",
          "scoreDisplay": "22.7%",
          "scoreNumeric": 22.7,
          "date": "2026-07-06",
          "config": "All Domains pass@1; PA 4.0%, UM 4.0%, CM 60.0%; submitted 2026-07-06; run date not published",
          "modelSlug": null,
          "scorePrinted": "22.7%",
          "variant": "All Domains pass@1; PA 4.0%, UM 4.0%, CM 60.0%; submitted 2026-07-06; run date not published",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "community submission",
          "confidence": "verified",
          "resultId": 20002,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 11, Accuracy column; submission date 2026-07-06"
          },
          "corroborating": []
        },
        {
          "model": "codex + gpt-5.5",
          "lab": "OpenAI",
          "scoreDisplay": "20.9%",
          "scoreNumeric": 20.9,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 29.3%, UM 32.0%, CM 1.3%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "20.9%",
          "variant": "All Domains pass@1; PA 29.3%, UM 32.0%, CM 1.3%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 326,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 12, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "claude-code + claude-sonnet-5",
          "lab": "Anthropic",
          "scoreDisplay": "20.0%",
          "scoreNumeric": 20,
          "date": "2026-07-06",
          "config": "All Domains pass@1; PA 24.0%, UM 24.0%, CM 12.0%; submitted 2026-07-06; run date not published",
          "modelSlug": null,
          "scorePrinted": "20.0%",
          "variant": "All Domains pass@1; PA 24.0%, UM 24.0%, CM 12.0%; submitted 2026-07-06; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 327,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 13, Accuracy column; submission date 2026-07-06"
          },
          "corroborating": []
        },
        {
          "model": "openai-agents + glm-5.1",
          "lab": "Zhipu AI",
          "scoreDisplay": "18.7%",
          "scoreNumeric": 18.7,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 18.7%, UM 33.3%, CM 4.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "18.7%",
          "variant": "All Domains pass@1; PA 18.7%, UM 33.3%, CM 4.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20003,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 14, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "hermes + glm-5.1",
          "lab": "Zhipu AI",
          "scoreDisplay": "18.7%",
          "scoreNumeric": 18.7,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 10.7%, UM 34.7%, CM 10.7%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "18.7%",
          "variant": "All Domains pass@1; PA 10.7%, UM 34.7%, CM 10.7%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20004,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 15, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openai-agents + glm-5.2",
          "lab": "Zhipu AI",
          "scoreDisplay": "18.7%",
          "scoreNumeric": 18.7,
          "date": "2026-07-06",
          "config": "All Domains pass@1; PA 20.0%, UM 32.0%, CM 4.0%; submitted 2026-07-06; run date not published",
          "modelSlug": null,
          "scorePrinted": "18.7%",
          "variant": "All Domains pass@1; PA 20.0%, UM 32.0%, CM 4.0%; submitted 2026-07-06; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20005,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 16, Accuracy column; submission date 2026-07-06"
          },
          "corroborating": []
        },
        {
          "model": "openclaw + claude-opus-4-7",
          "lab": "Anthropic",
          "scoreDisplay": "17.3%",
          "scoreNumeric": 17.3,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 18.7%, UM 13.3%, CM 20.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "17.3%",
          "variant": "All Domains pass@1; PA 18.7%, UM 13.3%, CM 20.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20006,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 17, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openclaw + glm-5.1",
          "lab": "Zhipu AI",
          "scoreDisplay": "16.9%",
          "scoreNumeric": 16.9,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 13.3%, UM 26.7%, CM 10.7%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "16.9%",
          "variant": "All Domains pass@1; PA 13.3%, UM 26.7%, CM 10.7%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20007,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 18, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "hermes + qwen-3.6-max",
          "lab": "Alibaba",
          "scoreDisplay": "16.4%",
          "scoreNumeric": 16.4,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 9.3%, UM 26.7%, CM 13.3%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "16.4%",
          "variant": "All Domains pass@1; PA 9.3%, UM 26.7%, CM 13.3%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20008,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 19, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "codex + gpt-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "16.0%",
          "scoreNumeric": 16,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 24.0%, UM 17.3%, CM 6.7%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "16.0%",
          "variant": "All Domains pass@1; PA 24.0%, UM 17.3%, CM 6.7%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20009,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 20, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openai-agents + qwen-3.6-max",
          "lab": "Alibaba",
          "scoreDisplay": "15.6%",
          "scoreNumeric": 15.6,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 16.0%, UM 26.7%, CM 4.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "15.6%",
          "variant": "All Domains pass@1; PA 16.0%, UM 26.7%, CM 4.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20010,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 21, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "hermes + kimi-k2.6",
          "lab": "Moonshot AI",
          "scoreDisplay": "15.6%",
          "scoreNumeric": 15.6,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 18.7%, UM 21.3%, CM 6.7%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "15.6%",
          "variant": "All Domains pass@1; PA 18.7%, UM 21.3%, CM 6.7%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20011,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 22, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openai-agents + kimi-k2.6",
          "lab": "Moonshot AI",
          "scoreDisplay": "15.1%",
          "scoreNumeric": 15.1,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 17.3%, UM 25.3%, CM 2.7%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "15.1%",
          "variant": "All Domains pass@1; PA 17.3%, UM 25.3%, CM 2.7%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20012,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 23, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openai-agents + deepseek-v4-pro",
          "lab": "DeepSeek",
          "scoreDisplay": "14.2%",
          "scoreNumeric": 14.2,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 10.7%, UM 28.0%, CM 4.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "14.2%",
          "variant": "All Domains pass@1; PA 10.7%, UM 28.0%, CM 4.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20013,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 24, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "hermes + deepseek-v4-pro",
          "lab": "DeepSeek",
          "scoreDisplay": "13.8%",
          "scoreNumeric": 13.8,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 8.0%, UM 25.3%, CM 8.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "13.8%",
          "variant": "All Domains pass@1; PA 8.0%, UM 25.3%, CM 8.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20014,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 25, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "codex + gpt-5.6-terra",
          "lab": "OpenAI",
          "scoreDisplay": "13.3%",
          "scoreNumeric": 13.3,
          "date": "2026-07-24",
          "config": "All Domains pass@1; PA 12.0%, UM 20.0%, CM 8.0%; submitted 2026-07-24; run date not published",
          "modelSlug": null,
          "scorePrinted": "13.3%",
          "variant": "All Domains pass@1; PA 12.0%, UM 20.0%, CM 8.0%; submitted 2026-07-24; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20015,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 26, Accuracy column; submission date 2026-07-24"
          },
          "corroborating": []
        },
        {
          "model": "codex + gpt-5.6-luna",
          "lab": "OpenAI",
          "scoreDisplay": "13.3%",
          "scoreNumeric": 13.3,
          "date": "2026-07-24",
          "config": "All Domains pass@1; PA 20.0%, UM 16.0%, CM 4.0%; submitted 2026-07-24; run date not published",
          "modelSlug": null,
          "scorePrinted": "13.3%",
          "variant": "All Domains pass@1; PA 20.0%, UM 16.0%, CM 4.0%; submitted 2026-07-24; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20016,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 27, Accuracy column; submission date 2026-07-24"
          },
          "corroborating": []
        },
        {
          "model": "gemini-cli + gemini-3-flash",
          "lab": "Google",
          "scoreDisplay": "12.5%",
          "scoreNumeric": 12.5,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 18.7%, UM 18.7%, CM 0.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "12.5%",
          "variant": "All Domains pass@1; PA 18.7%, UM 18.7%, CM 0.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20017,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 28, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openclaw + deepseek-v4-pro",
          "lab": "DeepSeek",
          "scoreDisplay": "11.1%",
          "scoreNumeric": 11.1,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 14.7%, UM 12.0%, CM 6.7%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "11.1%",
          "variant": "All Domains pass@1; PA 14.7%, UM 12.0%, CM 6.7%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20018,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 29, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "deepagents + glm-5.1",
          "lab": "Zhipu AI",
          "scoreDisplay": "11.1%",
          "scoreNumeric": 11.1,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 17.3%, UM 10.7%, CM 5.3%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "11.1%",
          "variant": "All Domains pass@1; PA 17.3%, UM 10.7%, CM 5.3%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20019,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 30, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "deepagents + deepseek-v4-pro",
          "lab": "DeepSeek",
          "scoreDisplay": "10.7%",
          "scoreNumeric": 10.7,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 14.7%, UM 10.7%, CM 6.7%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "10.7%",
          "variant": "All Domains pass@1; PA 14.7%, UM 10.7%, CM 6.7%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20020,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 31, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openclaw + kimi-k2.6",
          "lab": "Moonshot AI",
          "scoreDisplay": "10.2%",
          "scoreNumeric": 10.2,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 12.0%, UM 18.7%, CM 0.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "10.2%",
          "variant": "All Domains pass@1; PA 12.0%, UM 18.7%, CM 0.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20021,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 32, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "deepagents + qwen-3.6-max",
          "lab": "Alibaba",
          "scoreDisplay": "9.3%",
          "scoreNumeric": 9.3,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 12.0%, UM 10.7%, CM 5.3%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "9.3%",
          "variant": "All Domains pass@1; PA 12.0%, UM 10.7%, CM 5.3%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20022,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 33, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "codex + gpt-5.4-mini",
          "lab": "OpenAI",
          "scoreDisplay": "8.4%",
          "scoreNumeric": 8.4,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 10.7%, UM 13.3%, CM 1.3%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "8.4%",
          "variant": "All Domains pass@1; PA 10.7%, UM 13.3%, CM 1.3%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20023,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 34, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openai-agents + TML Inkling 256K",
          "lab": "Thinking Machines",
          "scoreDisplay": "8.0%",
          "scoreNumeric": 8,
          "date": "2026-07-24",
          "config": "All Domains pass@1; PA 4.0%, UM 16.0%, CM 4.0%; submitted 2026-07-24; run date not published",
          "modelSlug": null,
          "scorePrinted": "8.0%",
          "variant": "All Domains pass@1; PA 4.0%, UM 16.0%, CM 4.0%; submitted 2026-07-24; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20024,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 35, Accuracy column; submission date 2026-07-24"
          },
          "corroborating": []
        },
        {
          "model": "gemini-cli + gemini-3.1-pro",
          "lab": "Google",
          "scoreDisplay": "7.1%",
          "scoreNumeric": 7.1,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 14.7%, UM 6.7%, CM 0.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "7.1%",
          "variant": "All Domains pass@1; PA 14.7%, UM 6.7%, CM 0.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20025,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 36, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "claude-code + claude-haiku-4-5",
          "lab": "Anthropic",
          "scoreDisplay": "6.2%",
          "scoreNumeric": 6.2,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 0.0%, UM 14.7%, CM 4.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "6.2%",
          "variant": "All Domains pass@1; PA 0.0%, UM 14.7%, CM 4.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20026,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 37, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openai-agents + grok-4.3",
          "lab": "SpaceX AI",
          "scoreDisplay": "5.8%",
          "scoreNumeric": 5.8,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 0.0%, UM 16.0%, CM 1.3%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "5.8%",
          "variant": "All Domains pass@1; PA 0.0%, UM 16.0%, CM 1.3%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20027,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 38, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openclaw + qwen-3.6-max",
          "lab": "Alibaba",
          "scoreDisplay": "4.9%",
          "scoreNumeric": 4.9,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 10.7%, UM 4.0%, CM 0.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "4.9%",
          "variant": "All Domains pass@1; PA 10.7%, UM 4.0%, CM 0.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20028,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 39, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "hermes + grok-4.3",
          "lab": "SpaceX AI",
          "scoreDisplay": "4.4%",
          "scoreNumeric": 4.4,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 0.0%, UM 13.3%, CM 0.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "4.4%",
          "variant": "All Domains pass@1; PA 0.0%, UM 13.3%, CM 0.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20029,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 40, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "deepagents + kimi-k2.6",
          "lab": "Moonshot AI",
          "scoreDisplay": "3.1%",
          "scoreNumeric": 3.1,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 8.0%, UM 1.3%, CM 0.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "3.1%",
          "variant": "All Domains pass@1; PA 8.0%, UM 1.3%, CM 0.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20030,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 41, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "deepagents + grok-4.3",
          "lab": "SpaceX AI",
          "scoreDisplay": "2.2%",
          "scoreNumeric": 2.2,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 0.0%, UM 5.3%, CM 1.3%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "2.2%",
          "variant": "All Domains pass@1; PA 0.0%, UM 5.3%, CM 1.3%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20031,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 42, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openclaw + grok-4.3",
          "lab": "SpaceX AI",
          "scoreDisplay": "0.4%",
          "scoreNumeric": 0.4,
          "date": "2026-05-01",
          "config": "All Domains pass@1; PA 1.3%, UM 0.0%, CM 0.0%; submitted 2026-05-01; run date not published",
          "modelSlug": null,
          "scorePrinted": "0.4%",
          "variant": "All Domains pass@1; PA 1.3%, UM 0.0%, CM 0.0%; submitted 2026-05-01; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20032,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 43, Accuracy column; submission date 2026-05-01"
          },
          "corroborating": []
        },
        {
          "model": "openai-agents + Nemotron 3 Ultra 256K",
          "lab": "NVIDIA",
          "scoreDisplay": "0.0%",
          "scoreNumeric": 0,
          "date": "2026-07-24",
          "config": "All Domains pass@1; PA 0.0%, UM 0.0%, CM 0.0%; submitted 2026-07-24; run date not published",
          "modelSlug": null,
          "scorePrinted": "0.0%",
          "variant": "All Domains pass@1; PA 0.0%, UM 0.0%, CM 0.0%; submitted 2026-07-24; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20033,
          "source": {
            "id": 21,
            "title": "CHI-Bench leaderboard (actAVA)",
            "url": "https://actava.ai/benchmarks/leaderboards",
            "kind": "official_leaderboard",
            "publisher": "actAVA",
            "firstParty": true,
            "publishedAt": "2026-08-12",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "All Domains leaderboard, rank 44, Accuracy column; submission date 2026-07-24"
          },
          "corroborating": []
        }
      ],
      "unitShort": "75 operations workflows",
      "publisherShort": "actAVA",
      "sourceShort": "CHI-Bench leaderboard",
      "fieldSize": 45,
      "category": "agentic",
      "confidence": "verified",
      "officialUrl": "https://actava.ai/benchmarks/leaderboards",
      "paperUrl": null,
      "scaleKind": "percent",
      "summary": "Long-horizon healthcare operations workflows for agents: prior authorization, utilization management, and care management.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "All 45 submissions inspected; retained all 44 numeric all-domain results and excluded one PA-only result. No newer scores found on the official board.",
        "urls": [
          "https://actava.ai/benchmarks/leaderboards"
        ]
      }
    },
    {
      "slug": "medcode",
      "name": "MedCode (Vals AI)",
      "publisher": "Vals AI (dataset with Protege)",
      "released": "2026-02",
      "unit": "2,755 patient records",
      "scale": "percentage accuracy 0-100, higher better",
      "description": "ICD-10-CM diagnosis coding for entire hospital stays: models assign primary and secondary codes from discharge summaries plus progress/consult notes; ground truth double-annotated by certified professional coders.",
      "basis": "independent-run",
      "sourceName": "Vals AI MedCode leaderboard",
      "sourceUrl": "https://www.vals.ai/benchmarks/medcode",
      "ours": null,
      "lastUpdate": "2026-09-26",
      "notes": "Vals AI runs this benchmark. Full Overall data contains 102 scored model configurations in the 2026-09-26 snapshot, including older and reasoning variants. Scores use the published percent scale; numeric values retain source precision and displayed values round to two decimals. Snapshot update dates are not model measurement dates.",
      "results": [
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "63.57%",
          "scoreNumeric": 63.57,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.993 pp; $0.156845/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-5",
          "scorePrinted": "63.57%",
          "variant": "model ID anthropic/claude-opus-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.993 pp; $0.156845/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 145,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-5\"].accuracy; Overall leaderboard rank 1"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.1 Pro Preview (02/26)",
          "lab": "Google",
          "scoreDisplay": "59.06%",
          "scoreNumeric": 59.062,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.1-pro-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.996 pp; $0.024714/test; source snapshot 2026-09-26; run date not published",
          "alias": "Gemini 3.1 Pro",
          "modelSlug": "gemini-3.1-pro",
          "scorePrinted": "59.06%",
          "variant": "model ID google/gemini-3.1-pro-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.996 pp; $0.024714/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 146,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.1-pro-preview\"].accuracy; Overall leaderboard rank 2"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "56.07%",
          "scoreNumeric": 56.07,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-fable-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.203 pp; $0.591071/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-fable-5",
          "scorePrinted": "56.07%",
          "variant": "model ID anthropic/claude-fable-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.203 pp; $0.591071/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 147,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-fable-5\"].accuracy; Overall leaderboard rank 3"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3 Flash (12/25)",
          "lab": "Google",
          "scoreDisplay": "55.92%",
          "scoreNumeric": 55.92,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3-flash-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.112 pp; $0.006187/test; source snapshot 2026-09-26; run date not published",
          "alias": "Gemini 3 Flash",
          "modelSlug": "gemini-3-flash",
          "scorePrinted": "55.92%",
          "variant": "model ID google/gemini-3-flash-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.112 pp; $0.006187/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 148,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3-flash-preview\"].accuracy; Overall leaderboard rank 4"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.5 Flash",
          "lab": "Google",
          "scoreDisplay": "55.83%",
          "scoreNumeric": 55.825,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.5-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.113 pp; $0.073716/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.5-flash",
          "scorePrinted": "55.83%",
          "variant": "model ID google/gemini-3.5-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.113 pp; $0.073716/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 149,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.5-flash\"].accuracy; Overall leaderboard rank 5"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.7",
          "lab": "Anthropic",
          "scoreDisplay": "54.86%",
          "scoreNumeric": 54.858,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-7; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.205 pp; $0.226314/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.7",
          "scorePrinted": "54.86%",
          "variant": "model ID anthropic/claude-opus-4-7; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.205 pp; $0.226314/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 346,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-7\"].accuracy; Overall leaderboard rank 6"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5.1",
          "lab": "Anthropic",
          "scoreDisplay": "53.51%",
          "scoreNumeric": 53.509,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-fable-5-1; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 2.165 pp; $1.116862/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-fable-5.1",
          "scorePrinted": "53.51%",
          "variant": "model ID anthropic/claude-fable-5-1; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 2.165 pp; $1.116862/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 348,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-fable-5-1\"].accuracy; Overall leaderboard rank 7"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.7 Flash",
          "lab": "Google",
          "scoreDisplay": "53.39%",
          "scoreNumeric": 53.39,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.7-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.12 pp; $0.038331/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.7-flash",
          "scorePrinted": "53.39%",
          "variant": "model ID google/gemini-3.7-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.12 pp; $0.038331/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20034,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.7-flash\"].accuracy; Overall leaderboard rank 8"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.8",
          "lab": "Anthropic",
          "scoreDisplay": "53.22%",
          "scoreNumeric": 53.217,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-8; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.165 pp; $0.350925/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.8",
          "scorePrinted": "53.22%",
          "variant": "model ID anthropic/claude-opus-4-8; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.165 pp; $0.350925/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 350,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-8\"].accuracy; Overall leaderboard rank 9"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.6 Flash",
          "lab": "Google",
          "scoreDisplay": "53.15%",
          "scoreNumeric": 53.153,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.6-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.157 pp; $0.044216/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.6-flash",
          "scorePrinted": "53.15%",
          "variant": "model ID google/gemini-3.6-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.157 pp; $0.044216/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 352,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.6-flash\"].accuracy; Overall leaderboard rank 10"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "52.92%",
          "scoreNumeric": 52.92,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-5-5; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 2.119 pp; $0.391592/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-5.5",
          "scorePrinted": "52.92%",
          "variant": "model ID anthropic/claude-sonnet-5-5; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 2.119 pp; $0.391592/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20035,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-5-5\"].accuracy; Overall leaderboard rank 11"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.1",
          "lab": "OpenAI",
          "scoreDisplay": "52.73%",
          "scoreNumeric": 52.732,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.1-2025-11-13; reasoning_effort=high; max_output_tokens=30000; standard error 2.151 pp; $0.014371/test; source snapshot 2026-09-26; run date not published",
          "alias": "GPT-5.1",
          "modelSlug": "gpt-5.1",
          "scorePrinted": "52.73%",
          "variant": "model ID openai/gpt-5.1-2025-11-13; reasoning_effort=high; max_output_tokens=30000; standard error 2.151 pp; $0.014371/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 353,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.1-2025-11-13\"].accuracy; Overall leaderboard rank 12"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3 Pro (11/25)",
          "lab": "Google",
          "scoreDisplay": "52.20%",
          "scoreNumeric": 52.198,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3-pro-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.073 pp; $0.028248/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3-pro-11-25",
          "scorePrinted": "52.20%",
          "variant": "model ID google/gemini-3-pro-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.073 pp; $0.028248/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20036,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3-pro-preview\"].accuracy; Overall leaderboard rank 13"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark",
          "lab": "Meta",
          "scoreDisplay": "51.31%",
          "scoreNumeric": 51.31,
          "date": "2026-09-26",
          "config": "model ID meta/muse_spark; temperature=1; max_output_tokens=30000; standard error 2.244 pp; $0.005341/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "muse-spark",
          "scorePrinted": "51.31%",
          "variant": "model ID meta/muse_spark; temperature=1; max_output_tokens=30000; standard error 2.244 pp; $0.005341/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20037,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark\"].accuracy; Overall leaderboard rank 14"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Pro",
          "lab": "Google",
          "scoreDisplay": "50.59%",
          "scoreNumeric": 50.59,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-pro; temperature=1; max_output_tokens=30000; standard error 2.113 pp; $0.015389/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-pro",
          "scorePrinted": "50.59%",
          "variant": "model ID google/gemini-2.5-pro; temperature=1; max_output_tokens=30000; standard error 2.113 pp; $0.015389/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20038,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-pro\"].accuracy; Overall leaderboard rank 15"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "49.80%",
          "scoreNumeric": 49.797,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-5-5; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 2.273 pp; $0.658237/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-5.5",
          "scorePrinted": "49.80%",
          "variant": "model ID anthropic/claude-opus-5-5; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 2.273 pp; $0.658237/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20039,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-5-5\"].accuracy; Overall leaderboard rank 16"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.2",
          "lab": "OpenAI",
          "scoreDisplay": "49.75%",
          "scoreNumeric": 49.749,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.2-2025-12-11; reasoning_effort=xhigh; max_output_tokens=30000; standard error 2.262 pp; $0.018852/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.2",
          "scorePrinted": "49.75%",
          "variant": "model ID openai/gpt-5.2-2025-12-11; reasoning_effort=xhigh; max_output_tokens=30000; standard error 2.262 pp; $0.018852/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20040,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.2-2025-12-11\"].accuracy; Overall leaderboard rank 17"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5",
          "lab": "OpenAI",
          "scoreDisplay": "49.63%",
          "scoreNumeric": 49.634,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 2.098 pp; $0.045858/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5",
          "scorePrinted": "49.63%",
          "variant": "model ID openai/gpt-5-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 2.098 pp; $0.045858/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20041,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-2025-08-07\"].accuracy; Overall leaderboard rank 18"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.7",
          "lab": "SpaceXAI",
          "scoreDisplay": "49.55%",
          "scoreNumeric": 49.553,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.7; reasoning_effort=xhigh; temperature=1; top_p=0.95; standard error 2.171 pp; $0.105497/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.7",
          "scorePrinted": "49.55%",
          "variant": "model ID grok/grok-4.7; reasoning_effort=xhigh; temperature=1; top_p=0.95; standard error 2.171 pp; $0.105497/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20042,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.7\"].accuracy; Overall leaderboard rank 19"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark 1.2",
          "lab": "Meta",
          "scoreDisplay": "49.35%",
          "scoreNumeric": 49.346,
          "date": "2026-09-26",
          "config": "model ID meta/muse_spark_1_2; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 2.187 pp; $0.039023/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "muse-spark-1.2",
          "scorePrinted": "49.35%",
          "variant": "model ID meta/muse_spark_1_2; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 2.187 pp; $0.039023/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20043,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark_1_2\"].accuracy; Overall leaderboard rank 20"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.5 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "49.16%",
          "scoreNumeric": 49.156,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-5-20251101-thinking; compute_effort=high; temperature=1; max_output_tokens=30000; standard error 2.012 pp; $0.095846/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.5",
          "scorePrinted": "49.16%",
          "variant": "model ID anthropic/claude-opus-4-5-20251101-thinking; compute_effort=high; temperature=1; max_output_tokens=30000; standard error 2.012 pp; $0.095846/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20044,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-5-20251101-thinking\"].accuracy; Overall leaderboard rank 21"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.6 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "49.13%",
          "scoreNumeric": 49.129,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-6-thinking; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.085 pp; $0.244127/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "49.13%",
          "variant": "model ID anthropic/claude-opus-4-6-thinking; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.085 pp; $0.244127/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20045,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-6-thinking\"].accuracy; Overall leaderboard rank 22"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.5",
          "lab": "OpenAI",
          "scoreDisplay": "49.10%",
          "scoreNumeric": 49.1,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.5; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 2.188 pp; $0.160759/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.5",
          "scorePrinted": "49.10%",
          "variant": "model ID openai/gpt-5.5; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 2.188 pp; $0.160759/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20046,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.5\"].accuracy; Overall leaderboard rank 23"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K3",
          "lab": "Moonshot AI",
          "scoreDisplay": "48.88%",
          "scoreNumeric": 48.884,
          "date": "2026-09-26",
          "config": "model ID kimi/kimi-k3; temperature=1; max_output_tokens=30000; standard error 2.193 pp; $0.076379/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "kimi-k3",
          "scorePrinted": "48.88%",
          "variant": "model ID kimi/kimi-k3; temperature=1; max_output_tokens=30000; standard error 2.193 pp; $0.076379/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20047,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k3\"].accuracy; Overall leaderboard rank 24"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Astra",
          "lab": "OpenAI",
          "scoreDisplay": "48.49%",
          "scoreNumeric": 48.486,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-6-astra; reasoning_effort=max; max_output_tokens=128000; standard error 2.131 pp; $0.451358/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-6-astra",
          "scorePrinted": "48.49%",
          "variant": "model ID openai/gpt-6-astra; reasoning_effort=max; max_output_tokens=128000; standard error 2.131 pp; $0.451358/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 568,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-astra\"].accuracy; Overall leaderboard rank 25"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.6 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "48.24%",
          "scoreNumeric": 48.244,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-6; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.05 pp; $0.006180/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "48.24%",
          "variant": "model ID anthropic/claude-opus-4-6; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.05 pp; $0.006180/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20048,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-6\"].accuracy; Overall leaderboard rank 26"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.8 Flash",
          "lab": "Google",
          "scoreDisplay": "48.13%",
          "scoreNumeric": 48.135,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.8-flash; reasoning_effort=high; temperature=1; max_output_tokens=65536; standard error 2.18 pp; $0.017700/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.8-flash",
          "scorePrinted": "48.13%",
          "variant": "model ID google/gemini-3.8-flash; reasoning_effort=high; temperature=1; max_output_tokens=65536; standard error 2.18 pp; $0.017700/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20049,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.8-flash\"].accuracy; Overall leaderboard rank 27"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.1 Flash Lite Preview",
          "lab": "Google",
          "scoreDisplay": "47.60%",
          "scoreNumeric": 47.602,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.1-flash-lite-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.071 pp; $0.002029/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.1-flash-lite-preview",
          "scorePrinted": "47.60%",
          "variant": "model ID google/gemini-3.1-flash-lite-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.071 pp; $0.002029/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20050,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.1-flash-lite-preview\"].accuracy; Overall leaderboard rank 28"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5",
          "lab": "Anthropic",
          "scoreDisplay": "47.54%",
          "scoreNumeric": 47.54,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.274 pp; $0.278799/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-5",
          "scorePrinted": "47.54%",
          "variant": "model ID anthropic/claude-sonnet-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 2.274 pp; $0.278799/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20051,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-5\"].accuracy; Overall leaderboard rank 29"
          },
          "corroborating": []
        },
        {
          "model": "o3",
          "lab": "OpenAI",
          "scoreDisplay": "47.29%",
          "scoreNumeric": 47.29,
          "date": "2026-09-26",
          "config": "model ID openai/o3-2025-04-16; reasoning_effort=high; max_output_tokens=30000; standard error 2.161 pp; $0.029818/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "o3",
          "scorePrinted": "47.29%",
          "variant": "model ID openai/o3-2025-04-16; reasoning_effort=high; max_output_tokens=30000; standard error 2.161 pp; $0.029818/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20052,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/o3-2025-04-16\"].accuracy; Overall leaderboard rank 30"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.1 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "47.23%",
          "scoreNumeric": 47.235,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-1-20250805-thinking; temperature=1; max_output_tokens=30000; standard error 2.067 pp; $0.269254/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.1",
          "scorePrinted": "47.23%",
          "variant": "model ID anthropic/claude-opus-4-1-20250805-thinking; temperature=1; max_output_tokens=30000; standard error 2.067 pp; $0.269254/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20053,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-1-20250805-thinking\"].accuracy; Overall leaderboard rank 31"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "47.07%",
          "scoreNumeric": 47.072,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-6-sol; reasoning_effort=max; max_output_tokens=128000; standard error 2.119 pp; $0.085827/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-6-sol",
          "scorePrinted": "47.07%",
          "variant": "model ID openai/gpt-6-sol; reasoning_effort=max; max_output_tokens=128000; standard error 2.119 pp; $0.085827/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20054,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-sol\"].accuracy; Overall leaderboard rank 32"
          },
          "corroborating": []
        },
        {
          "model": "MiniMax-M3",
          "lab": "MiniMax",
          "scoreDisplay": "46.29%",
          "scoreNumeric": 46.289,
          "date": "2026-09-26",
          "config": "model ID minimax/MiniMax-M3; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.104 pp; $0.012125/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "minimax-m3",
          "scorePrinted": "46.29%",
          "variant": "model ID minimax/MiniMax-M3; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.104 pp; $0.012125/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20055,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M3\"].accuracy; Overall leaderboard rank 33"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.5 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "45.17%",
          "scoreNumeric": 45.174,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-5-20251101; compute_effort=high; temperature=1; max_output_tokens=30000; standard error 1.888 pp; $0.006826/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.5",
          "scorePrinted": "45.17%",
          "variant": "model ID anthropic/claude-opus-4-5-20251101; compute_effort=high; temperature=1; max_output_tokens=30000; standard error 1.888 pp; $0.006826/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20056,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-5-20251101\"].accuracy; Overall leaderboard rank 34"
          },
          "corroborating": []
        },
        {
          "model": "MiMo V2.6 Pro",
          "lab": "Xiaomi",
          "scoreDisplay": "44.97%",
          "scoreNumeric": 44.969,
          "date": "2026-09-26",
          "config": "model ID xiaomi/mimo-v2.6-pro; temperature=1; top_p=0.95; max_output_tokens=128000; standard error 2.097 pp; $0.009528/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mimo-v2.6-pro",
          "scorePrinted": "44.97%",
          "variant": "model ID xiaomi/mimo-v2.6-pro; temperature=1; top_p=0.95; max_output_tokens=128000; standard error 2.097 pp; $0.009528/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20057,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.6-pro\"].accuracy; Overall leaderboard rank 35"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.6",
          "lab": "SpaceXAI",
          "scoreDisplay": "44.71%",
          "scoreNumeric": 44.712,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.6; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.256 pp; $0.050335/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.6",
          "scorePrinted": "44.71%",
          "variant": "model ID grok/grok-4.6; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.256 pp; $0.050335/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20058,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.6\"].accuracy; Overall leaderboard rank 36"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "44.69%",
          "scoreNumeric": 44.685,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-6-luna; reasoning_effort=max; max_output_tokens=128000; standard error 2.303 pp; $0.007323/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-6-luna",
          "scorePrinted": "44.69%",
          "variant": "model ID openai/gpt-6-luna; reasoning_effort=max; max_output_tokens=128000; standard error 2.303 pp; $0.007323/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20059,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-luna\"].accuracy; Overall leaderboard rank 37"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4.5 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "44.13%",
          "scoreNumeric": 44.134,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-4-5-20250929-thinking; temperature=1; max_output_tokens=30000; standard error 1.998 pp; $0.101495/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-4.5",
          "scorePrinted": "44.13%",
          "variant": "model ID anthropic/claude-sonnet-4-5-20250929-thinking; temperature=1; max_output_tokens=30000; standard error 1.998 pp; $0.101495/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20060,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-5-20250929-thinking\"].accuracy; Overall leaderboard rank 38"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "43.97%",
          "scoreNumeric": 43.974,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.6-sol; reasoning_effort=max; max_output_tokens=30000; standard error 2.258 pp; $0.280517/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "43.97%",
          "variant": "model ID openai/gpt-5.6-sol; reasoning_effort=max; max_output_tokens=30000; standard error 2.258 pp; $0.280517/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20061,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-sol\"].accuracy; Overall leaderboard rank 39"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.5 Flash Lite",
          "lab": "Google",
          "scoreDisplay": "43.49%",
          "scoreNumeric": 43.489,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.5-flash-lite; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.951 pp; $0.008093/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.5-flash-lite",
          "scorePrinted": "43.49%",
          "variant": "model ID google/gemini-3.5-flash-lite; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.951 pp; $0.008093/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20062,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.5-flash-lite\"].accuracy; Overall leaderboard rank 40"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Terra",
          "lab": "OpenAI",
          "scoreDisplay": "43.41%",
          "scoreNumeric": 43.414,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.6-terra; reasoning_effort=xhigh; max_output_tokens=30000; standard error 2.173 pp; $0.046558/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.6-terra",
          "scorePrinted": "43.41%",
          "variant": "model ID openai/gpt-5.6-terra; reasoning_effort=xhigh; max_output_tokens=30000; standard error 2.173 pp; $0.046558/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20063,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-terra\"].accuracy; Overall leaderboard rank 41"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.5",
          "lab": "SpaceXAI",
          "scoreDisplay": "43.29%",
          "scoreNumeric": 43.291,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.5; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.313 pp; $0.048461/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.5",
          "scorePrinted": "43.29%",
          "variant": "model ID grok/grok-4.5; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.313 pp; $0.048461/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20064,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.5\"].accuracy; Overall leaderboard rank 42"
          },
          "corroborating": []
        },
        {
          "model": "Hy4 Preview",
          "lab": "Tencent",
          "scoreDisplay": "43.25%",
          "scoreNumeric": 43.247,
          "date": "2026-09-26",
          "config": "model ID tencent/hy4-preview; temperature=1; top_p=1; max_output_tokens=64000; standard error 2.134 pp; $0.062053/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "hy4-preview",
          "scorePrinted": "43.25%",
          "variant": "model ID tencent/hy4-preview; temperature=1; top_p=1; max_output_tokens=64000; standard error 2.134 pp; $0.062053/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20065,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"tencent/hy4-preview\"].accuracy; Overall leaderboard rank 43"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5 Mini",
          "lab": "OpenAI",
          "scoreDisplay": "43.05%",
          "scoreNumeric": 43.045,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5-mini-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 2.045 pp; $0.005560/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5-mini",
          "scorePrinted": "43.05%",
          "variant": "model ID openai/gpt-5-mini-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 2.045 pp; $0.005560/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20066,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-mini-2025-08-07\"].accuracy; Overall leaderboard rank 44"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.3",
          "lab": "zAI",
          "scoreDisplay": "42.86%",
          "scoreNumeric": 42.864,
          "date": "2026-09-26",
          "config": "model ID zai/glm-5.3; reasoning_effort=max; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.111 pp; $0.081905/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "glm-5.3",
          "scorePrinted": "42.86%",
          "variant": "model ID zai/glm-5.3; reasoning_effort=max; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.111 pp; $0.081905/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20067,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.3\"].accuracy; Overall leaderboard rank 45"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V4 Pro 0813",
          "lab": "DeepSeek",
          "scoreDisplay": "42.47%",
          "scoreNumeric": 42.47,
          "date": "2026-09-26",
          "config": "model ID deepseek/deepseek-v4-pro-0813; reasoning_effort=max; max_output_tokens=30000; standard error 2.16 pp; $0.061060/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "deepseek-v4-pro-0813",
          "scorePrinted": "42.47%",
          "variant": "model ID deepseek/deepseek-v4-pro-0813; reasoning_effort=max; max_output_tokens=30000; standard error 2.16 pp; $0.061060/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20068,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-pro-0813\"].accuracy; Overall leaderboard rank 46"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "42.39%",
          "scoreNumeric": 42.391,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.6-luna; reasoning_effort=max; max_output_tokens=30000; standard error 2.27 pp; $0.015970/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.6-luna",
          "scorePrinted": "42.39%",
          "variant": "model ID openai/gpt-5.6-luna; reasoning_effort=max; max_output_tokens=30000; standard error 2.27 pp; $0.015970/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20069,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-luna\"].accuracy; Overall leaderboard rank 47"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.1",
          "lab": "zAI",
          "scoreDisplay": "41.60%",
          "scoreNumeric": 41.604,
          "date": "2026-09-26",
          "config": "model ID zai/glm-5.1; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.124 pp; $0.024196/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "glm-5.1",
          "scorePrinted": "41.60%",
          "variant": "model ID zai/glm-5.1; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.124 pp; $0.024196/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20070,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.1\"].accuracy; Overall leaderboard rank 48"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V4 Flash 0731",
          "lab": "DeepSeek",
          "scoreDisplay": "41.41%",
          "scoreNumeric": 41.415,
          "date": "2026-09-26",
          "config": "model ID deepseek/deepseek-v4-flash-0731; reasoning_effort=high; max_output_tokens=30000; standard error 2.15 pp; $0.019659/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "deepseek-v4-flash-0731",
          "scorePrinted": "41.41%",
          "variant": "model ID deepseek/deepseek-v4-flash-0731; reasoning_effort=high; max_output_tokens=30000; standard error 2.15 pp; $0.019659/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20071,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-flash-0731\"].accuracy; Overall leaderboard rank 49"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.1 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "41.37%",
          "scoreNumeric": 41.372,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-1-20250805; temperature=1; max_output_tokens=30000; standard error 1.958 pp; $0.206270/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.1",
          "scorePrinted": "41.37%",
          "variant": "model ID anthropic/claude-opus-4-1-20250805; temperature=1; max_output_tokens=30000; standard error 1.958 pp; $0.206270/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20072,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-1-20250805\"].accuracy; Overall leaderboard rank 50"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.4 (xhigh)",
          "lab": "OpenAI",
          "scoreDisplay": "41.29%",
          "scoreNumeric": 41.292,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.4-2026-03-05; reasoning_effort=xhigh; max_output_tokens=30000; standard error 2.148 pp; $0.212108/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.4-xhigh",
          "scorePrinted": "41.29%",
          "variant": "model ID openai/gpt-5.4-2026-03-05; reasoning_effort=xhigh; max_output_tokens=30000; standard error 2.148 pp; $0.212108/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20073,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.4-2026-03-05\"].accuracy; Overall leaderboard rank 51"
          },
          "corroborating": []
        },
        {
          "model": "Inkling",
          "lab": "Thinking Machines",
          "scoreDisplay": "41.19%",
          "scoreNumeric": 41.19,
          "date": "2026-09-26",
          "config": "model ID thinkingmachines/inkling; reasoning_effort=0.99; temperature=1; top_p=1; max_output_tokens=30000; standard error 2.23 pp; $0.127025/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "inkling",
          "scorePrinted": "41.19%",
          "variant": "model ID thinkingmachines/inkling; reasoning_effort=0.99; temperature=1; top_p=1; max_output_tokens=30000; standard error 2.23 pp; $0.127025/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20074,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"thinkingmachines/inkling\"].accuracy; Overall leaderboard rank 52"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V4.1 Flash",
          "lab": "DeepSeek",
          "scoreDisplay": "41.17%",
          "scoreNumeric": 41.173,
          "date": "2026-09-26",
          "config": "model ID deepseek/deepseek-v4.1-flash; reasoning_effort=high; max_output_tokens=384000; standard error 2.042 pp; $0.012450/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "deepseek-v4.1-flash",
          "scorePrinted": "41.17%",
          "variant": "model ID deepseek/deepseek-v4.1-flash; reasoning_effort=high; max_output_tokens=384000; standard error 2.042 pp; $0.012450/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20075,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4.1-flash\"].accuracy; Overall leaderboard rank 53"
          },
          "corroborating": []
        },
        {
          "model": "MiMo V2.6 Flash",
          "lab": "Xiaomi",
          "scoreDisplay": "41.06%",
          "scoreNumeric": 41.057,
          "date": "2026-09-26",
          "config": "model ID xiaomi/mimo-v2.6-flash; temperature=1; top_p=0.95; max_output_tokens=128000; standard error 2.032 pp; $0.002865/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mimo-v2.6-flash",
          "scorePrinted": "41.06%",
          "variant": "model ID xiaomi/mimo-v2.6-flash; temperature=1; top_p=0.95; max_output_tokens=128000; standard error 2.032 pp; $0.002865/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20076,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.6-flash\"].accuracy; Overall leaderboard rank 54"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.4 Nano",
          "lab": "OpenAI",
          "scoreDisplay": "41.03%",
          "scoreNumeric": 41.029,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.4-nano-2026-03-17; reasoning_effort=high; max_output_tokens=30000; standard error 2.256 pp; $0.000844/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.4-nano",
          "scorePrinted": "41.03%",
          "variant": "model ID openai/gpt-5.4-nano-2026-03-17; reasoning_effort=high; max_output_tokens=30000; standard error 2.256 pp; $0.000844/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20077,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.4-nano-2026-03-17\"].accuracy; Overall leaderboard rank 55"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.2",
          "lab": "zAI",
          "scoreDisplay": "40.77%",
          "scoreNumeric": 40.771,
          "date": "2026-09-26",
          "config": "model ID zai/glm-5.2; temperature=1; max_output_tokens=30000; standard error 2.166 pp; $0.045010/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "glm-5.2",
          "scorePrinted": "40.77%",
          "variant": "model ID zai/glm-5.2; temperature=1; max_output_tokens=30000; standard error 2.166 pp; $0.045010/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20078,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.2\"].accuracy; Overall leaderboard rank 56"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.8 Max",
          "lab": "Alibaba",
          "scoreDisplay": "40.67%",
          "scoreNumeric": 40.668,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.8-max; temperature=1; max_output_tokens=30000; standard error 2.029 pp; $0.120827/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen3.8-max",
          "scorePrinted": "40.67%",
          "variant": "model ID alibaba/qwen3.8-max; temperature=1; max_output_tokens=30000; standard error 2.029 pp; $0.120827/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20079,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.8-max\"].accuracy; Overall leaderboard rank 57"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4.5 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "40.57%",
          "scoreNumeric": 40.569,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-4-5-20250929; temperature=1; max_output_tokens=30000; standard error 1.995 pp; $0.042403/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-4.5",
          "scorePrinted": "40.57%",
          "variant": "model ID anthropic/claude-sonnet-4-5-20250929; temperature=1; max_output_tokens=30000; standard error 1.995 pp; $0.042403/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20080,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-5-20250929\"].accuracy; Overall leaderboard rank 58"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Preview (9/25) (Nonthinking)",
          "lab": "Google",
          "scoreDisplay": "40.54%",
          "scoreNumeric": 40.538,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-preview-09-2025; temperature=1; max_output_tokens=30000; standard error 1.932 pp; $0.003692/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-preview-9-25",
          "scorePrinted": "40.54%",
          "variant": "model ID google/gemini-2.5-flash-preview-09-2025; temperature=1; max_output_tokens=30000; standard error 1.932 pp; $0.003692/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20081,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-preview-09-2025\"].accuracy; Overall leaderboard rank 59"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V4",
          "lab": "DeepSeek",
          "scoreDisplay": "40.45%",
          "scoreNumeric": 40.455,
          "date": "2026-09-26",
          "config": "model ID deepseek/deepseek-v4-pro; reasoning_effort=max; max_output_tokens=128000; standard error 2.122 pp; $0.060710/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "deepseek-v4",
          "scorePrinted": "40.45%",
          "variant": "model ID deepseek/deepseek-v4-pro; reasoning_effort=max; max_output_tokens=128000; standard error 2.122 pp; $0.060710/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20082,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-pro\"].accuracy; Overall leaderboard rank 60"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash (7/17) (Thinking)",
          "lab": "Google",
          "scoreDisplay": "40.36%",
          "scoreNumeric": 40.357,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-thinking; temperature=1; max_output_tokens=30000; standard error 1.952 pp; $0.003661/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-7-17",
          "scorePrinted": "40.36%",
          "variant": "model ID google/gemini-2.5-flash-thinking; temperature=1; max_output_tokens=30000; standard error 1.952 pp; $0.003661/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20083,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-thinking\"].accuracy; Overall leaderboard rank 61"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Preview (9/25) (Thinking)",
          "lab": "Google",
          "scoreDisplay": "40.33%",
          "scoreNumeric": 40.33,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-preview-09-2025-thinking; temperature=1; max_output_tokens=30000; standard error 1.915 pp; $0.003653/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-preview-9-25",
          "scorePrinted": "40.33%",
          "variant": "model ID google/gemini-2.5-flash-preview-09-2025-thinking; temperature=1; max_output_tokens=30000; standard error 1.915 pp; $0.003653/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20084,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-preview-09-2025-thinking\"].accuracy; Overall leaderboard rank 62"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K2.6",
          "lab": "Moonshot AI",
          "scoreDisplay": "40.14%",
          "scoreNumeric": 40.142,
          "date": "2026-09-26",
          "config": "model ID kimi/kimi-k2.6; top_p=0.95; max_output_tokens=30000; standard error 2.041 pp; $0.041295/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "kimi-k2.6",
          "scorePrinted": "40.14%",
          "variant": "model ID kimi/kimi-k2.6; top_p=0.95; max_output_tokens=30000; standard error 2.041 pp; $0.041295/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20085,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k2.6\"].accuracy; Overall leaderboard rank 63"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K2.5",
          "lab": "Moonshot AI",
          "scoreDisplay": "39.32%",
          "scoreNumeric": 39.316,
          "date": "2026-09-26",
          "config": "model ID kimi/kimi-k2.5-thinking; top_p=0.95; max_output_tokens=30000; standard error 2.119 pp; $0.017275/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "kimi-k2.5",
          "scorePrinted": "39.32%",
          "variant": "model ID kimi/kimi-k2.5-thinking; top_p=0.95; max_output_tokens=30000; standard error 2.119 pp; $0.017275/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20086,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k2.5-thinking\"].accuracy; Overall leaderboard rank 64"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.7 Max",
          "lab": "Alibaba",
          "scoreDisplay": "38.75%",
          "scoreNumeric": 38.751,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.7-max; temperature=1; max_output_tokens=30000; standard error 2.196 pp; $0.042362/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3.7-max",
          "scorePrinted": "38.75%",
          "variant": "model ID alibaba/qwen3.7-max; temperature=1; max_output_tokens=30000; standard error 2.196 pp; $0.042362/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20087,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.7-max\"].accuracy; Overall leaderboard rank 65"
          },
          "corroborating": []
        },
        {
          "model": "Nemotron 3 Ultra",
          "lab": "NVIDIA",
          "scoreDisplay": "38.62%",
          "scoreNumeric": 38.621,
          "date": "2026-09-26",
          "config": "model ID nvidia/nemotron-3-ultra-550b-a55b; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.001 pp; source snapshot 2026-09-26; run date not published",
          "modelSlug": "nemotron-3-ultra",
          "scorePrinted": "38.62%",
          "variant": "model ID nvidia/nemotron-3-ultra-550b-a55b; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.001 pp; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20088,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"nvidia/nemotron-3-ultra-550b-a55b\"].accuracy; Overall leaderboard rank 66"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash (7/17) (Nonthinking)",
          "lab": "Google",
          "scoreDisplay": "38.42%",
          "scoreNumeric": 38.425,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash; temperature=1; max_output_tokens=30000; standard error 1.923 pp; $0.003698/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-7-17",
          "scorePrinted": "38.42%",
          "variant": "model ID google/gemini-2.5-flash; temperature=1; max_output_tokens=30000; standard error 1.923 pp; $0.003698/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20089,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash\"].accuracy; Overall leaderboard rank 67"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4",
          "lab": "SpaceXAI",
          "scoreDisplay": "38.08%",
          "scoreNumeric": 38.078,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-0709; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.206 pp; $0.034103/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4",
          "scorePrinted": "38.08%",
          "variant": "model ID grok/grok-4-0709; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.206 pp; $0.034103/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20090,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-0709\"].accuracy; Overall leaderboard rank 68"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.3",
          "lab": "SpaceXAI",
          "scoreDisplay": "38.07%",
          "scoreNumeric": 38.068,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.3; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.081 pp; $0.022202/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.3",
          "scorePrinted": "38.07%",
          "variant": "model ID grok/grok-4.3; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.081 pp; $0.022202/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20091,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.3\"].accuracy; Overall leaderboard rank 69"
          },
          "corroborating": []
        },
        {
          "model": "Inkling Small",
          "lab": "Thinking Machines",
          "scoreDisplay": "37.89%",
          "scoreNumeric": 37.893,
          "date": "2026-09-26",
          "config": "model ID thinkingmachines/inkling-small; reasoning_effort=0.99; temperature=1; top_p=1; max_output_tokens=30000; standard error 2.206 pp; $0.016656/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "inkling-small",
          "scorePrinted": "37.89%",
          "variant": "model ID thinkingmachines/inkling-small; reasoning_effort=0.99; temperature=1; top_p=1; max_output_tokens=30000; standard error 2.206 pp; $0.016656/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20092,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"thinkingmachines/inkling-small\"].accuracy; Overall leaderboard rank 70"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4 Fast (Reasoning)",
          "lab": "SpaceXAI",
          "scoreDisplay": "37.38%",
          "scoreNumeric": 37.385,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-fast-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.941 pp; $0.002143/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4-fast-reasoning",
          "scorePrinted": "37.38%",
          "variant": "model ID grok/grok-4-fast-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.941 pp; $0.002143/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20093,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-fast-reasoning\"].accuracy; Overall leaderboard rank 71"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.6 Plus",
          "lab": "Alibaba",
          "scoreDisplay": "36.89%",
          "scoreNumeric": 36.894,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.6-plus; temperature=1; max_output_tokens=30000; standard error 2.017 pp; $0.015673/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen3.6-plus",
          "scorePrinted": "36.89%",
          "variant": "model ID alibaba/qwen3.6-plus; temperature=1; max_output_tokens=30000; standard error 2.017 pp; $0.015673/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20094,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.6-plus\"].accuracy; Overall leaderboard rank 72"
          },
          "corroborating": []
        },
        {
          "model": "Llama 4 Maverick",
          "lab": "Meta",
          "scoreDisplay": "36.51%",
          "scoreNumeric": 36.514,
          "date": "2026-09-26",
          "config": "model ID fireworks/llama4-maverick-instruct-basic; temperature=1; max_output_tokens=30000; standard error 1.994 pp; $0.002888/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "llama-4-maverick",
          "scorePrinted": "36.51%",
          "variant": "model ID fireworks/llama4-maverick-instruct-basic; temperature=1; max_output_tokens=30000; standard error 1.994 pp; $0.002888/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20095,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"fireworks/llama4-maverick-instruct-basic\"].accuracy; Overall leaderboard rank 73"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "34.96%",
          "scoreNumeric": 34.959,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-4-20250514-thinking; max_output_tokens=30000; standard error 1.939 pp; $0.069896/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-4",
          "scorePrinted": "34.96%",
          "variant": "model ID anthropic/claude-sonnet-4-20250514-thinking; max_output_tokens=30000; standard error 1.939 pp; $0.069896/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20096,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-20250514-thinking\"].accuracy; Overall leaderboard rank 74"
          },
          "corroborating": []
        },
        {
          "model": "MiniMax-M2.7",
          "lab": "MiniMax",
          "scoreDisplay": "34.44%",
          "scoreNumeric": 34.44,
          "date": "2026-09-26",
          "config": "model ID minimax/MiniMax-M2.7; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.985 pp; $0.007424/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "minimax-m2.7",
          "scorePrinted": "34.44%",
          "variant": "model ID minimax/MiniMax-M2.7; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.985 pp; $0.007424/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20097,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M2.7\"].accuracy; Overall leaderboard rank 75"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Lite (9/25) (Thinking)",
          "lab": "Google",
          "scoreDisplay": "34.19%",
          "scoreNumeric": 34.191,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-lite-preview-09-2025-thinking; temperature=1; max_output_tokens=30000; standard error 1.736 pp; $0.001182/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-lite-9-25",
          "scorePrinted": "34.19%",
          "variant": "model ID google/gemini-2.5-flash-lite-preview-09-2025-thinking; temperature=1; max_output_tokens=30000; standard error 1.736 pp; $0.001182/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20098,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite-preview-09-2025-thinking\"].accuracy; Overall leaderboard rank 76"
          },
          "corroborating": []
        },
        {
          "model": "MiniMax-M2.1",
          "lab": "MiniMax",
          "scoreDisplay": "34.08%",
          "scoreNumeric": 34.083,
          "date": "2026-09-26",
          "config": "model ID minimax/MiniMax-M2.1; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.943 pp; source snapshot 2026-09-26; run date not published",
          "modelSlug": "minimax-m2.1",
          "scorePrinted": "34.08%",
          "variant": "model ID minimax/MiniMax-M2.1; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.943 pp; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20099,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M2.1\"].accuracy; Overall leaderboard rank 77"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "33.94%",
          "scoreNumeric": 33.943,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-4-20250514; temperature=1; max_output_tokens=30000; standard error 1.906 pp; $0.039460/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-4",
          "scorePrinted": "33.94%",
          "variant": "model ID anthropic/claude-sonnet-4-20250514; temperature=1; max_output_tokens=30000; standard error 1.906 pp; $0.039460/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20100,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-20250514\"].accuracy; Overall leaderboard rank 78"
          },
          "corroborating": []
        },
        {
          "model": "o4 Mini",
          "lab": "OpenAI",
          "scoreDisplay": "33.79%",
          "scoreNumeric": 33.791,
          "date": "2026-09-26",
          "config": "model ID openai/o4-mini-2025-04-16; reasoning_effort=high; max_output_tokens=30000; standard error 2.021 pp; $0.017605/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "o4-mini",
          "scorePrinted": "33.79%",
          "variant": "model ID openai/o4-mini-2025-04-16; reasoning_effort=high; max_output_tokens=30000; standard error 2.021 pp; $0.017605/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20101,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/o4-mini-2025-04-16\"].accuracy; Overall leaderboard rank 79"
          },
          "corroborating": []
        },
        {
          "model": "Mistral Medium 3.5",
          "lab": "Mistral",
          "scoreDisplay": "33.75%",
          "scoreNumeric": 33.752,
          "date": "2026-09-26",
          "config": "model ID mistralai/mistral-medium-3.5; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.148 pp; $0.053370/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mistral-medium-3.5",
          "scorePrinted": "33.75%",
          "variant": "model ID mistralai/mistral-medium-3.5; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.148 pp; $0.053370/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20102,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"mistralai/mistral-medium-3.5\"].accuracy; Overall leaderboard rank 80"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.5 Flash",
          "lab": "Alibaba",
          "scoreDisplay": "33.00%",
          "scoreNumeric": 32.997,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.5-flash; temperature=1; max_output_tokens=30000; standard error 1.787 pp; $0.003934/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3.5-flash",
          "scorePrinted": "33.00%",
          "variant": "model ID alibaba/qwen3.5-flash; temperature=1; max_output_tokens=30000; standard error 1.787 pp; $0.003934/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20103,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.5-flash\"].accuracy; Overall leaderboard rank 81"
          },
          "corroborating": []
        },
        {
          "model": "GLM 4.7",
          "lab": "zAI",
          "scoreDisplay": "32.77%",
          "scoreNumeric": 32.772,
          "date": "2026-09-26",
          "config": "model ID zai/glm-4.7; temperature=1; top_p=1; max_output_tokens=30000; standard error 1.996 pp; $0.006710/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "glm-4.7",
          "scorePrinted": "32.77%",
          "variant": "model ID zai/glm-4.7; temperature=1; top_p=1; max_output_tokens=30000; standard error 1.996 pp; $0.006710/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20104,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-4.7\"].accuracy; Overall leaderboard rank 82"
          },
          "corroborating": []
        },
        {
          "model": "Claude Haiku 4.5 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "32.68%",
          "scoreNumeric": 32.678,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-haiku-4-5-20251001-thinking; temperature=1; max_output_tokens=30000; standard error 1.998 pp; $0.020099/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-haiku-4.5",
          "scorePrinted": "32.68%",
          "variant": "model ID anthropic/claude-haiku-4-5-20251001-thinking; temperature=1; max_output_tokens=30000; standard error 1.998 pp; $0.020099/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20105,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-haiku-4-5-20251001-thinking\"].accuracy; Overall leaderboard rank 83"
          },
          "corroborating": []
        },
        {
          "model": "MiMo V2.5 Pro",
          "lab": "Xiaomi",
          "scoreDisplay": "32.48%",
          "scoreNumeric": 32.484,
          "date": "2026-09-26",
          "config": "model ID xiaomi/mimo-v2.5-pro; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.907 pp; $0.006719/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mimo-v2.5-pro",
          "scorePrinted": "32.48%",
          "variant": "model ID xiaomi/mimo-v2.5-pro; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.907 pp; $0.006719/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20106,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.5-pro\"].accuracy; Overall leaderboard rank 84"
          },
          "corroborating": []
        },
        {
          "model": "Ling 3.0 Flash",
          "lab": "Ant Group",
          "scoreDisplay": "32.27%",
          "scoreNumeric": 32.275,
          "date": "2026-09-26",
          "config": "model ID ant/ling-3.0-flash-2607; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.908 pp; $0.001645/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "ling-3.0-flash",
          "scorePrinted": "32.27%",
          "variant": "model ID ant/ling-3.0-flash-2607; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.908 pp; $0.001645/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20107,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"ant/ling-3.0-flash-2607\"].accuracy; Overall leaderboard rank 85"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.20 (Reasoning)",
          "lab": "SpaceXAI",
          "scoreDisplay": "32.16%",
          "scoreNumeric": 32.156,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.20-0309-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.124 pp; $0.036189/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.20-reasoning",
          "scorePrinted": "32.16%",
          "variant": "model ID grok/grok-4.20-0309-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.124 pp; $0.036189/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20108,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.20-0309-reasoning\"].accuracy; Overall leaderboard rank 86"
          },
          "corroborating": []
        },
        {
          "model": "MiMo V2.5",
          "lab": "Xiaomi",
          "scoreDisplay": "31.89%",
          "scoreNumeric": 31.895,
          "date": "2026-09-26",
          "config": "model ID xiaomi/mimo-v2.5; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.025 pp; $0.002162/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mimo-v2.5",
          "scorePrinted": "31.89%",
          "variant": "model ID xiaomi/mimo-v2.5; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.025 pp; $0.002162/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20109,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.5\"].accuracy; Overall leaderboard rank 87"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3 VL Plus",
          "lab": "Alibaba",
          "scoreDisplay": "31.65%",
          "scoreNumeric": 31.651,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3-vl-plus-2025-09-23; temperature=1; max_output_tokens=30000; standard error 1.845 pp; $0.002519/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3-vl-plus",
          "scorePrinted": "31.65%",
          "variant": "model ID alibaba/qwen3-vl-plus-2025-09-23; temperature=1; max_output_tokens=30000; standard error 1.845 pp; $0.002519/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20110,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3-vl-plus-2025-09-23\"].accuracy; Overall leaderboard rank 88"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3 Max Thinking",
          "lab": "Alibaba",
          "scoreDisplay": "31.37%",
          "scoreNumeric": 31.373,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3-max-2026-01-23; temperature=1; max_output_tokens=30000; standard error 1.888 pp; $0.014776/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3-max-thinking",
          "scorePrinted": "31.37%",
          "variant": "model ID alibaba/qwen3-max-2026-01-23; temperature=1; max_output_tokens=30000; standard error 1.888 pp; $0.014776/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20111,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3-max-2026-01-23\"].accuracy; Overall leaderboard rank 89"
          },
          "corroborating": []
        },
        {
          "model": "Mercury 2.5",
          "lab": "Inception",
          "scoreDisplay": "31.33%",
          "scoreNumeric": 31.326,
          "date": "2026-09-26",
          "config": "model ID inception/mercury-2.5; reasoning_effort=high; temperature=1; max_output_tokens=65536; standard error 1.953 pp; $0.004535/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mercury-2.5",
          "scorePrinted": "31.33%",
          "variant": "model ID inception/mercury-2.5; reasoning_effort=high; temperature=1; max_output_tokens=65536; standard error 1.953 pp; $0.004535/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20112,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"inception/mercury-2.5\"].accuracy; Overall leaderboard rank 90"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5 Nano",
          "lab": "OpenAI",
          "scoreDisplay": "30.44%",
          "scoreNumeric": 30.441,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5-nano-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 1.948 pp; $0.001729/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5-nano",
          "scorePrinted": "30.44%",
          "variant": "model ID openai/gpt-5-nano-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 1.948 pp; $0.001729/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20113,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-nano-2025-08-07\"].accuracy; Overall leaderboard rank 91"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4 Fast (Non-Reasoning)",
          "lab": "SpaceXAI",
          "scoreDisplay": "30.04%",
          "scoreNumeric": 30.036,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-fast-non-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.974 pp; $0.002149/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4-fast-non-reasoning",
          "scorePrinted": "30.04%",
          "variant": "model ID grok/grok-4-fast-non-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.974 pp; $0.002149/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20114,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-fast-non-reasoning\"].accuracy; Overall leaderboard rank 92"
          },
          "corroborating": []
        },
        {
          "model": "Ling 3.0 Flash Fin",
          "lab": "Ant Group",
          "scoreDisplay": "29.30%",
          "scoreNumeric": 29.305,
          "date": "2026-09-26",
          "config": "model ID ant/ling-3.0-flash-af-rc3; temperature=1; top_p=0.95; max_output_tokens=131072; standard error 1.943 pp; $0.001354/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "ling-3.0-flash-fin",
          "scorePrinted": "29.30%",
          "variant": "model ID ant/ling-3.0-flash-af-rc3; temperature=1; top_p=0.95; max_output_tokens=131072; standard error 1.943 pp; $0.001354/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20115,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"ant/ling-3.0-flash-af-rc3\"].accuracy; Overall leaderboard rank 93"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.8 27B",
          "lab": "Alibaba",
          "scoreDisplay": "28.70%",
          "scoreNumeric": 28.698,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.8-27b; reasoning_effort=xhigh; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.971 pp; $0.050145/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3.8-27b",
          "scorePrinted": "28.70%",
          "variant": "model ID alibaba/qwen3.8-27b; reasoning_effort=xhigh; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.971 pp; $0.050145/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20116,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.8-27b\"].accuracy; Overall leaderboard rank 94"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.1 Fast Non-Reasoning",
          "lab": "SpaceXAI",
          "scoreDisplay": "28.35%",
          "scoreNumeric": 28.349,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-1-fast-non-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.921 pp; $0.002193/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.1-fast-non-reasoning",
          "scorePrinted": "28.35%",
          "variant": "model ID grok/grok-4-1-fast-non-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.921 pp; $0.002193/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20117,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-1-fast-non-reasoning\"].accuracy; Overall leaderboard rank 95"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.1 Fast (Reasoning)",
          "lab": "SpaceXAI",
          "scoreDisplay": "28.08%",
          "scoreNumeric": 28.08,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-1-fast-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.992 pp; $0.002108/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.1-fast-reasoning",
          "scorePrinted": "28.08%",
          "variant": "model ID grok/grok-4-1-fast-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.992 pp; $0.002108/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20118,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-1-fast-reasoning\"].accuracy; Overall leaderboard rank 96"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Lite (Nonthinking)",
          "lab": "Google",
          "scoreDisplay": "27.11%",
          "scoreNumeric": 27.115,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-lite; temperature=1; max_output_tokens=30000; standard error 1.843 pp; $0.001342/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-lite",
          "scorePrinted": "27.11%",
          "variant": "model ID google/gemini-2.5-flash-lite; temperature=1; max_output_tokens=30000; standard error 1.843 pp; $0.001342/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20119,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite\"].accuracy; Overall leaderboard rank 97"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Lite (9/25) (Nonthinking)",
          "lab": "Google",
          "scoreDisplay": "27.08%",
          "scoreNumeric": 27.079,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-lite-preview-09-2025; temperature=1; max_output_tokens=30000; standard error 1.911 pp; $0.001440/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-lite-9-25",
          "scorePrinted": "27.08%",
          "variant": "model ID google/gemini-2.5-flash-lite-preview-09-2025; temperature=1; max_output_tokens=30000; standard error 1.911 pp; $0.001440/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20120,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite-preview-09-2025\"].accuracy; Overall leaderboard rank 98"
          },
          "corroborating": []
        },
        {
          "model": "Llama 4 Scout",
          "lab": "Meta",
          "scoreDisplay": "23.31%",
          "scoreNumeric": 23.311,
          "date": "2026-09-26",
          "config": "model ID together/meta-llama/Llama-4-Scout-17B-16E-Instruct; temperature=1; max_output_tokens=30000; standard error 1.749 pp; $0.002176/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "llama-4-scout",
          "scorePrinted": "23.31%",
          "variant": "model ID together/meta-llama/Llama-4-Scout-17B-16E-Instruct; temperature=1; max_output_tokens=30000; standard error 1.749 pp; $0.002176/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20121,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"together/meta-llama/Llama-4-Scout-17B-16E-Instruct\"].accuracy; Overall leaderboard rank 99"
          },
          "corroborating": []
        },
        {
          "model": "Laguna M.1",
          "lab": "Poolside",
          "scoreDisplay": "23.11%",
          "scoreNumeric": 23.106,
          "date": "2026-09-26",
          "config": "model ID poolside/laguna-m.1; temperature=1; max_output_tokens=30000; standard error 1.693 pp; $0.002973/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "laguna-m.1",
          "scorePrinted": "23.11%",
          "variant": "model ID poolside/laguna-m.1; temperature=1; max_output_tokens=30000; standard error 1.693 pp; $0.002973/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20122,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"poolside/laguna-m.1\"].accuracy; Overall leaderboard rank 100"
          },
          "corroborating": []
        },
        {
          "model": "Laguna XS.2",
          "lab": "Poolside",
          "scoreDisplay": "21.25%",
          "scoreNumeric": 21.251,
          "date": "2026-09-26",
          "config": "model ID poolside/laguna-xs.2; temperature=1; max_output_tokens=30000; standard error 1.703 pp; $0.001445/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "laguna-xs.2",
          "scorePrinted": "21.25%",
          "variant": "model ID poolside/laguna-xs.2; temperature=1; max_output_tokens=30000; standard error 1.703 pp; $0.001445/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20123,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"poolside/laguna-xs.2\"].accuracy; Overall leaderboard rank 101"
          },
          "corroborating": []
        },
        {
          "model": "Command A+",
          "lab": "Cohere",
          "scoreDisplay": "19.72%",
          "scoreNumeric": 19.718,
          "date": "2026-09-26",
          "config": "model ID cohere/command-a-plus-05-2026; temperature=1; top_p=0.95; max_output_tokens=64000; standard error 1.835 pp; $0.055695/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "command-a",
          "scorePrinted": "19.72%",
          "variant": "model ID cohere/command-a-plus-05-2026; temperature=1; top_p=0.95; max_output_tokens=64000; standard error 1.835 pp; $0.055695/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20124,
          "source": {
            "id": 22,
            "title": "Vals AI MedCode leaderboard",
            "url": "https://www.vals.ai/benchmarks/medcode",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"cohere/command-a-plus-05-2026\"].accuracy; Overall leaderboard rank 102"
          },
          "corroborating": []
        }
      ],
      "unitShort": "2,755 patient records",
      "publisherShort": "Vals AI",
      "sourceShort": "Vals AI MedCode leaderboard",
      "fieldSize": 102,
      "category": "documentation",
      "confidence": "verified",
      "officialUrl": "https://www.vals.ai/benchmarks/medcode",
      "paperUrl": null,
      "scaleKind": "percent",
      "summary": "ICD-10-CM coding of whole hospital stays from discharge summaries and notes, checked against professional coders.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "All 102 scored configurations checked against the first-party embedded table; source parameters captured per row. Updated 2026-09-26; individual run dates unavailable.",
        "urls": [
          "https://www.vals.ai/benchmarks/medcode"
        ]
      }
    },
    {
      "slug": "medscribe",
      "name": "MedScribe (Vals AI)",
      "publisher": "Vals AI (dataset with Protege)",
      "released": "2026-02",
      "unit": "100 rubric-scored SOAP-note cases",
      "scale": "percentage accuracy 0-100, higher better",
      "description": "Clinical documentation support: quality of SOAP notes generated from clinical visits, scored against rubrics for documentation quality and compliance.",
      "basis": "independent-run",
      "sourceName": "Vals AI MedScribe leaderboard",
      "sourceUrl": "https://www.vals.ai/benchmarks/medscribe",
      "ours": null,
      "lastUpdate": "2026-09-26",
      "notes": "Vals AI runs this benchmark. Full Overall data contains 104 scored model configurations in the 2026-09-26 snapshot, including older and reasoning variants. Scores use the published percent scale; numeric values retain source precision and displayed values round to two decimals. Snapshot update dates are not model measurement dates.",
      "results": [
        {
          "model": "Claude Opus 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "91.43%",
          "scoreNumeric": 91.43,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-5-5; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 1.932 pp; $1.154156/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-5.5",
          "scorePrinted": "91.43%",
          "variant": "model ID anthropic/claude-opus-5-5; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 1.932 pp; $1.154156/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20125,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-5-5\"].accuracy; Overall leaderboard rank 1"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5.1",
          "lab": "Anthropic",
          "scoreDisplay": "91.29%",
          "scoreNumeric": 91.294,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-fable-5-1; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 1.953 pp; $0.963500/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-fable-5.1",
          "scorePrinted": "91.29%",
          "variant": "model ID anthropic/claude-fable-5-1; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 1.953 pp; $0.963500/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 354,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-fable-5-1\"].accuracy; Overall leaderboard rank 2"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "91.10%",
          "scoreNumeric": 91.101,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-5-5; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 1.96 pp; $0.508604/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-5.5",
          "scorePrinted": "91.10%",
          "variant": "model ID anthropic/claude-sonnet-5-5; compute_effort=max; temperature=1; max_output_tokens=128000; standard error 1.96 pp; $0.508604/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20126,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-5-5\"].accuracy; Overall leaderboard rank 3"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "90.98%",
          "scoreNumeric": 90.985,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.916 pp; $0.236275/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-5",
          "scorePrinted": "90.98%",
          "variant": "model ID anthropic/claude-opus-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.916 pp; $0.236275/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 150,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-5\"].accuracy; Overall leaderboard rank 4"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark 1.2",
          "lab": "Meta",
          "scoreDisplay": "90.06%",
          "scoreNumeric": 90.062,
          "date": "2026-09-26",
          "config": "model ID meta/muse_spark_1_2; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 1.958 pp; $0.037779/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "muse-spark-1.2",
          "scorePrinted": "90.06%",
          "variant": "model ID meta/muse_spark_1_2; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 1.958 pp; $0.037779/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 355,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark_1_2\"].accuracy; Overall leaderboard rank 5"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.7",
          "lab": "SpaceXAI",
          "scoreDisplay": "89.38%",
          "scoreNumeric": 89.377,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.7; reasoning_effort=xhigh; temperature=1; top_p=0.95; standard error 1.886 pp; $0.074505/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.7",
          "scorePrinted": "89.38%",
          "variant": "model ID grok/grok-4.7; reasoning_effort=xhigh; temperature=1; top_p=0.95; standard error 1.886 pp; $0.074505/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20127,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.7\"].accuracy; Overall leaderboard rank 6"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.3 Flash",
          "lab": "zAI",
          "scoreDisplay": "88.94%",
          "scoreNumeric": 88.936,
          "date": "2026-09-26",
          "config": "model ID zai/glm-5.3-flash; reasoning_effort=max; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.907 pp; $0.002235/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "glm-5.3-flash",
          "scorePrinted": "88.94%",
          "variant": "model ID zai/glm-5.3-flash; reasoning_effort=max; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.907 pp; $0.002235/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20128,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.3-flash\"].accuracy; Overall leaderboard rank 7"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark 1.1",
          "lab": "Meta",
          "scoreDisplay": "88.89%",
          "scoreNumeric": 88.888,
          "date": "2026-09-26",
          "config": "model ID meta/muse_spark_1_1; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 1.95 pp; $0.034628/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "muse-spark-1.1",
          "scorePrinted": "88.89%",
          "variant": "model ID meta/muse_spark_1_1; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 1.95 pp; $0.034628/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 356,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark_1_1\"].accuracy; Overall leaderboard rank 8"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.3",
          "lab": "zAI",
          "scoreDisplay": "88.81%",
          "scoreNumeric": 88.81,
          "date": "2026-09-26",
          "config": "model ID zai/glm-5.3; reasoning_effort=max; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.999 pp; $0.057236/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "glm-5.3",
          "scorePrinted": "88.81%",
          "variant": "model ID zai/glm-5.3; reasoning_effort=max; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.999 pp; $0.057236/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20129,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.3\"].accuracy; Overall leaderboard rank 9"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "88.52%",
          "scoreNumeric": 88.522,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-fable-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.945 pp; $0.583239/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-fable-5",
          "scorePrinted": "88.52%",
          "variant": "model ID anthropic/claude-fable-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.945 pp; $0.583239/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 357,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-fable-5\"].accuracy; Overall leaderboard rank 10"
          },
          "corroborating": []
        },
        {
          "model": "MiMo V2.6 Pro",
          "lab": "Xiaomi",
          "scoreDisplay": "88.31%",
          "scoreNumeric": 88.307,
          "date": "2026-09-26",
          "config": "model ID xiaomi/mimo-v2.6-pro; temperature=1; top_p=0.95; max_output_tokens=128000; standard error 1.938 pp; $0.009880/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mimo-v2.6-pro",
          "scorePrinted": "88.31%",
          "variant": "model ID xiaomi/mimo-v2.6-pro; temperature=1; top_p=0.95; max_output_tokens=128000; standard error 1.938 pp; $0.009880/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20130,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.6-pro\"].accuracy; Overall leaderboard rank 11"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.1",
          "lab": "OpenAI",
          "scoreDisplay": "88.09%",
          "scoreNumeric": 88.09,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.1-2025-11-13; reasoning_effort=high; max_output_tokens=30000; standard error 1.942 pp; $0.096508/test; source snapshot 2026-09-26; run date not published",
          "alias": "GPT-5.1",
          "modelSlug": "gpt-5.1",
          "scorePrinted": "88.09%",
          "variant": "model ID openai/gpt-5.1-2025-11-13; reasoning_effort=high; max_output_tokens=30000; standard error 1.942 pp; $0.096508/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 358,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.1-2025-11-13\"].accuracy; Overall leaderboard rank 12"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K3",
          "lab": "Moonshot AI",
          "scoreDisplay": "87.96%",
          "scoreNumeric": 87.958,
          "date": "2026-09-26",
          "config": "model ID kimi/kimi-k3; temperature=1; max_output_tokens=30000; standard error 1.891 pp; $0.118005/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "kimi-k3",
          "scorePrinted": "87.96%",
          "variant": "model ID kimi/kimi-k3; temperature=1; max_output_tokens=30000; standard error 1.891 pp; $0.118005/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 359,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k3\"].accuracy; Overall leaderboard rank 13"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Astra",
          "lab": "OpenAI",
          "scoreDisplay": "87.91%",
          "scoreNumeric": 87.908,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-6-astra; reasoning_effort=max; max_output_tokens=128000; standard error 1.938 pp; $0.581991/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-6-astra",
          "scorePrinted": "87.91%",
          "variant": "model ID openai/gpt-6-astra; reasoning_effort=max; max_output_tokens=128000; standard error 1.938 pp; $0.581991/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 567,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-astra\"].accuracy; Overall leaderboard rank 14"
          },
          "corroborating": []
        },
        {
          "model": "MiniMax-M3",
          "lab": "MiniMax",
          "scoreDisplay": "87.25%",
          "scoreNumeric": 87.252,
          "date": "2026-09-26",
          "config": "model ID minimax/MiniMax-M3; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.957 pp; $0.013748/test; source snapshot 2026-09-26; run date not published",
          "alias": "MiniMax M3",
          "modelSlug": "minimax-m3",
          "scorePrinted": "87.25%",
          "variant": "model ID minimax/MiniMax-M3; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.957 pp; $0.013748/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 360,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M3\"].accuracy; Overall leaderboard rank 15"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.5",
          "lab": "SpaceXAI",
          "scoreDisplay": "86.88%",
          "scoreNumeric": 86.884,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.5; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.944 pp; $0.033208/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.5",
          "scorePrinted": "86.88%",
          "variant": "model ID grok/grok-4.5; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.944 pp; $0.033208/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20131,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.5\"].accuracy; Overall leaderboard rank 16"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.5",
          "lab": "OpenAI",
          "scoreDisplay": "86.87%",
          "scoreNumeric": 86.868,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.5; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 1.932 pp; $0.142988/test; source snapshot 2026-09-26; run date not published",
          "alias": "GPT-5.5",
          "modelSlug": "gpt-5.5",
          "scorePrinted": "86.87%",
          "variant": "model ID openai/gpt-5.5; reasoning_effort=xhigh; temperature=1; max_output_tokens=30000; standard error 1.932 pp; $0.142988/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 361,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.5\"].accuracy; Overall leaderboard rank 17"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.6 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "86.74%",
          "scoreNumeric": 86.738,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-6; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.942 pp; $0.115121/test; source snapshot 2026-09-26; run date not published",
          "alias": "Claude Opus 4.6",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "86.74%",
          "variant": "model ID anthropic/claude-opus-4-6; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.942 pp; $0.115121/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 362,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-6\"].accuracy; Overall leaderboard rank 18"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.6",
          "lab": "xAI",
          "scoreDisplay": "86.53%",
          "scoreNumeric": 86.534,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.6; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.956 pp; $0.037190/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.6",
          "scorePrinted": "86.53%",
          "variant": "model ID grok/grok-4.6; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.956 pp; $0.037190/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "official leaderboard",
          "confidence": "verified",
          "resultId": 363,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.6\"].accuracy; Overall leaderboard rank 19"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.6 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "86.13%",
          "scoreNumeric": 86.13,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-6-thinking; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.944 pp; $0.224735/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "86.13%",
          "variant": "model ID anthropic/claude-opus-4-6-thinking; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.944 pp; $0.224735/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20132,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-6-thinking\"].accuracy; Overall leaderboard rank 20"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark",
          "lab": "Meta",
          "scoreDisplay": "85.90%",
          "scoreNumeric": 85.902,
          "date": "2026-09-26",
          "config": "model ID meta/muse_spark; temperature=1; max_output_tokens=30000; standard error 1.847 pp; $0.007681/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "muse-spark",
          "scorePrinted": "85.90%",
          "variant": "model ID meta/muse_spark; temperature=1; max_output_tokens=30000; standard error 1.847 pp; $0.007681/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20133,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark\"].accuracy; Overall leaderboard rank 21"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.8",
          "lab": "Anthropic",
          "scoreDisplay": "85.75%",
          "scoreNumeric": 85.755,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-8; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.928 pp; $0.259121/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.8",
          "scorePrinted": "85.75%",
          "variant": "model ID anthropic/claude-opus-4-8; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.928 pp; $0.259121/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20134,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-8\"].accuracy; Overall leaderboard rank 22"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V4.1 Flash",
          "lab": "DeepSeek",
          "scoreDisplay": "85.50%",
          "scoreNumeric": 85.5,
          "date": "2026-09-26",
          "config": "model ID deepseek/deepseek-v4.1-flash; reasoning_effort=high; max_output_tokens=384000; standard error 1.918 pp; $0.015397/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "deepseek-v4.1-flash",
          "scorePrinted": "85.50%",
          "variant": "model ID deepseek/deepseek-v4.1-flash; reasoning_effort=high; max_output_tokens=384000; standard error 1.918 pp; $0.015397/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20135,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4.1-flash\"].accuracy; Overall leaderboard rank 23"
          },
          "corroborating": []
        },
        {
          "model": "Inkling",
          "lab": "Thinking Machines",
          "scoreDisplay": "85.41%",
          "scoreNumeric": 85.405,
          "date": "2026-09-26",
          "config": "model ID thinkingmachines/inkling; reasoning_effort=0.99; temperature=1; top_p=1; max_output_tokens=30000; standard error 1.844 pp; $0.165561/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "inkling",
          "scorePrinted": "85.41%",
          "variant": "model ID thinkingmachines/inkling; reasoning_effort=0.99; temperature=1; top_p=1; max_output_tokens=30000; standard error 1.844 pp; $0.165561/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20136,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"thinkingmachines/inkling\"].accuracy; Overall leaderboard rank 24"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.5 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "85.32%",
          "scoreNumeric": 85.321,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-5-20251101-thinking; compute_effort=high; temperature=1; max_output_tokens=30000; standard error 1.896 pp; $0.410224/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.5",
          "scorePrinted": "85.32%",
          "variant": "model ID anthropic/claude-opus-4-5-20251101-thinking; compute_effort=high; temperature=1; max_output_tokens=30000; standard error 1.896 pp; $0.410224/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20137,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-5-20251101-thinking\"].accuracy; Overall leaderboard rank 25"
          },
          "corroborating": []
        },
        {
          "model": "MiMo V2.6 Flash",
          "lab": "Xiaomi",
          "scoreDisplay": "85.28%",
          "scoreNumeric": 85.275,
          "date": "2026-09-26",
          "config": "model ID xiaomi/mimo-v2.6-flash; temperature=1; top_p=0.95; max_output_tokens=128000; standard error 1.986 pp; $0.002184/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mimo-v2.6-flash",
          "scorePrinted": "85.28%",
          "variant": "model ID xiaomi/mimo-v2.6-flash; temperature=1; top_p=0.95; max_output_tokens=128000; standard error 1.986 pp; $0.002184/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20138,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.6-flash\"].accuracy; Overall leaderboard rank 26"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "85.23%",
          "scoreNumeric": 85.233,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.6-sol; reasoning_effort=max; max_output_tokens=30000; standard error 1.973 pp; $0.276691/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "85.23%",
          "variant": "model ID openai/gpt-5.6-sol; reasoning_effort=max; max_output_tokens=30000; standard error 1.973 pp; $0.276691/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20139,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-sol\"].accuracy; Overall leaderboard rank 27"
          },
          "corroborating": []
        },
        {
          "model": "Claude Haiku 4.5 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "85.23%",
          "scoreNumeric": 85.23,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-haiku-4-5-20251001-thinking; temperature=1; max_output_tokens=30000; standard error 1.899 pp; $0.042375/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-haiku-4.5",
          "scorePrinted": "85.23%",
          "variant": "model ID anthropic/claude-haiku-4-5-20251001-thinking; temperature=1; max_output_tokens=30000; standard error 1.899 pp; $0.042375/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20140,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-haiku-4-5-20251001-thinking\"].accuracy; Overall leaderboard rank 28"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.8 Max",
          "lab": "Alibaba",
          "scoreDisplay": "84.95%",
          "scoreNumeric": 84.947,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.8-max; temperature=1; max_output_tokens=30000; standard error 1.999 pp; $0.089616/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen3.8-max",
          "scorePrinted": "84.95%",
          "variant": "model ID alibaba/qwen3.8-max; temperature=1; max_output_tokens=30000; standard error 1.999 pp; $0.089616/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20141,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.8-max\"].accuracy; Overall leaderboard rank 29"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4.5 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "84.52%",
          "scoreNumeric": 84.515,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-4-5-20250929; temperature=1; max_output_tokens=30000; standard error 1.929 pp; $0.054649/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-4.5",
          "scorePrinted": "84.52%",
          "variant": "model ID anthropic/claude-sonnet-4-5-20250929; temperature=1; max_output_tokens=30000; standard error 1.929 pp; $0.054649/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20142,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-5-20250929\"].accuracy; Overall leaderboard rank 30"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.8 Flash",
          "lab": "Google",
          "scoreDisplay": "84.50%",
          "scoreNumeric": 84.496,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.8-flash; reasoning_effort=high; temperature=1; max_output_tokens=65536; standard error 1.943 pp; $0.025238/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.8-flash",
          "scorePrinted": "84.50%",
          "variant": "model ID google/gemini-3.8-flash; reasoning_effort=high; temperature=1; max_output_tokens=65536; standard error 1.943 pp; $0.025238/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20143,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.8-flash\"].accuracy; Overall leaderboard rank 31"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "84.39%",
          "scoreNumeric": 84.391,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.6-luna; reasoning_effort=max; max_output_tokens=30000; standard error 2.585 pp; $0.022813/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.6-luna",
          "scorePrinted": "84.39%",
          "variant": "model ID openai/gpt-5.6-luna; reasoning_effort=max; max_output_tokens=30000; standard error 2.585 pp; $0.022813/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20144,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-luna\"].accuracy; Overall leaderboard rank 32"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.2",
          "lab": "OpenAI",
          "scoreDisplay": "84.39%",
          "scoreNumeric": 84.387,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.2-2025-12-11; reasoning_effort=xhigh; max_output_tokens=30000; standard error 1.856 pp; $0.115422/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.2",
          "scorePrinted": "84.39%",
          "variant": "model ID openai/gpt-5.2-2025-12-11; reasoning_effort=xhigh; max_output_tokens=30000; standard error 1.856 pp; $0.115422/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20145,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.2-2025-12-11\"].accuracy; Overall leaderboard rank 33"
          },
          "corroborating": []
        },
        {
          "model": "Inkling Small",
          "lab": "Thinking Machines",
          "scoreDisplay": "84.11%",
          "scoreNumeric": 84.112,
          "date": "2026-09-26",
          "config": "model ID thinkingmachines/inkling-small; reasoning_effort=0.99; temperature=1; top_p=1; max_output_tokens=30000; standard error 1.87 pp; $0.019018/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "inkling-small",
          "scorePrinted": "84.11%",
          "variant": "model ID thinkingmachines/inkling-small; reasoning_effort=0.99; temperature=1; top_p=1; max_output_tokens=30000; standard error 1.87 pp; $0.019018/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20146,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"thinkingmachines/inkling-small\"].accuracy; Overall leaderboard rank 34"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4.5 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "84.10%",
          "scoreNumeric": 84.101,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-4-5-20250929-thinking; temperature=1; max_output_tokens=30000; standard error 1.873 pp; $0.082281/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-4.5",
          "scorePrinted": "84.10%",
          "variant": "model ID anthropic/claude-sonnet-4-5-20250929-thinking; temperature=1; max_output_tokens=30000; standard error 1.873 pp; $0.082281/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20147,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-5-20250929-thinking\"].accuracy; Overall leaderboard rank 35"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.7 Flash",
          "lab": "Google",
          "scoreDisplay": "83.94%",
          "scoreNumeric": 83.942,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.7-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.004 pp; $0.058736/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.7-flash",
          "scorePrinted": "83.94%",
          "variant": "model ID google/gemini-3.7-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.004 pp; $0.058736/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20148,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.7-flash\"].accuracy; Overall leaderboard rank 36"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.8 27B",
          "lab": "Alibaba",
          "scoreDisplay": "83.85%",
          "scoreNumeric": 83.849,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.8-27b; reasoning_effort=xhigh; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.977 pp; $0.046732/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3.8-27b",
          "scorePrinted": "83.85%",
          "variant": "model ID alibaba/qwen3.8-27b; reasoning_effort=xhigh; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.977 pp; $0.046732/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20149,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.8-27b\"].accuracy; Overall leaderboard rank 37"
          },
          "corroborating": []
        },
        {
          "model": "MiMo V2.5 Pro",
          "lab": "Xiaomi",
          "scoreDisplay": "83.73%",
          "scoreNumeric": 83.73,
          "date": "2026-09-26",
          "config": "model ID xiaomi/mimo-v2.5-pro; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.063 pp; $0.006030/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mimo-v2.5-pro",
          "scorePrinted": "83.73%",
          "variant": "model ID xiaomi/mimo-v2.5-pro; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.063 pp; $0.006030/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20150,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.5-pro\"].accuracy; Overall leaderboard rank 38"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Luna",
          "lab": "OpenAI",
          "scoreDisplay": "83.71%",
          "scoreNumeric": 83.71,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-6-luna; reasoning_effort=max; max_output_tokens=128000; standard error 1.949 pp; $0.009562/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-6-luna",
          "scorePrinted": "83.71%",
          "variant": "model ID openai/gpt-6-luna; reasoning_effort=max; max_output_tokens=128000; standard error 1.949 pp; $0.009562/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20151,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-luna\"].accuracy; Overall leaderboard rank 39"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5",
          "lab": "OpenAI",
          "scoreDisplay": "83.65%",
          "scoreNumeric": 83.65,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 1.936 pp; $0.101500/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5",
          "scorePrinted": "83.65%",
          "variant": "model ID openai/gpt-5-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 1.936 pp; $0.101500/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20152,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-2025-08-07\"].accuracy; Overall leaderboard rank 40"
          },
          "corroborating": []
        },
        {
          "model": "Hy4 Preview",
          "lab": "Tencent",
          "scoreDisplay": "83.60%",
          "scoreNumeric": 83.6,
          "date": "2026-09-26",
          "config": "model ID tencent/hy4-preview; temperature=1; top_p=1; max_output_tokens=64000; standard error 2.065 pp; $0.053243/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "hy4-preview",
          "scorePrinted": "83.60%",
          "variant": "model ID tencent/hy4-preview; temperature=1; top_p=1; max_output_tokens=64000; standard error 2.065 pp; $0.053243/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20153,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"tencent/hy4-preview\"].accuracy; Overall leaderboard rank 41"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.2",
          "lab": "zAI",
          "scoreDisplay": "83.53%",
          "scoreNumeric": 83.534,
          "date": "2026-09-26",
          "config": "model ID zai/glm-5.2; temperature=1; max_output_tokens=30000; standard error 2.002 pp; $0.044912/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "glm-5.2",
          "scorePrinted": "83.53%",
          "variant": "model ID zai/glm-5.2; temperature=1; max_output_tokens=30000; standard error 2.002 pp; $0.044912/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20154,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.2\"].accuracy; Overall leaderboard rank 42"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.5 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "83.25%",
          "scoreNumeric": 83.246,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-5-20251101; compute_effort=high; temperature=1; max_output_tokens=30000; standard error 1.926 pp; $0.281674/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.5",
          "scorePrinted": "83.25%",
          "variant": "model ID anthropic/claude-opus-4-5-20251101; compute_effort=high; temperature=1; max_output_tokens=30000; standard error 1.926 pp; $0.281674/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20155,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-5-20251101\"].accuracy; Overall leaderboard rank 43"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash (7/17) (Thinking)",
          "lab": "Google",
          "scoreDisplay": "82.98%",
          "scoreNumeric": 82.983,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-thinking; temperature=1; max_output_tokens=30000; standard error 1.908 pp; $0.014824/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-7-17",
          "scorePrinted": "82.98%",
          "variant": "model ID google/gemini-2.5-flash-thinking; temperature=1; max_output_tokens=30000; standard error 1.908 pp; $0.014824/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20156,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-thinking\"].accuracy; Overall leaderboard rank 44"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.7",
          "lab": "Anthropic",
          "scoreDisplay": "82.95%",
          "scoreNumeric": 82.953,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-7; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.977 pp; $0.177841/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.7",
          "scorePrinted": "82.95%",
          "variant": "model ID anthropic/claude-opus-4-7; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 1.977 pp; $0.177841/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20157,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-7\"].accuracy; Overall leaderboard rank 45"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash (7/17) (Nonthinking)",
          "lab": "Google",
          "scoreDisplay": "82.87%",
          "scoreNumeric": 82.869,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash; temperature=1; max_output_tokens=30000; standard error 1.909 pp; $0.014869/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-7-17",
          "scorePrinted": "82.87%",
          "variant": "model ID google/gemini-2.5-flash; temperature=1; max_output_tokens=30000; standard error 1.909 pp; $0.014869/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20158,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash\"].accuracy; Overall leaderboard rank 46"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Terra",
          "lab": "OpenAI",
          "scoreDisplay": "82.87%",
          "scoreNumeric": 82.869,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.6-terra; reasoning_effort=xhigh; max_output_tokens=30000; standard error 1.948 pp; $0.062588/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.6-terra",
          "scorePrinted": "82.87%",
          "variant": "model ID openai/gpt-5.6-terra; reasoning_effort=xhigh; max_output_tokens=30000; standard error 1.948 pp; $0.062588/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20159,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-terra\"].accuracy; Overall leaderboard rank 47"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "82.03%",
          "scoreNumeric": 82.034,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-6-sol; reasoning_effort=max; max_output_tokens=128000; standard error 1.942 pp; $0.083138/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-6-sol",
          "scorePrinted": "82.03%",
          "variant": "model ID openai/gpt-6-sol; reasoning_effort=max; max_output_tokens=128000; standard error 1.942 pp; $0.083138/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20160,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-sol\"].accuracy; Overall leaderboard rank 48"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4 Fast (Reasoning)",
          "lab": "SpaceXAI",
          "scoreDisplay": "81.63%",
          "scoreNumeric": 81.632,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-fast-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.137 pp; $0.002535/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4-fast-reasoning",
          "scorePrinted": "81.63%",
          "variant": "model ID grok/grok-4-fast-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.137 pp; $0.002535/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20161,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-fast-reasoning\"].accuracy; Overall leaderboard rank 49"
          },
          "corroborating": []
        },
        {
          "model": "Ling 3.0 Flash",
          "lab": "Ant Group",
          "scoreDisplay": "80.90%",
          "scoreNumeric": 80.904,
          "date": "2026-09-26",
          "config": "model ID ant/ling-3.0-flash-2607; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.039 pp; $0.001358/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "ling-3.0-flash",
          "scorePrinted": "80.90%",
          "variant": "model ID ant/ling-3.0-flash-2607; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.039 pp; $0.001358/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20162,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"ant/ling-3.0-flash-2607\"].accuracy; Overall leaderboard rank 50"
          },
          "corroborating": []
        },
        {
          "model": "MiniMax-M2.1",
          "lab": "MiniMax",
          "scoreDisplay": "80.78%",
          "scoreNumeric": 80.777,
          "date": "2026-09-26",
          "config": "model ID minimax/MiniMax-M2.1; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.831 pp; $0.005087/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "minimax-m2.1",
          "scorePrinted": "80.78%",
          "variant": "model ID minimax/MiniMax-M2.1; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.831 pp; $0.005087/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20163,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M2.1\"].accuracy; Overall leaderboard rank 51"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5 Mini",
          "lab": "OpenAI",
          "scoreDisplay": "80.58%",
          "scoreNumeric": 80.577,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5-mini-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 1.924 pp; $0.033478/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5-mini",
          "scorePrinted": "80.58%",
          "variant": "model ID openai/gpt-5-mini-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 1.924 pp; $0.033478/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20164,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-mini-2025-08-07\"].accuracy; Overall leaderboard rank 52"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V4 Flash 0731",
          "lab": "DeepSeek",
          "scoreDisplay": "80.36%",
          "scoreNumeric": 80.363,
          "date": "2026-09-26",
          "config": "model ID deepseek/deepseek-v4-flash-0731; reasoning_effort=high; max_output_tokens=30000; standard error 1.973 pp; $0.014247/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "deepseek-v4-flash-0731",
          "scorePrinted": "80.36%",
          "variant": "model ID deepseek/deepseek-v4-flash-0731; reasoning_effort=high; max_output_tokens=30000; standard error 1.973 pp; $0.014247/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20165,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-flash-0731\"].accuracy; Overall leaderboard rank 53"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V4 Pro 0813",
          "lab": "DeepSeek",
          "scoreDisplay": "80.17%",
          "scoreNumeric": 80.174,
          "date": "2026-09-26",
          "config": "model ID deepseek/deepseek-v4-pro-0813; reasoning_effort=max; max_output_tokens=30000; standard error 2.004 pp; $0.041127/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "deepseek-v4-pro-0813",
          "scorePrinted": "80.17%",
          "variant": "model ID deepseek/deepseek-v4-pro-0813; reasoning_effort=max; max_output_tokens=30000; standard error 2.004 pp; $0.041127/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20166,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-pro-0813\"].accuracy; Overall leaderboard rank 54"
          },
          "corroborating": []
        },
        {
          "model": "MiniMax-M2.7",
          "lab": "MiniMax",
          "scoreDisplay": "79.87%",
          "scoreNumeric": 79.867,
          "date": "2026-09-26",
          "config": "model ID minimax/MiniMax-M2.7; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.86 pp; $0.005124/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "minimax-m2.7",
          "scorePrinted": "79.87%",
          "variant": "model ID minimax/MiniMax-M2.7; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.86 pp; $0.005124/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20167,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M2.7\"].accuracy; Overall leaderboard rank 55"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4 Fast (Non-Reasoning)",
          "lab": "SpaceXAI",
          "scoreDisplay": "79.72%",
          "scoreNumeric": 79.722,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-fast-non-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.871 pp; $0.002056/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4-fast-non-reasoning",
          "scorePrinted": "79.72%",
          "variant": "model ID grok/grok-4-fast-non-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.871 pp; $0.002056/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20168,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-fast-non-reasoning\"].accuracy; Overall leaderboard rank 56"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.6 Flash",
          "lab": "Google",
          "scoreDisplay": "79.66%",
          "scoreNumeric": 79.661,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.6-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.861 pp; $0.073178/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.6-flash",
          "scorePrinted": "79.66%",
          "variant": "model ID google/gemini-3.6-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.861 pp; $0.073178/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20169,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.6-flash\"].accuracy; Overall leaderboard rank 57"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.7 Max",
          "lab": "Alibaba",
          "scoreDisplay": "79.40%",
          "scoreNumeric": 79.396,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.7-max; temperature=1; max_output_tokens=30000; standard error 1.907 pp; $0.069070/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3.7-max",
          "scorePrinted": "79.40%",
          "variant": "model ID alibaba/qwen3.7-max; temperature=1; max_output_tokens=30000; standard error 1.907 pp; $0.069070/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20170,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.7-max\"].accuracy; Overall leaderboard rank 58"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.1 Fast (Reasoning)",
          "lab": "SpaceXAI",
          "scoreDisplay": "78.73%",
          "scoreNumeric": 78.732,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-1-fast-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.866 pp; $0.002387/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.1-fast-reasoning",
          "scorePrinted": "78.73%",
          "variant": "model ID grok/grok-4-1-fast-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.866 pp; $0.002387/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20171,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-1-fast-reasoning\"].accuracy; Overall leaderboard rank 59"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Preview (9/25) (Thinking)",
          "lab": "Google",
          "scoreDisplay": "78.50%",
          "scoreNumeric": 78.497,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-preview-09-2025-thinking; temperature=1; max_output_tokens=30000; standard error 1.993 pp; $0.014526/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-preview-9-25",
          "scorePrinted": "78.50%",
          "variant": "model ID google/gemini-2.5-flash-preview-09-2025-thinking; temperature=1; max_output_tokens=30000; standard error 1.993 pp; $0.014526/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20172,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-preview-09-2025-thinking\"].accuracy; Overall leaderboard rank 60"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4",
          "lab": "SpaceXAI",
          "scoreDisplay": "78.15%",
          "scoreNumeric": 78.152,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-0709; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.084 pp; $0.063955/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4",
          "scorePrinted": "78.15%",
          "variant": "model ID grok/grok-4-0709; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.084 pp; $0.063955/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20173,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-0709\"].accuracy; Overall leaderboard rank 61"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K2.6",
          "lab": "Moonshot AI",
          "scoreDisplay": "78.15%",
          "scoreNumeric": 78.149,
          "date": "2026-09-26",
          "config": "model ID kimi/kimi-k2.6; top_p=0.95; max_output_tokens=30000; standard error 1.792 pp; $0.055962/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "kimi-k2.6",
          "scorePrinted": "78.15%",
          "variant": "model ID kimi/kimi-k2.6; top_p=0.95; max_output_tokens=30000; standard error 1.792 pp; $0.055962/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20174,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k2.6\"].accuracy; Overall leaderboard rank 62"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Preview (9/25) (Nonthinking)",
          "lab": "Google",
          "scoreDisplay": "77.95%",
          "scoreNumeric": 77.946,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-preview-09-2025; temperature=1; max_output_tokens=30000; standard error 1.923 pp; $0.014385/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-preview-9-25",
          "scorePrinted": "77.95%",
          "variant": "model ID google/gemini-2.5-flash-preview-09-2025; temperature=1; max_output_tokens=30000; standard error 1.923 pp; $0.014385/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20175,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-preview-09-2025\"].accuracy; Overall leaderboard rank 63"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.4 (xhigh)",
          "lab": "OpenAI",
          "scoreDisplay": "77.55%",
          "scoreNumeric": 77.549,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.4-2026-03-05; reasoning_effort=xhigh; max_output_tokens=30000; standard error 3.316 pp; $0.639282/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.4-xhigh",
          "scorePrinted": "77.55%",
          "variant": "model ID openai/gpt-5.4-2026-03-05; reasoning_effort=xhigh; max_output_tokens=30000; standard error 3.316 pp; $0.639282/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20176,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.4-2026-03-05\"].accuracy; Overall leaderboard rank 64"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.1 Fast Non-Reasoning",
          "lab": "SpaceXAI",
          "scoreDisplay": "77.46%",
          "scoreNumeric": 77.464,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4-1-fast-non-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.04 pp; $0.001782/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.1-fast-non-reasoning",
          "scorePrinted": "77.46%",
          "variant": "model ID grok/grok-4-1-fast-non-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.04 pp; $0.001782/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20177,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-1-fast-non-reasoning\"].accuracy; Overall leaderboard rank 65"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3 VL Plus",
          "lab": "Alibaba",
          "scoreDisplay": "77.13%",
          "scoreNumeric": 77.129,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3-vl-plus-2025-09-23; temperature=1; max_output_tokens=30000; standard error 1.916 pp; $0.020220/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3-vl-plus",
          "scorePrinted": "77.13%",
          "variant": "model ID alibaba/qwen3-vl-plus-2025-09-23; temperature=1; max_output_tokens=30000; standard error 1.916 pp; $0.020220/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20178,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3-vl-plus-2025-09-23\"].accuracy; Overall leaderboard rank 66"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5.4 Nano",
          "lab": "OpenAI",
          "scoreDisplay": "77.09%",
          "scoreNumeric": 77.09,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5.4-nano-2026-03-17; reasoning_effort=high; max_output_tokens=30000; standard error 1.891 pp; $0.001800/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5.4-nano",
          "scorePrinted": "77.09%",
          "variant": "model ID openai/gpt-5.4-nano-2026-03-17; reasoning_effort=high; max_output_tokens=30000; standard error 1.891 pp; $0.001800/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20179,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.4-nano-2026-03-17\"].accuracy; Overall leaderboard rank 67"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.6 Plus",
          "lab": "Alibaba",
          "scoreDisplay": "76.96%",
          "scoreNumeric": 76.963,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.6-plus; temperature=1; max_output_tokens=30000; standard error 1.917 pp; $0.029294/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen3.6-plus",
          "scorePrinted": "76.96%",
          "variant": "model ID alibaba/qwen3.6-plus; temperature=1; max_output_tokens=30000; standard error 1.917 pp; $0.029294/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20180,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.6-plus\"].accuracy; Overall leaderboard rank 68"
          },
          "corroborating": []
        },
        {
          "model": "o3",
          "lab": "OpenAI",
          "scoreDisplay": "76.65%",
          "scoreNumeric": 76.654,
          "date": "2026-09-26",
          "config": "model ID openai/o3-2025-04-16; reasoning_effort=high; max_output_tokens=30000; standard error 1.871 pp; $0.040334/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "o3",
          "scorePrinted": "76.65%",
          "variant": "model ID openai/o3-2025-04-16; reasoning_effort=high; max_output_tokens=30000; standard error 1.871 pp; $0.040334/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20181,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/o3-2025-04-16\"].accuracy; Overall leaderboard rank 69"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.5 Flash",
          "lab": "Google",
          "scoreDisplay": "76.57%",
          "scoreNumeric": 76.574,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.5-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.923 pp; $0.166341/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.5-flash",
          "scorePrinted": "76.57%",
          "variant": "model ID google/gemini-3.5-flash; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.923 pp; $0.166341/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20182,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.5-flash\"].accuracy; Overall leaderboard rank 70"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K2.5",
          "lab": "Moonshot AI",
          "scoreDisplay": "76.44%",
          "scoreNumeric": 76.442,
          "date": "2026-09-26",
          "config": "model ID kimi/kimi-k2.5-thinking; top_p=0.95; max_output_tokens=30000; standard error 1.986 pp; $0.024890/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "kimi-k2.5",
          "scorePrinted": "76.44%",
          "variant": "model ID kimi/kimi-k2.5-thinking; top_p=0.95; max_output_tokens=30000; standard error 1.986 pp; $0.024890/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20183,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k2.5-thinking\"].accuracy; Overall leaderboard rank 71"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.1 Pro Preview (02/26)",
          "lab": "Google",
          "scoreDisplay": "76.11%",
          "scoreNumeric": 76.114,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.1-pro-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.915 pp; $0.097954/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.1-pro",
          "scorePrinted": "76.11%",
          "variant": "model ID google/gemini-3.1-pro-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.915 pp; $0.097954/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20184,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.1-pro-preview\"].accuracy; Overall leaderboard rank 72"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5",
          "lab": "Anthropic",
          "scoreDisplay": "76.05%",
          "scoreNumeric": 76.054,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 3.05 pp; $0.433684/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-5",
          "scorePrinted": "76.05%",
          "variant": "model ID anthropic/claude-sonnet-5; compute_effort=max; temperature=1; max_output_tokens=30000; standard error 3.05 pp; $0.433684/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20185,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-5\"].accuracy; Overall leaderboard rank 73"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Lite (9/25) (Nonthinking)",
          "lab": "Google",
          "scoreDisplay": "75.82%",
          "scoreNumeric": 75.824,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-lite-preview-09-2025; temperature=1; max_output_tokens=30000; standard error 1.851 pp; $0.001332/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-lite-9-25",
          "scorePrinted": "75.82%",
          "variant": "model ID google/gemini-2.5-flash-lite-preview-09-2025; temperature=1; max_output_tokens=30000; standard error 1.851 pp; $0.001332/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20186,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite-preview-09-2025\"].accuracy; Overall leaderboard rank 74"
          },
          "corroborating": []
        },
        {
          "model": "Ling 3.0 Flash Fin",
          "lab": "Ant Group",
          "scoreDisplay": "75.59%",
          "scoreNumeric": 75.593,
          "date": "2026-09-26",
          "config": "model ID ant/ling-3.0-flash-af-rc3; temperature=1; top_p=0.95; max_output_tokens=131072; standard error 2.026 pp; $0.001618/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "ling-3.0-flash-fin",
          "scorePrinted": "75.59%",
          "variant": "model ID ant/ling-3.0-flash-af-rc3; temperature=1; top_p=0.95; max_output_tokens=131072; standard error 2.026 pp; $0.001618/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20187,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"ant/ling-3.0-flash-af-rc3\"].accuracy; Overall leaderboard rank 75"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V4",
          "lab": "DeepSeek",
          "scoreDisplay": "75.14%",
          "scoreNumeric": 75.144,
          "date": "2026-09-26",
          "config": "model ID deepseek/deepseek-v4-pro; reasoning_effort=max; max_output_tokens=128000; standard error 2.002 pp; $0.053954/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "deepseek-v4",
          "scorePrinted": "75.14%",
          "variant": "model ID deepseek/deepseek-v4-pro; reasoning_effort=max; max_output_tokens=128000; standard error 2.002 pp; $0.053954/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20188,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-pro\"].accuracy; Overall leaderboard rank 76"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.3",
          "lab": "SpaceXAI",
          "scoreDisplay": "74.40%",
          "scoreNumeric": 74.399,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.3; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.019 pp; $0.015293/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.3",
          "scorePrinted": "74.40%",
          "variant": "model ID grok/grok-4.3; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.019 pp; $0.015293/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20189,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.3\"].accuracy; Overall leaderboard rank 77"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.1 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "73.90%",
          "scoreNumeric": 73.901,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-1-20250805-thinking; temperature=1; max_output_tokens=30000; standard error 1.965 pp; $0.263427/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.1",
          "scorePrinted": "73.90%",
          "variant": "model ID anthropic/claude-opus-4-1-20250805-thinking; temperature=1; max_output_tokens=30000; standard error 1.965 pp; $0.263427/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20190,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-1-20250805-thinking\"].accuracy; Overall leaderboard rank 78"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Pro",
          "lab": "Google",
          "scoreDisplay": "73.55%",
          "scoreNumeric": 73.552,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-pro; temperature=1; max_output_tokens=30000; standard error 1.91 pp; $0.046379/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-pro",
          "scorePrinted": "73.55%",
          "variant": "model ID google/gemini-2.5-pro; temperature=1; max_output_tokens=30000; standard error 1.91 pp; $0.046379/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20191,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-pro\"].accuracy; Overall leaderboard rank 79"
          },
          "corroborating": []
        },
        {
          "model": "GPT 5 Nano",
          "lab": "OpenAI",
          "scoreDisplay": "72.86%",
          "scoreNumeric": 72.865,
          "date": "2026-09-26",
          "config": "model ID openai/gpt-5-nano-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 1.891 pp; $0.006961/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gpt-5-nano",
          "scorePrinted": "72.86%",
          "variant": "model ID openai/gpt-5-nano-2025-08-07; reasoning_effort=high; max_output_tokens=30000; standard error 1.891 pp; $0.006961/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20192,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-nano-2025-08-07\"].accuracy; Overall leaderboard rank 80"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Lite (Nonthinking)",
          "lab": "Google",
          "scoreDisplay": "72.83%",
          "scoreNumeric": 72.832,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-lite; temperature=1; max_output_tokens=30000; standard error 1.982 pp; $0.001211/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-lite",
          "scorePrinted": "72.83%",
          "variant": "model ID google/gemini-2.5-flash-lite; temperature=1; max_output_tokens=30000; standard error 1.982 pp; $0.001211/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20193,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite\"].accuracy; Overall leaderboard rank 81"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3 Max Thinking",
          "lab": "Alibaba",
          "scoreDisplay": "72.71%",
          "scoreNumeric": 72.709,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3-max-2026-01-23; temperature=1; max_output_tokens=30000; standard error 1.905 pp; $0.085327/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3-max-thinking",
          "scorePrinted": "72.71%",
          "variant": "model ID alibaba/qwen3-max-2026-01-23; temperature=1; max_output_tokens=30000; standard error 1.905 pp; $0.085327/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20194,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3-max-2026-01-23\"].accuracy; Overall leaderboard rank 82"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "72.41%",
          "scoreNumeric": 72.411,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-4-20250514; temperature=1; max_output_tokens=30000; standard error 1.929 pp; $0.038973/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-4",
          "scorePrinted": "72.41%",
          "variant": "model ID anthropic/claude-sonnet-4-20250514; temperature=1; max_output_tokens=30000; standard error 1.929 pp; $0.038973/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20195,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-20250514\"].accuracy; Overall leaderboard rank 83"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.1",
          "lab": "zAI",
          "scoreDisplay": "72.27%",
          "scoreNumeric": 72.27,
          "date": "2026-09-26",
          "config": "model ID zai/glm-5.1; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.064 pp; $0.023717/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "glm-5.1",
          "scorePrinted": "72.27%",
          "variant": "model ID zai/glm-5.1; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.064 pp; $0.023717/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20196,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.1\"].accuracy; Overall leaderboard rank 84"
          },
          "corroborating": []
        },
        {
          "model": "MiMo V2.5",
          "lab": "Xiaomi",
          "scoreDisplay": "72.15%",
          "scoreNumeric": 72.151,
          "date": "2026-09-26",
          "config": "model ID xiaomi/mimo-v2.5; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.851 pp; $0.001351/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mimo-v2.5",
          "scorePrinted": "72.15%",
          "variant": "model ID xiaomi/mimo-v2.5; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 1.851 pp; $0.001351/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20197,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.5\"].accuracy; Overall leaderboard rank 85"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3 Pro (11/25)",
          "lab": "Google",
          "scoreDisplay": "72.04%",
          "scoreNumeric": 72.036,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3-pro-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.9 pp; $0.061162/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3-pro-11-25",
          "scorePrinted": "72.04%",
          "variant": "model ID google/gemini-3-pro-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.9 pp; $0.061162/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20198,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3-pro-preview\"].accuracy; Overall leaderboard rank 86"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.1 (Nonthinking)",
          "lab": "Anthropic",
          "scoreDisplay": "71.75%",
          "scoreNumeric": 71.753,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-opus-4-1-20250805; temperature=1; max_output_tokens=30000; standard error 2.021 pp; $0.187162/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-opus-4.1",
          "scorePrinted": "71.75%",
          "variant": "model ID anthropic/claude-opus-4-1-20250805; temperature=1; max_output_tokens=30000; standard error 2.021 pp; $0.187162/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20199,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-1-20250805\"].accuracy; Overall leaderboard rank 87"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.5 Flash Lite",
          "lab": "Google",
          "scoreDisplay": "70.89%",
          "scoreNumeric": 70.886,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.5-flash-lite; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.031 pp; $0.019867/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.5-flash-lite",
          "scorePrinted": "70.89%",
          "variant": "model ID google/gemini-3.5-flash-lite; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 2.031 pp; $0.019867/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20200,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.5-flash-lite\"].accuracy; Overall leaderboard rank 88"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.5 Flash",
          "lab": "Alibaba",
          "scoreDisplay": "70.62%",
          "scoreNumeric": 70.619,
          "date": "2026-09-26",
          "config": "model ID alibaba/qwen3.5-flash; temperature=1; max_output_tokens=30000; standard error 2.09 pp; $0.004425/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "qwen-3.5-flash",
          "scorePrinted": "70.62%",
          "variant": "model ID alibaba/qwen3.5-flash; temperature=1; max_output_tokens=30000; standard error 2.09 pp; $0.004425/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20201,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.5-flash\"].accuracy; Overall leaderboard rank 89"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3 Flash (12/25)",
          "lab": "Google",
          "scoreDisplay": "69.92%",
          "scoreNumeric": 69.917,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3-flash-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.899 pp; $0.014379/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3-flash",
          "scorePrinted": "69.92%",
          "variant": "model ID google/gemini-3-flash-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.899 pp; $0.014379/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20202,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3-flash-preview\"].accuracy; Overall leaderboard rank 90"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4 (Thinking)",
          "lab": "Anthropic",
          "scoreDisplay": "69.35%",
          "scoreNumeric": 69.353,
          "date": "2026-09-26",
          "config": "model ID anthropic/claude-sonnet-4-20250514-thinking; max_output_tokens=30000; standard error 2.212 pp; $0.053443/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "claude-sonnet-4",
          "scorePrinted": "69.35%",
          "variant": "model ID anthropic/claude-sonnet-4-20250514-thinking; max_output_tokens=30000; standard error 2.212 pp; $0.053443/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20203,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-20250514-thinking\"].accuracy; Overall leaderboard rank 91"
          },
          "corroborating": []
        },
        {
          "model": "o4 Mini",
          "lab": "OpenAI",
          "scoreDisplay": "69.14%",
          "scoreNumeric": 69.139,
          "date": "2026-09-26",
          "config": "model ID openai/o4-mini-2025-04-16; reasoning_effort=high; max_output_tokens=30000; standard error 1.957 pp; $0.040605/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "o4-mini",
          "scorePrinted": "69.14%",
          "variant": "model ID openai/o4-mini-2025-04-16; reasoning_effort=high; max_output_tokens=30000; standard error 1.957 pp; $0.040605/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20204,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/o4-mini-2025-04-16\"].accuracy; Overall leaderboard rank 92"
          },
          "corroborating": []
        },
        {
          "model": "GLM 4.7",
          "lab": "zAI",
          "scoreDisplay": "68.63%",
          "scoreNumeric": 68.629,
          "date": "2026-09-26",
          "config": "model ID zai/glm-4.7; temperature=1; top_p=1; max_output_tokens=30000; standard error 2.123 pp; $0.019082/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "glm-4.7",
          "scorePrinted": "68.63%",
          "variant": "model ID zai/glm-4.7; temperature=1; top_p=1; max_output_tokens=30000; standard error 2.123 pp; $0.019082/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20205,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-4.7\"].accuracy; Overall leaderboard rank 93"
          },
          "corroborating": []
        },
        {
          "model": "Mistral Medium 3.5",
          "lab": "Mistral",
          "scoreDisplay": "67.73%",
          "scoreNumeric": 67.728,
          "date": "2026-09-26",
          "config": "model ID mistralai/mistral-medium-3.5; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.011 pp; $0.157641/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mistral-medium-3.5",
          "scorePrinted": "67.73%",
          "variant": "model ID mistralai/mistral-medium-3.5; reasoning_effort=high; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.011 pp; $0.157641/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20206,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"mistralai/mistral-medium-3.5\"].accuracy; Overall leaderboard rank 94"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash Lite (9/25) (Thinking)",
          "lab": "Google",
          "scoreDisplay": "66.88%",
          "scoreNumeric": 66.877,
          "date": "2026-09-26",
          "config": "model ID google/gemini-2.5-flash-lite-preview-09-2025-thinking; temperature=1; max_output_tokens=30000; standard error 1.923 pp; $0.002567/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-2.5-flash-lite-9-25",
          "scorePrinted": "66.88%",
          "variant": "model ID google/gemini-2.5-flash-lite-preview-09-2025-thinking; temperature=1; max_output_tokens=30000; standard error 1.923 pp; $0.002567/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20207,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite-preview-09-2025-thinking\"].accuracy; Overall leaderboard rank 95"
          },
          "corroborating": []
        },
        {
          "model": "Laguna M.1",
          "lab": "Poolside",
          "scoreDisplay": "65.91%",
          "scoreNumeric": 65.911,
          "date": "2026-09-26",
          "config": "model ID poolside/laguna-m.1; temperature=1; max_output_tokens=30000; standard error 2.007 pp; $0.002202/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "laguna-m.1",
          "scorePrinted": "65.91%",
          "variant": "model ID poolside/laguna-m.1; temperature=1; max_output_tokens=30000; standard error 2.007 pp; $0.002202/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20208,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"poolside/laguna-m.1\"].accuracy; Overall leaderboard rank 96"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.1 Flash Lite Preview",
          "lab": "Google",
          "scoreDisplay": "63.90%",
          "scoreNumeric": 63.902,
          "date": "2026-09-26",
          "config": "model ID google/gemini-3.1-flash-lite-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.823 pp; $0.002195/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "gemini-3.1-flash-lite-preview",
          "scorePrinted": "63.90%",
          "variant": "model ID google/gemini-3.1-flash-lite-preview; reasoning_effort=high; temperature=1; max_output_tokens=30000; standard error 1.823 pp; $0.002195/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20209,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.1-flash-lite-preview\"].accuracy; Overall leaderboard rank 97"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.20 (Reasoning)",
          "lab": "SpaceXAI",
          "scoreDisplay": "63.41%",
          "scoreNumeric": 63.412,
          "date": "2026-09-26",
          "config": "model ID grok/grok-4.20-0309-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.095 pp; $0.031303/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "grok-4.20-reasoning",
          "scorePrinted": "63.41%",
          "variant": "model ID grok/grok-4.20-0309-reasoning; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 2.095 pp; $0.031303/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20210,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.20-0309-reasoning\"].accuracy; Overall leaderboard rank 98"
          },
          "corroborating": []
        },
        {
          "model": "Laguna XS.2",
          "lab": "Poolside",
          "scoreDisplay": "61.43%",
          "scoreNumeric": 61.432,
          "date": "2026-09-26",
          "config": "model ID poolside/laguna-xs.2; temperature=1; max_output_tokens=30000; standard error 2.349 pp; $0.001126/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "laguna-xs.2",
          "scorePrinted": "61.43%",
          "variant": "model ID poolside/laguna-xs.2; temperature=1; max_output_tokens=30000; standard error 2.349 pp; $0.001126/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20211,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"poolside/laguna-xs.2\"].accuracy; Overall leaderboard rank 99"
          },
          "corroborating": []
        },
        {
          "model": "Command A+",
          "lab": "Cohere",
          "scoreDisplay": "55.68%",
          "scoreNumeric": 55.682,
          "date": "2026-09-26",
          "config": "model ID cohere/command-a-plus-05-2026; temperature=1; top_p=0.95; max_output_tokens=64000; standard error 3.646 pp; $0.140316/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "command-a",
          "scorePrinted": "55.68%",
          "variant": "model ID cohere/command-a-plus-05-2026; temperature=1; top_p=0.95; max_output_tokens=64000; standard error 3.646 pp; $0.140316/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20212,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"cohere/command-a-plus-05-2026\"].accuracy; Overall leaderboard rank 100"
          },
          "corroborating": []
        },
        {
          "model": "Mercury 2.5",
          "lab": "Inception",
          "scoreDisplay": "55.09%",
          "scoreNumeric": 55.093,
          "date": "2026-09-26",
          "config": "model ID inception/mercury-2.5; reasoning_effort=high; temperature=1; max_output_tokens=65536; standard error 2.095 pp; $0.004476/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "mercury-2.5",
          "scorePrinted": "55.09%",
          "variant": "model ID inception/mercury-2.5; reasoning_effort=high; temperature=1; max_output_tokens=65536; standard error 2.095 pp; $0.004476/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20213,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"inception/mercury-2.5\"].accuracy; Overall leaderboard rank 101"
          },
          "corroborating": []
        },
        {
          "model": "Llama 4 Maverick",
          "lab": "Meta",
          "scoreDisplay": "54.22%",
          "scoreNumeric": 54.219,
          "date": "2026-09-26",
          "config": "model ID fireworks/llama4-maverick-instruct-basic; temperature=1; max_output_tokens=30000; standard error 1.871 pp; $0.002460/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "llama-4-maverick",
          "scorePrinted": "54.22%",
          "variant": "model ID fireworks/llama4-maverick-instruct-basic; temperature=1; max_output_tokens=30000; standard error 1.871 pp; $0.002460/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20214,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"fireworks/llama4-maverick-instruct-basic\"].accuracy; Overall leaderboard rank 102"
          },
          "corroborating": []
        },
        {
          "model": "Llama 4 Scout",
          "lab": "Meta",
          "scoreDisplay": "50.59%",
          "scoreNumeric": 50.593,
          "date": "2026-09-26",
          "config": "model ID together/meta-llama/Llama-4-Scout-17B-16E-Instruct; temperature=1; max_output_tokens=30000; standard error 1.901 pp; $0.001700/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "llama-4-scout",
          "scorePrinted": "50.59%",
          "variant": "model ID together/meta-llama/Llama-4-Scout-17B-16E-Instruct; temperature=1; max_output_tokens=30000; standard error 1.901 pp; $0.001700/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20215,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"together/meta-llama/Llama-4-Scout-17B-16E-Instruct\"].accuracy; Overall leaderboard rank 103"
          },
          "corroborating": []
        },
        {
          "model": "Nemotron 3.5 Lightning",
          "lab": "NVIDIA",
          "scoreDisplay": "4.27%",
          "scoreNumeric": 4.267,
          "date": "2026-09-26",
          "config": "model ID fireworks/nemotron-lightning-3p5-30b-a3b; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 0.492 pp; $0.006241/test; source snapshot 2026-09-26; run date not published",
          "modelSlug": "nemotron-3.5-lightning",
          "scorePrinted": "4.27%",
          "variant": "model ID fireworks/nemotron-lightning-3p5-30b-a3b; temperature=1; top_p=0.95; max_output_tokens=30000; standard error 0.492 pp; $0.006241/test; source snapshot 2026-09-26; run date not published",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20216,
          "source": {
            "id": 23,
            "title": "Vals AI MedScribe leaderboard",
            "url": "https://www.vals.ai/benchmarks/medscribe",
            "kind": "official_leaderboard",
            "publisher": "Vals AI",
            "firstParty": true,
            "publishedAt": "2026-09-26",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded BenchmarkView data, tasks.overall[\"fireworks/nemotron-lightning-3p5-30b-a3b\"].accuracy; Overall leaderboard rank 104"
          },
          "corroborating": []
        }
      ],
      "unitShort": "100 SOAP-note cases",
      "publisherShort": "Vals AI",
      "sourceShort": "Vals AI MedScribe leaderboard",
      "fieldSize": 104,
      "category": "documentation",
      "confidence": "verified",
      "officialUrl": "https://www.vals.ai/benchmarks/medscribe",
      "paperUrl": null,
      "scaleKind": "percent",
      "summary": "Quality and compliance of SOAP notes generated from clinical visits, scored against documentation rubrics.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "All 104 scored configurations checked against the first-party embedded table; source parameters captured per row. Updated 2026-09-26; individual run dates unavailable.",
        "urls": [
          "https://www.vals.ai/benchmarks/medscribe"
        ]
      }
    },
    {
      "slug": "medxpertqa-mm",
      "name": "MedXpertQA (MM)",
      "publisher": "TsinghuaC3I (Tsinghua University)",
      "released": "2025-01",
      "unit": "2,000 multimodal questions (MM subset)",
      "scale": "percentage accuracy 0-100, higher better",
      "description": "Expert-level multimodal medical multiple-choice QA covering clinical images (X-ray, histology, dermatology, charts) across 17 specialties; MM subset of the 4,460-question MedXpertQA benchmark.",
      "basis": "mixed",
      "sourceName": "Introducing Muse Spark: Scaling Towards Personal Superintelligence",
      "sourceUrl": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
      "ours": null,
      "lastUpdate": "2026-08",
      "notes": "Assembled table, not a single board. Rows come from several vendors' own launch documents: Meta's Muse Spark evaluation (whose values live only in a rendered table image and cover Meta's own model plus four competitors it re-ran or quoted), Alibaba's Qwen3.5, 3.7 and 3.8 posts (Alibaba's own runs of its models and of competitors), and Google's Gemma 4 model card. Protocols are not known to match across rows; each row's source line says which document it came from and who ran it. benchlm.ai presents a subset as one mirrored view while citing only Meta's methodology. Checked 2026-09-28: Meta image table visually verified from its methodology PDF; Qwen3.5 and Google model-card tables verified. Six Qwen3.7/3.8 blog rows retain their historical retrieval dates and partial confidence because those primary pages currently return no article content.",
      "results": [
        {
          "model": "GPT-5.6 Sol",
          "lab": "OpenAI",
          "scoreDisplay": "81.5",
          "scoreNumeric": 81.5,
          "date": null,
          "config": "Qwen-run comparison in the Qwen3.8-Max launch post; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "81.5",
          "variant": "Qwen-run comparison in the Qwen3.8-Max launch post; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "partial",
          "resultId": 548,
          "source": {
            "id": 152,
            "title": "Qwen3.8-Max: A New Bar for Coding and Cowork",
            "url": "https://qwen.ai/blog?id=qwen3.8",
            "kind": "launch_post",
            "publisher": "Alibaba",
            "firstParty": true,
            "publishedAt": "2026-08-02",
            "retrieved": "2026-09-07",
            "quote": null,
            "locator": "Full Benchmark Table, second (multimodal) table, row MedXpertQA-MM; columns Opus4.8 / Fable5 / Gemini3.1-Pro / GPT5.6-Sol / Qwen3.7-Plus / Qwen3.8-Max"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "81.3%",
          "scoreNumeric": 81.3,
          "date": "2026-04",
          "config": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "modelSlug": "gemini-3.1-pro",
          "scorePrinted": "81.3%",
          "variant": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "measured": "2026-04",
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 156,
          "source": {
            "id": 71,
            "title": "Introducing Muse Spark: Scaling Towards Personal Superintelligence",
            "url": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
            "kind": "launch_post",
            "publisher": "Meta",
            "firstParty": true,
            "publishedAt": "2026-04-08",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column Gemini 3.1 Pro High; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'"
          },
          "corroborating": [
            {
              "id": 74,
              "title": "Muse Spark Eval Methodology",
              "url": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
              "kind": "model_card",
              "publisher": "Meta",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "81.3",
              "quote": null,
              "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28"
            },
            {
              "id": 24,
              "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
              "url": "https://benchlm.ai/benchmarks/medxpertqamm",
              "kind": "mirror",
              "publisher": "benchlm.ai",
              "firstParty": false,
              "role": "mirror",
              "scoreDisplay": "81.3%",
              "quote": null,
              "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 1; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology"
            }
          ]
        },
        {
          "model": "Qwen3.8 Max",
          "lab": "Alibaba",
          "scoreDisplay": "80.4%",
          "scoreNumeric": 80.4,
          "date": "2026-08",
          "config": "Alibaba's own Qwen3.8 launch table; protocol not stated; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "modelSlug": "qwen3.8-max",
          "scorePrinted": "80.4%",
          "variant": "Alibaba's own Qwen3.8 launch table; protocol not stated; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "measured": "2026-08",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "partial",
          "resultId": 157,
          "source": {
            "id": 152,
            "title": "Qwen3.8-Max: A New Bar for Coding and Cowork",
            "url": "https://qwen.ai/blog?id=qwen3.8",
            "kind": "launch_post",
            "publisher": "Alibaba",
            "firstParty": true,
            "publishedAt": "2026-08-02",
            "retrieved": "2026-09-07",
            "quote": null,
            "locator": "Multimodal Benchmarks table, Multimodal Reasoning section, row MedXpertQA-MM, column Qwen3.8-Max; header row: |     | Opus4.8 | Fable5 | Gemini3.1-Pro | GPT5.6-Sol | Qwen3.7-Plus | Qwen3.8-Max |"
          },
          "corroborating": [
            {
              "id": 24,
              "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
              "url": "https://benchlm.ai/benchmarks/medxpertqamm",
              "kind": "mirror",
              "publisher": "benchlm.ai",
              "firstParty": false,
              "role": "mirror",
              "scoreDisplay": "80.4%",
              "quote": null,
              "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 2; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology"
            }
          ]
        },
        {
          "model": "Claude Fable 5",
          "lab": "Anthropic",
          "scoreDisplay": "80.0",
          "scoreNumeric": 80,
          "date": null,
          "config": "Qwen-run comparison in the Qwen3.8-Max launch post; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "modelSlug": "claude-fable-5",
          "scorePrinted": "80.0",
          "variant": "Qwen-run comparison in the Qwen3.8-Max launch post; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "partial",
          "resultId": 547,
          "source": {
            "id": 152,
            "title": "Qwen3.8-Max: A New Bar for Coding and Cowork",
            "url": "https://qwen.ai/blog?id=qwen3.8",
            "kind": "launch_post",
            "publisher": "Alibaba",
            "firstParty": true,
            "publishedAt": "2026-08-02",
            "retrieved": "2026-09-07",
            "quote": null,
            "locator": "Full Benchmark Table, second (multimodal) table, row MedXpertQA-MM; columns Opus4.8 / Fable5 / Gemini3.1-Pro / GPT5.6-Sol / Qwen3.7-Plus / Qwen3.8-Max"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark",
          "lab": "Meta",
          "scoreDisplay": "78.4%",
          "scoreNumeric": 78.4,
          "date": "2026-04",
          "config": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "modelSlug": "muse-spark",
          "scorePrinted": "78.4%",
          "variant": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "measured": "2026-04",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 158,
          "source": {
            "id": 71,
            "title": "Introducing Muse Spark: Scaling Towards Personal Superintelligence",
            "url": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
            "kind": "launch_post",
            "publisher": "Meta",
            "firstParty": true,
            "publishedAt": "2026-04-08",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column Muse Spark Thinking; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'"
          },
          "corroborating": [
            {
              "id": 74,
              "title": "Muse Spark Eval Methodology",
              "url": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
              "kind": "model_card",
              "publisher": "Meta",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "78.4",
              "quote": null,
              "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28"
            },
            {
              "id": 24,
              "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
              "url": "https://benchlm.ai/benchmarks/medxpertqamm",
              "kind": "mirror",
              "publisher": "benchlm.ai",
              "firstParty": false,
              "role": "mirror",
              "scoreDisplay": "78.4%",
              "quote": null,
              "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 3; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology"
            }
          ]
        },
        {
          "model": "GPT-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "77.1%",
          "scoreNumeric": 77.1,
          "date": "2026-04",
          "config": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "77.1%",
          "variant": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "measured": "2026-04",
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 159,
          "source": {
            "id": 71,
            "title": "Introducing Muse Spark: Scaling Towards Personal Superintelligence",
            "url": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
            "kind": "launch_post",
            "publisher": "Meta",
            "firstParty": true,
            "publishedAt": "2026-04-08",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column GPT 5.4 Xhigh; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'"
          },
          "corroborating": [
            {
              "id": 74,
              "title": "Muse Spark Eval Methodology",
              "url": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
              "kind": "model_card",
              "publisher": "Meta",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "77.1",
              "quote": null,
              "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28"
            },
            {
              "id": 24,
              "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
              "url": "https://benchlm.ai/benchmarks/medxpertqamm",
              "kind": "mirror",
              "publisher": "benchlm.ai",
              "firstParty": false,
              "role": "mirror",
              "scoreDisplay": "77.1%",
              "quote": null,
              "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 4; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology"
            }
          ]
        },
        {
          "model": "Gemini 3 Pro",
          "lab": "Google",
          "scoreDisplay": "76.0%",
          "scoreNumeric": 76,
          "date": "",
          "config": "Qwen3.5 model card comparison; source-specific evaluation, not harmonized across vendors",
          "modelSlug": "gemini-3-pro",
          "scorePrinted": "76.0%",
          "variant": "Qwen3.5 model card comparison; source-specific evaluation, not harmonized across vendors",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 20252,
          "source": {
            "id": 318,
            "title": "Qwen/Qwen3.5-397B-A17B model card",
            "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
            "kind": "model_card",
            "publisher": "Alibaba / Qwen",
            "firstParty": true,
            "publishedAt": "2026-02-16",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.2",
          "lab": "OpenAI",
          "scoreDisplay": "73.3",
          "scoreNumeric": 73.3,
          "date": null,
          "config": "Qwen-run comparison in the Qwen3.5-397B-A17B model card",
          "modelSlug": "gpt-5.2",
          "scorePrinted": "73.3",
          "variant": "Qwen-run comparison in the Qwen3.5-397B-A17B model card",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 550,
          "source": {
            "id": 318,
            "title": "Qwen/Qwen3.5-397B-A17B model card",
            "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
            "kind": "model_card",
            "publisher": "Alibaba / Qwen",
            "firstParty": true,
            "publishedAt": "2026-02-16",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.8",
          "lab": "Anthropic",
          "scoreDisplay": "71.7",
          "scoreNumeric": 71.7,
          "date": null,
          "config": "Qwen-run comparison in the Qwen3.8-Max launch post; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "modelSlug": "claude-opus-4.8",
          "scorePrinted": "71.7",
          "variant": "Qwen-run comparison in the Qwen3.8-Max launch post; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "partial",
          "resultId": 546,
          "source": {
            "id": 152,
            "title": "Qwen3.8-Max: A New Bar for Coding and Cowork",
            "url": "https://qwen.ai/blog?id=qwen3.8",
            "kind": "launch_post",
            "publisher": "Alibaba",
            "firstParty": true,
            "publishedAt": "2026-08-02",
            "retrieved": "2026-09-07",
            "quote": null,
            "locator": "Full Benchmark Table, second (multimodal) table, row MedXpertQA-MM; columns Opus4.8 / Fable5 / Gemini3.1-Pro / GPT5.6-Sol / Qwen3.7-Plus / Qwen3.8-Max"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3.7 Plus",
          "lab": "Alibaba",
          "scoreDisplay": "71.0%",
          "scoreNumeric": 71,
          "date": "2026-05",
          "config": "Alibaba's own Qwen3.7 Plus launch table; protocol not stated; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "modelSlug": "qwen3.7-plus",
          "scorePrinted": "71.0%",
          "variant": "Alibaba's own Qwen3.7 Plus launch table; protocol not stated; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "measured": "2026-05",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "partial",
          "resultId": 160,
          "source": {
            "id": 205,
            "title": "Qwen3.7-Plus: Multimodal Agent Intelligence",
            "url": "https://qwen.ai/blog?id=qwen3.7-plus",
            "kind": "launch_post",
            "publisher": "Alibaba",
            "firstParty": true,
            "publishedAt": "2026-05-31",
            "retrieved": "2026-09-07",
            "quote": null,
            "locator": "Multimodal Benchmarks table, Multimodal Reasoning section, row MedXpertQA-MM, column Qwen3.7-Plus; header row: |     | GPT-5.4 (xhigh) | Opus-4.6 Max | Gemini-3.1 Pro | Qwen3.6-Plus | Qwen3.7-Plus |"
          },
          "corroborating": [
            {
              "id": 152,
              "title": "Qwen3.8-Max: A New Bar for Coding and Cowork",
              "url": "https://qwen.ai/blog?id=qwen3.8",
              "kind": "launch_post",
              "publisher": "Alibaba",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "71.0",
              "quote": null,
              "locator": "Qwen3.8 launch post, Multimodal Benchmarks table, row MedXpertQA-MM, column Qwen3.7-Plus (same value carried forward)"
            },
            {
              "id": 24,
              "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
              "url": "https://benchlm.ai/benchmarks/medxpertqamm",
              "kind": "mirror",
              "publisher": "benchlm.ai",
              "firstParty": false,
              "role": "mirror",
              "scoreDisplay": "71.0%",
              "quote": null,
              "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 5; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology"
            }
          ]
        },
        {
          "model": "Qwen3.5 397B A17B",
          "lab": "Alibaba",
          "scoreDisplay": "70.0",
          "scoreNumeric": 70,
          "date": null,
          "config": "self-reported in the Qwen3.5-397B-A17B model card",
          "modelSlug": "qwen3.5-397b-a17b",
          "scorePrinted": "70.0",
          "variant": "self-reported in the Qwen3.5-397B-A17B model card",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 552,
          "source": {
            "id": 318,
            "title": "Qwen/Qwen3.5-397B-A17B model card",
            "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
            "kind": "model_card",
            "publisher": "Alibaba / Qwen",
            "firstParty": true,
            "publishedAt": "2026-02-16",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3.6 Plus",
          "lab": "Alibaba",
          "scoreDisplay": "68.7",
          "scoreNumeric": 68.7,
          "date": null,
          "config": "Qwen-run comparison in the Qwen3.7-Plus launch post; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "modelSlug": "qwen3.6-plus",
          "scorePrinted": "68.7",
          "variant": "Qwen-run comparison in the Qwen3.7-Plus launch post; Primary blog currently renders an empty shell; retained historical value, not reverified on 2026-09-28.",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "partial",
          "resultId": 549,
          "source": {
            "id": 205,
            "title": "Qwen3.7-Plus: Multimodal Agent Intelligence",
            "url": "https://qwen.ai/blog?id=qwen3.7-plus",
            "kind": "launch_post",
            "publisher": "Alibaba",
            "firstParty": true,
            "publishedAt": "2026-05-31",
            "retrieved": "2026-09-07",
            "quote": null,
            "locator": "Multimodal Benchmarks table, row MedXpertQA-MM; columns GPT-5.4 (xhigh) / Opus-4.6 Max / Gemini-3.1 Pro / Qwen3.6-Plus / Qwen3.7-Plus"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.20",
          "lab": "xAI",
          "scoreDisplay": "65.8%",
          "scoreNumeric": 65.8,
          "date": "2026-04",
          "config": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "modelSlug": "grok-4.20",
          "scorePrinted": "65.8%",
          "variant": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "measured": "2026-04",
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 161,
          "source": {
            "id": 71,
            "title": "Introducing Muse Spark: Scaling Towards Personal Superintelligence",
            "url": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
            "kind": "launch_post",
            "publisher": "Meta",
            "firstParty": true,
            "publishedAt": "2026-04-08",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column Grok 4.2 Reasoning; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'"
          },
          "corroborating": [
            {
              "id": 74,
              "title": "Muse Spark Eval Methodology",
              "url": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
              "kind": "model_card",
              "publisher": "Meta",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "65.8",
              "quote": null,
              "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28"
            },
            {
              "id": 24,
              "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
              "url": "https://benchlm.ai/benchmarks/medxpertqamm",
              "kind": "mirror",
              "publisher": "benchlm.ai",
              "firstParty": false,
              "role": "mirror",
              "scoreDisplay": "65.8%",
              "quote": null,
              "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 6; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology"
            }
          ]
        },
        {
          "model": "Kimi K2.5",
          "lab": "Moonshot AI",
          "scoreDisplay": "65.3",
          "scoreNumeric": 65.3,
          "date": null,
          "config": "Qwen-run comparison in the Qwen3.5-397B-A17B model card (K2.5-1T-A32B column)",
          "modelSlug": "kimi-k2.5",
          "scorePrinted": "65.3",
          "variant": "Qwen-run comparison in the Qwen3.5-397B-A17B model card (K2.5-1T-A32B column)",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 551,
          "source": {
            "id": 318,
            "title": "Qwen/Qwen3.5-397B-A17B model card",
            "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
            "kind": "model_card",
            "publisher": "Alibaba / Qwen",
            "firstParty": true,
            "publishedAt": "2026-02-16",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "64.8%",
          "scoreNumeric": 64.8,
          "date": "2026-04",
          "config": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "64.8%",
          "variant": "Meta's Muse Spark launch table (Meta reports the better of vendor self-reports and its own reproduction)",
          "measured": "2026-04",
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 162,
          "source": {
            "id": 71,
            "title": "Introducing Muse Spark: Scaling Towards Personal Superintelligence",
            "url": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
            "kind": "launch_post",
            "publisher": "Meta",
            "firstParty": true,
            "publishedAt": "2026-04-08",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column Opus 4.6 Max; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'"
          },
          "corroborating": [
            {
              "id": 74,
              "title": "Muse Spark Eval Methodology",
              "url": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
              "kind": "model_card",
              "publisher": "Meta",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "64.8",
              "quote": null,
              "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28"
            },
            {
              "id": 24,
              "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
              "url": "https://benchlm.ai/benchmarks/medxpertqamm",
              "kind": "mirror",
              "publisher": "benchlm.ai",
              "firstParty": false,
              "role": "mirror",
              "scoreDisplay": "64.8%",
              "quote": null,
              "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 7; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology"
            }
          ]
        },
        {
          "model": "Claude Opus 4.5",
          "lab": "Anthropic",
          "scoreDisplay": "63.6%",
          "scoreNumeric": 63.6,
          "date": "",
          "config": "Qwen3.5 model card comparison; source-specific evaluation, not harmonized across vendors",
          "modelSlug": "claude-opus-4.5",
          "scorePrinted": "63.6%",
          "variant": "Qwen3.5 model card comparison; source-specific evaluation, not harmonized across vendors",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "independent run",
          "confidence": "verified",
          "resultId": 20253,
          "source": {
            "id": 318,
            "title": "Qwen/Qwen3.5-397B-A17B model card",
            "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
            "kind": "model_card",
            "publisher": "Alibaba / Qwen",
            "firstParty": true,
            "publishedAt": "2026-02-16",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B"
          },
          "corroborating": []
        },
        {
          "model": "Gemma 4 31B",
          "lab": "Google",
          "scoreDisplay": "61.3%",
          "scoreNumeric": 61.3,
          "date": "",
          "config": "Google model card; MedXPertQA MM row; vendor-reported; protocol differs from other source families",
          "modelSlug": "gemma-4-31b",
          "scorePrinted": "61.3%",
          "variant": "Google model card; MedXPertQA MM row; vendor-reported; protocol differs from other source families",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20248,
          "source": {
            "id": 202,
            "title": "Gemma 4 model card",
            "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "kind": "model_card",
            "publisher": "Google",
            "firstParty": true,
            "publishedAt": "2026-04-02",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Evaluation Results table, MedXPertQA MM row, Gemma 4 31B column"
          },
          "corroborating": []
        },
        {
          "model": "Gemma 4 26B A4B",
          "lab": "Google",
          "scoreDisplay": "58.1%",
          "scoreNumeric": 58.1,
          "date": "",
          "config": "Google model card; MedXPertQA MM row; vendor-reported; protocol differs from other source families",
          "modelSlug": "gemma-4-26b-a4b",
          "scorePrinted": "58.1%",
          "variant": "Google model card; MedXPertQA MM row; vendor-reported; protocol differs from other source families",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20249,
          "source": {
            "id": 202,
            "title": "Gemma 4 model card",
            "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "kind": "model_card",
            "publisher": "Google",
            "firstParty": true,
            "publishedAt": "2026-04-02",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Evaluation Results table, MedXPertQA MM row, Gemma 4 26B A4B column"
          },
          "corroborating": []
        },
        {
          "model": "Gemma 4 12B",
          "lab": "Google",
          "scoreDisplay": "48.7%",
          "scoreNumeric": 48.7,
          "date": "2026-04",
          "config": "Google's Gemma 4 model card, Unified 12B; protocol not stated",
          "modelSlug": "gemma-4-12b",
          "scorePrinted": "48.7%",
          "variant": "Google's Gemma 4 model card, Unified 12B; protocol not stated",
          "measured": "2026-04",
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 163,
          "source": {
            "id": 202,
            "title": "Gemma 4 model card",
            "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "kind": "model_card",
            "publisher": "Google",
            "firstParty": true,
            "publishedAt": "2026-04-02",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Benchmark Results table, Vision section, row MedXPertQA MM, column Gemma 4 12B Unified; header row: |     | Gemma 4 31B | Gemma 4 26B A4B | Gemma 4 12B Unified | Gemma 4 E4B | Gemma 4 E2B | Gemma 3 27B (no think) |"
          },
          "corroborating": [
            {
              "id": 212,
              "title": "Gemma 4 Technical Report",
              "url": "https://arxiv.org/pdf/2607.02770",
              "kind": "paper",
              "publisher": "Google DeepMind",
              "firstParty": false,
              "role": "corroborating",
              "scoreDisplay": "48.7",
              "quote": null,
              "locator": "Gemma 4 Technical Report p. 6, Table 6 (vision benchmarks, thinking), row MedXPertQA MM, column 12B"
            },
            {
              "id": 24,
              "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
              "url": "https://benchlm.ai/benchmarks/medxpertqamm",
              "kind": "mirror",
              "publisher": "benchlm.ai",
              "firstParty": false,
              "role": "mirror",
              "scoreDisplay": "48.7%",
              "quote": null,
              "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 8; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology"
            }
          ]
        },
        {
          "model": "Qwen3-VL-235B-A22B",
          "lab": "Alibaba",
          "scoreDisplay": "47.6%",
          "scoreNumeric": 47.6,
          "date": "",
          "config": "Qwen3.5 model card comparison; source-specific evaluation, not harmonized across vendors",
          "modelSlug": "qwen3-vl-235b-a22b",
          "scorePrinted": "47.6%",
          "variant": "Qwen3.5 model card comparison; source-specific evaluation, not harmonized across vendors",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20254,
          "source": {
            "id": 318,
            "title": "Qwen/Qwen3.5-397B-A17B model card",
            "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
            "kind": "model_card",
            "publisher": "Alibaba / Qwen",
            "firstParty": true,
            "publishedAt": "2026-02-16",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B"
          },
          "corroborating": []
        },
        {
          "model": "Gemma 4 E4B",
          "lab": "Google",
          "scoreDisplay": "28.7%",
          "scoreNumeric": 28.7,
          "date": "",
          "config": "Google model card; MedXPertQA MM row; vendor-reported; protocol differs from other source families",
          "modelSlug": "gemma-4-e4b",
          "scorePrinted": "28.7%",
          "variant": "Google model card; MedXPertQA MM row; vendor-reported; protocol differs from other source families",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20250,
          "source": {
            "id": 202,
            "title": "Gemma 4 model card",
            "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "kind": "model_card",
            "publisher": "Google",
            "firstParty": true,
            "publishedAt": "2026-04-02",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Evaluation Results table, MedXPertQA MM row, Gemma 4 E4B column"
          },
          "corroborating": []
        },
        {
          "model": "Gemma 4 E2B",
          "lab": "Google",
          "scoreDisplay": "23.5%",
          "scoreNumeric": 23.5,
          "date": "",
          "config": "Google model card; MedXPertQA MM row; vendor-reported; protocol differs from other source families",
          "modelSlug": "gemma-4-e2b",
          "scorePrinted": "23.5%",
          "variant": "Google model card; MedXPertQA MM row; vendor-reported; protocol differs from other source families",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20251,
          "source": {
            "id": 202,
            "title": "Gemma 4 model card",
            "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
            "kind": "model_card",
            "publisher": "Google",
            "firstParty": true,
            "publishedAt": "2026-04-02",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Evaluation Results table, MedXPertQA MM row, Gemma 4 E2B column"
          },
          "corroborating": []
        }
      ],
      "unitShort": "2,000 multimodal questions",
      "publisherShort": "Tsinghua University",
      "sourceShort": "Meta Muse Spark eval",
      "fieldSize": 22,
      "category": "knowledge",
      "confidence": "partial",
      "officialUrl": "https://github.com/TsinghuaC3I/MedXpertQA",
      "paperUrl": null,
      "scaleKind": "percent",
      "summary": "Expert-level multiple-choice questions over clinical images across 17 specialties.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "partial",
        "summary": "Meta, Qwen3.5 and Google score tables verified; all published Gemma 4 and Qwen3.5 comparison columns included. Six Qwen3.7/3.8 blog values remain historical and partial because primary pages are empty. No cross-vendor protocol equivalence implied.",
        "urls": [
          "https://ai.meta.com/blog/introducing-muse-spark-msl/",
          "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
          "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
          "https://ai.google.dev/gemma/docs/core/model_card_4",
          "https://qwen.ai/blog?id=qwen3.8",
          "https://qwen.ai/blog?id=qwen3.7-plus"
        ]
      }
    },
    {
      "slug": "artificial-analysis-healthcare",
      "name": "Artificial Analysis Healthcare & Medical Index",
      "publisher": "Artificial Analysis",
      "released": "2026-08",
      "unit": "Six-evaluation composite; 25 default-selected models indexed",
      "scale": "index score, higher better",
      "description": "Artificial Analysis independently evaluates a weighted healthcare composite: knowledge 30%, agentic knowledge work 25%, long-context reasoning 15%, non-hallucination 10%, reasoning 10%, and agentic tool use 10%. Components include AA-Omniscience, GDPval-AA v2.1, AA-Briefcase v1.1, MLCR-AA, Humanity’s Last Exam, and AutomationBench-AA.",
      "basis": "independent-run",
      "sourceName": "Best AI for Healthcare & Medical: LLM Leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
      "ours": null,
      "lastUpdate": "2026-09",
      "notes": "Current component versions and weights were checked September 28, 2026. This snapshot indexes the 25 default-selected models with numeric data embedded in the official page; the page offers 77 model variants overall. Historical rows absent from that snapshot are omitted rather than mixed with the prior five-evaluation composition. Scores are rounded whole index points; underlying values are recorded in source locators. Individual model evaluation dates are not published.",
      "results": [
        {
          "model": "Claude Opus 5.5 (Adaptive Reasoning, Max Effort, Default Fallback)",
          "lab": "Anthropic",
          "scoreDisplay": "61",
          "scoreNumeric": 61,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "claude-opus-5.5",
          "scorePrinted": "61",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10041,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=claude-opus-5-5, weightedIndex=60.5359874288733; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5.1 (Adaptive Reasoning, Max Effort, Default Fallback)",
          "lab": "Anthropic",
          "scoreDisplay": "58",
          "scoreNumeric": 58,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "claude-fable-5.1",
          "scorePrinted": "58",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 528,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=claude-fable-5-1, weightedIndex=58.0046680298741; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 5 (Adaptive Reasoning, Max Effort)",
          "lab": "Anthropic",
          "scoreDisplay": "53",
          "scoreNumeric": 53,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "claude-opus-5",
          "scorePrinted": "53",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 164,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=claude-opus-5, weightedIndex=53.3389883833501; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Astra (max)",
          "lab": "OpenAI",
          "scoreDisplay": "52",
          "scoreNumeric": 52,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "gpt-6-astra",
          "scorePrinted": "52",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10043,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-6-astra, weightedIndex=51.6668965494132; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Muse Spark 1.3 (max)",
          "lab": "Meta",
          "scoreDisplay": "50",
          "scoreNumeric": 50,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "muse-spark-1.3",
          "scorePrinted": "50",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10045,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=muse-spark-1-3, weightedIndex=49.6912519301107; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4.7 (xhigh)",
          "lab": "SpaceXAI",
          "scoreDisplay": "47",
          "scoreNumeric": 47,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "grok-4.7",
          "scorePrinted": "47",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10048,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=grok-4-7, weightedIndex=47.1903068758764; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "GLM-5.3 (max)",
          "lab": "Z AI",
          "scoreDisplay": "47",
          "scoreNumeric": 47,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "glm-5.3",
          "scorePrinted": "47",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10051,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=glm-5-3, weightedIndex=46.6708125611712; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Sol (max)",
          "lab": "OpenAI",
          "scoreDisplay": "45",
          "scoreNumeric": 45,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "gpt-5.6-sol",
          "scorePrinted": "45",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 530,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-5-6-sol, weightedIndex=45.4043575131731; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Kimi K3 (max)",
          "lab": "Kimi",
          "scoreDisplay": "45",
          "scoreNumeric": 45,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "kimi-k3",
          "scorePrinted": "45",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 531,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=kimi-k3, weightedIndex=45.0434943845881; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "GLM 5.3 Flash",
          "lab": "Z AI",
          "scoreDisplay": "45",
          "scoreNumeric": 45,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "glm-5.3-flash",
          "scorePrinted": "45",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10054,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=glm-5-3-flash, weightedIndex=44.8426192521491; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Sol (max)",
          "lab": "OpenAI",
          "scoreDisplay": "43",
          "scoreNumeric": 43,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "gpt-6-sol",
          "scorePrinted": "43",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10046,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-6-sol, weightedIndex=43.4880669048048; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "MiMo-V2.6-Pro",
          "lab": "Xiaomi",
          "scoreDisplay": "42",
          "scoreNumeric": 42,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "mimo-v2.6-pro",
          "scorePrinted": "42",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10049,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=mimo-v2-6-pro, weightedIndex=41.8204296860305; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.8 Flash (high)",
          "lab": "Google",
          "scoreDisplay": "42",
          "scoreNumeric": 42,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "gemini-3.8-flash",
          "scorePrinted": "42",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10055,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gemini-3-8-flash, weightedIndex=41.7830626872104; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3.8 Max (0902)",
          "lab": "Alibaba",
          "scoreDisplay": "41",
          "scoreNumeric": 41,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "qwen3.8-max",
          "scorePrinted": "41",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10050,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=qwen3-8-max, weightedIndex=41.4279151085327; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Step 5 Preview",
          "lab": "StepFun",
          "scoreDisplay": "41",
          "scoreNumeric": 41,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "step-5",
          "scorePrinted": "41",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10052,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=step-5, weightedIndex=40.8171769208551; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V4.1 Flash (Reasoning, Max Effort)",
          "lab": "DeepSeek",
          "scoreDisplay": "41",
          "scoreNumeric": 41,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "deepseek-v4.1-flash",
          "scorePrinted": "41",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10056,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=deepseek-v4-1-flash, weightedIndex=40.6388737813166; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "GPT-6 Luna (max)",
          "lab": "OpenAI",
          "scoreDisplay": "37",
          "scoreNumeric": 37,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "gpt-6-luna",
          "scorePrinted": "37",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10058,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-6-luna, weightedIndex=36.7252603853009; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.6 Luna (max)",
          "lab": "OpenAI",
          "scoreDisplay": "36",
          "scoreNumeric": 36,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "gpt-5.6-luna",
          "scorePrinted": "36",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 533,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-5-6-luna, weightedIndex=35.740444759121; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3.8 27B (xhigh)",
          "lab": "Alibaba",
          "scoreDisplay": "34",
          "scoreNumeric": 34,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "qwen3.8-27b",
          "scorePrinted": "34",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10059,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=qwen3-8-27b, weightedIndex=34.4310443115205; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "MiniMax-M3",
          "lab": "MiniMax",
          "scoreDisplay": "30",
          "scoreNumeric": 30,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "minimax-m3",
          "scorePrinted": "30",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 534,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=minimax-m3, weightedIndex=29.7042505129902; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Inkling (xhigh)",
          "lab": "Thinking Machines",
          "scoreDisplay": "25",
          "scoreNumeric": 25,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "inkling",
          "scorePrinted": "25",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 535,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=inkling, weightedIndex=25.3933378261902; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Nemotron 3 Ultra 550B A55B (Reasoning)",
          "lab": "NVIDIA",
          "scoreDisplay": "23",
          "scoreNumeric": 23,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "nvidia-nemotron-3-ultra-550b-a55b",
          "scorePrinted": "23",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10062,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=nvidia-nemotron-3-ultra-550b-a55b, weightedIndex=23.0046725383136; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 3.5 Flash-Lite",
          "lab": "Google",
          "scoreDisplay": "23",
          "scoreNumeric": 23,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "gemini-3.5-flash-lite",
          "scorePrinted": "23",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10063,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gemini-3-5-flash-lite, weightedIndex=23.074121756108; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Muse Glimmer (high)",
          "lab": "Meta",
          "scoreDisplay": "18",
          "scoreNumeric": 18,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "muse-glimmer",
          "scorePrinted": "18",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 10064,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=muse-glimmer, weightedIndex=17.5980321302316; round to nearest whole point"
          },
          "corroborating": []
        },
        {
          "model": "Mistral Medium 3.5",
          "lab": "Mistral",
          "scoreDisplay": "14",
          "scoreNumeric": 14,
          "date": "2026-09",
          "config": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "modelSlug": "mistral-medium-3.5",
          "scorePrinted": "14",
          "variant": "Artificial Analysis independent evaluation; current six-evaluation healthcare composite, including GDPval-AA v2.1, AA-Briefcase v1.1 and AutomationBench-AA. Configuration is shown in the model name. Board snapshot checked September 28, 2026; individual run dates are not disclosed. Display rounded to whole index points.",
          "measured": null,
          "reportedBy": "third_party",
          "reportedByLabel": "third-party run",
          "confidence": "verified",
          "resultId": 536,
          "source": {
            "id": 10547,
            "title": "Artificial Analysis Healthcare & Medical Index",
            "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
            "kind": "official_leaderboard",
            "publisher": "Artificial Analysis",
            "firstParty": true,
            "publishedAt": null,
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=mistral-medium-3-5, weightedIndex=13.5556785369912; round to nearest whole point"
          },
          "corroborating": []
        }
      ],
      "unitShort": "6-evaluation composite",
      "publisherShort": "Artificial Analysis",
      "sourceShort": "Artificial Analysis",
      "fieldSize": 77,
      "category": "composite",
      "confidence": "partial",
      "officialUrl": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
      "paperUrl": null,
      "scaleKind": "index",
      "summary": "A healthcare-weighted composite of six evaluations, covering medical knowledge, long records, knowledge work, reasoning and tools.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "partial",
        "summary": "Verified the current six-evaluation methodology and 25 numeric model rows from first-party embedded chart data. This is partial coverage of 77 available variants; individual evaluation dates are undisclosed. Older incompatible index rows were removed.",
        "urls": [
          "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical"
        ]
      }
    },
    {
      "slug": "physicianbench",
      "name": "PhysicianBench",
      "publisher": "Academic team (Ruoqi Liu, Imran Q. Mohiuddin et al., arXiv 2605.02240); also a MAST component",
      "released": "2026-05",
      "unit": "100 real-world clinical tasks, 21 specialties, 670 structured checkpoints (~27 tool calls per task)",
      "scale": "pass@1 success rate %, higher better (3 independent runs; Pass^3 also reported)",
      "description": "LLM agents on long-horizon composite physician workflows inside real EHR environments, with execution-grounded verification against actual EHR systems via standard commercial APIs.",
      "basis": "mixed",
      "sourceName": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
      "sourceUrl": "https://arxiv.org/pdf/2605.02240",
      "ours": null,
      "lastUpdate": "2026-09-28",
      "notes": "Two evaluator cohorts are shown on the same public 100-task set: the May 2026 benchmark paper (12 models, shared FHIR loop, 3 runs), and Anthropic’s September 28 system card (9 model/effort configurations, shared vendor harness, Opus 5 rubric grader). The vendor cohort is not a replication of the paper protocol. Read configuration and reporting labels before comparing cohorts. Opus 5.5 leads Anthropic’s max-effort cohort at 68.4%; GPT-5.5 leads the paper cohort at 46.3%. Differences of 5–6 points are near the resolution of this 100-task benchmark.",
      "results": [
        {
          "model": "Claude Opus 5.5 (max)",
          "alias": "Claude Opus 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "68.4%",
          "scoreNumeric": 68.4,
          "date": "2026-09-28",
          "config": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "modelSlug": "claude-opus-5.5",
          "scorePrinted": "68.4%",
          "variant": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20255,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5.5 (max)",
          "alias": "Claude Sonnet 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "63.2%",
          "scoreNumeric": 63.2,
          "date": "2026-09-28",
          "config": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "modelSlug": "claude-sonnet-5.5",
          "scorePrinted": "63.2%",
          "variant": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20256,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1"
          },
          "corroborating": []
        },
        {
          "model": "Claude Fable 5.1 (max)",
          "alias": "Claude Fable 5.1",
          "lab": "Anthropic",
          "scoreDisplay": "61.0%",
          "scoreNumeric": 61,
          "date": "2026-09-28",
          "config": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "modelSlug": "claude-fable-5.1",
          "scorePrinted": "61.0%",
          "variant": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20257,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 5 (max)",
          "alias": "Claude Opus 5",
          "lab": "Anthropic",
          "scoreDisplay": "57.6%",
          "scoreNumeric": 57.6,
          "date": "2026-09-28",
          "config": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "modelSlug": "claude-opus-5",
          "scorePrinted": "57.6%",
          "variant": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20258,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5.5 (xhigh)",
          "alias": "Claude Sonnet 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "56.4%",
          "scoreNumeric": 56.4,
          "date": "2026-09-28",
          "config": "Anthropic-run pass@1 on 100 tasks; xhigh effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "modelSlug": "claude-sonnet-5.5",
          "scorePrinted": "56.4%",
          "variant": "Anthropic-run pass@1 on 100 tasks; xhigh effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20259,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 5.5 (high)",
          "alias": "Claude Sonnet 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "47.6%",
          "scoreNumeric": 47.6,
          "date": "2026-09-28",
          "config": "Anthropic-run pass@1 on 100 tasks; high effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "modelSlug": "claude-sonnet-5.5",
          "scorePrinted": "47.6%",
          "variant": "Anthropic-run pass@1 on 100 tasks; high effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20260,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.5",
          "lab": "OpenAI",
          "scoreDisplay": "46.3 ± 1.2",
          "scoreNumeric": 46.3,
          "date": "2026-05",
          "config": "pass@1; Pass^3 28.0",
          "modelSlug": "gpt-5.5",
          "scorePrinted": "46.3 ± 1.2",
          "variant": "pass@1; Pass^3 28.0",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 167,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": [
            {
              "id": 26,
              "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
              "url": "https://arxiv.org/abs/2605.02240",
              "kind": "paper",
              "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "46.3 ± 1.2",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "Claude Sonnet 5 (max)",
          "alias": "Claude Sonnet 5",
          "lab": "Anthropic",
          "scoreDisplay": "37.4%",
          "scoreNumeric": 37.4,
          "date": "2026-09-28",
          "config": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "modelSlug": "claude-sonnet-5",
          "scorePrinted": "37.4%",
          "variant": "Anthropic-run pass@1 on 100 tasks; max effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20261,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "31.7 ± 2.3",
          "scoreNumeric": 31.7,
          "date": "2026-05",
          "config": "Pass^3 18.0",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "31.7 ± 2.3",
          "variant": "Pass^3 18.0",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 168,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": [
            {
              "id": 26,
              "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
              "url": "https://arxiv.org/abs/2605.02240",
              "kind": "paper",
              "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "31.7 ± 2.3",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "Claude Sonnet 5.5 (medium)",
          "alias": "Claude Sonnet 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "30.0%",
          "scoreNumeric": 30,
          "date": "2026-09-28",
          "config": "Anthropic-run pass@1 on 100 tasks; medium effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "modelSlug": "claude-sonnet-5.5",
          "scorePrinted": "30.0%",
          "variant": "Anthropic-run pass@1 on 100 tasks; medium effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20262,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4.7",
          "lab": "Anthropic",
          "scoreDisplay": "29.3 ± 2.5",
          "scoreNumeric": 29.3,
          "date": "2026-05",
          "config": "Pass^3 18.0",
          "modelSlug": "claude-opus-4.7",
          "scorePrinted": "29.3 ± 2.5",
          "variant": "Pass^3 18.0",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 169,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": [
            {
              "id": 26,
              "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
              "url": "https://arxiv.org/abs/2605.02240",
              "kind": "paper",
              "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "29.3 ± 2.5",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "GPT-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "27.7 ± 1.5",
          "scoreNumeric": 27.7,
          "date": "2026-05",
          "config": "",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "27.7 ± 1.5",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 170,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": [
            {
              "id": 26,
              "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
              "url": "https://arxiv.org/abs/2605.02240",
              "kind": "paper",
              "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "27.7 ± 1.5",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "Claude Sonnet 5.5 (low)",
          "alias": "Claude Sonnet 5.5",
          "lab": "Anthropic",
          "scoreDisplay": "27.2%",
          "scoreNumeric": 27.2,
          "date": "2026-09-28",
          "config": "Anthropic-run pass@1 on 100 tasks; low effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "modelSlug": "claude-sonnet-5.5",
          "scorePrinted": "27.2%",
          "variant": "Anthropic-run pass@1 on 100 tasks; low effort; shared Anthropic harness; Opus 5 rubric grader; safety classifiers enabled; differs from benchmark paper protocol; run date unpublished",
          "measured": null,
          "reportedBy": "vendor",
          "reportedByLabel": "vendor-reported",
          "confidence": "verified",
          "resultId": 20263,
          "source": {
            "id": 10514,
            "title": "Claude Sonnet 5.5 System Card",
            "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
            "kind": "system_card",
            "publisher": "Anthropic",
            "firstParty": true,
            "publishedAt": "2026-09-28",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "23.0 ± 2.6",
          "scoreNumeric": 23,
          "date": "2026-05",
          "config": "",
          "modelSlug": "claude-sonnet-4.6",
          "scorePrinted": "23.0 ± 2.6",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 171,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": [
            {
              "id": 26,
              "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
              "url": "https://arxiv.org/abs/2605.02240",
              "kind": "paper",
              "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "23.0 ± 2.6",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "DeepSeek V4-Pro",
          "lab": "DeepSeek",
          "scoreDisplay": "18.7 ± 2.9",
          "scoreNumeric": 18.7,
          "date": "2026-05",
          "config": "Pass@1 over 3 runs; Pass^3 6.0; shared FHIR tool harness; up to 100 turns; high reasoning when supported",
          "modelSlug": "deepseek-v4-pro",
          "scorePrinted": "18.7 ± 2.9",
          "variant": "Pass@1 over 3 runs; Pass^3 6.0; shared FHIR tool harness; up to 100 turns; high reasoning when supported",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20217,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": []
        },
        {
          "model": "Kimi-K2.6",
          "lab": "Moonshot AI",
          "scoreDisplay": "17.0 ± 2.6",
          "scoreNumeric": 17,
          "date": "2026-05",
          "config": "open source",
          "alias": "Kimi K2.6",
          "modelSlug": "kimi-k2.6",
          "scorePrinted": "17.0 ± 2.6",
          "variant": "open source",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 172,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": [
            {
              "id": 26,
              "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
              "url": "https://arxiv.org/abs/2605.02240",
              "kind": "paper",
              "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "17.0 ± 2.6",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "MiMo-v2.5-Pro",
          "lab": "Xiaomi",
          "scoreDisplay": "16.7 ± 4.0",
          "scoreNumeric": 16.7,
          "date": "2026-05",
          "config": "Pass@1 over 3 runs; Pass^3 6.0; shared FHIR tool harness; up to 100 turns; high reasoning when supported",
          "modelSlug": "mimo-v2.5-pro",
          "scorePrinted": "16.7 ± 4.0",
          "variant": "Pass@1 over 3 runs; Pass^3 6.0; shared FHIR tool harness; up to 100 turns; high reasoning when supported",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20218,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3.6-Plus",
          "lab": "Alibaba",
          "scoreDisplay": "13.7 ± 4.0",
          "scoreNumeric": 13.7,
          "date": "2026-05",
          "config": "",
          "alias": "Qwen3.6 Plus",
          "modelSlug": "qwen3.6-plus",
          "scorePrinted": "13.7 ± 4.0",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 173,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": [
            {
              "id": 26,
              "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
              "url": "https://arxiv.org/abs/2605.02240",
              "kind": "paper",
              "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "13.7 ± 4.0",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "MiniMax M2.7",
          "lab": "MiniMax",
          "scoreDisplay": "8.7 ± 1.2",
          "scoreNumeric": 8.7,
          "date": "2026-05",
          "config": "Pass@1 over 3 runs; Pass^3 1.0; shared FHIR tool harness; up to 100 turns; high reasoning when supported",
          "modelSlug": "minimax-m2.7",
          "scorePrinted": "8.7 ± 1.2",
          "variant": "Pass@1 over 3 runs; Pass^3 1.0; shared FHIR tool harness; up to 100 turns; high reasoning when supported",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20219,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": []
        },
        {
          "model": "Gemini Pro 3.1",
          "lab": "Google",
          "scoreDisplay": "6.0 ± 1.0",
          "scoreNumeric": 6,
          "date": "2026-05",
          "config": "",
          "alias": "Gemini 3.1 Pro",
          "modelSlug": "gemini-3.1-pro",
          "scorePrinted": "6.0 ± 1.0",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 174,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)"
          },
          "corroborating": [
            {
              "id": 26,
              "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
              "url": "https://arxiv.org/abs/2605.02240",
              "kind": "paper",
              "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "6.0 ± 1.0",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "Grok-4.20",
          "lab": "xAI",
          "scoreDisplay": "5.3 ± 3.2",
          "scoreNumeric": 5.3,
          "date": "2026-05",
          "config": "",
          "alias": "Grok 4.20",
          "modelSlug": "grok-4.20",
          "scorePrinted": "5.3 ± 3.2",
          "variant": "",
          "measured": "2026-05",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 295,
          "source": {
            "id": 94,
            "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
            "url": "https://arxiv.org/pdf/2605.02240",
            "kind": "paper",
            "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
            "firstParty": true,
            "publishedAt": "2026-05-04",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Table 2 (Proprietary Models block)"
          },
          "corroborating": []
        }
      ],
      "unitShort": "100 clinical tasks",
      "publisherShort": "academic team",
      "sourceShort": "PhysicianBench paper",
      "fieldSize": 21,
      "category": "agentic",
      "confidence": "verified",
      "officialUrl": "https://arxiv.org/abs/2605.02240",
      "paperUrl": "https://arxiv.org/abs/2605.02240",
      "scaleKind": "percent",
      "summary": "Agents carrying out long-horizon physician workflows inside real EHR systems, verified by execution against those systems.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "Verified all 12 original-paper rows plus 9 vendor-reported model/effort rows from today’s Sonnet 5.5 system card, pp. 137–139. Same public task set; vendor grader/harness cohort explicitly distinguished.",
        "urls": [
          "https://arxiv.org/pdf/2605.02240",
          "https://arxiv.org/abs/2605.02240",
          "https://www.anthropic.com/claude-sonnet-5-5-system-card"
        ]
      }
    },
    {
      "slug": "ehr-complex",
      "name": "EHR-Complex",
      "publisher": "Academic team (Qiao et al., Ant Group-affiliated; arXiv 2606.23301)",
      "released": "2026-06",
      "unit": "~52,000 tasks (3,915-task test set) over 365K patients, 31 tables, 500M+ records",
      "scale": "exact-match accuracy, 0-1, higher better",
      "description": "Agentic clinical reasoning over MIMIC-IV EHR databases via SQL and Python across six clinical intents, at patient and population level with temporal evidence paths.",
      "basis": "independent-run",
      "sourceName": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
      "sourceUrl": "https://arxiv.org/pdf/2606.23301",
      "ours": null,
      "lastUpdate": "2026-06",
      "notes": "The paper reports 12 base models, 2 benchmark-specific SFT variants, and 4 commercial configurations used for human validation. All 18 are shown with configuration labels. The macro-average gives equal weight to 12 intent/scope columns; it is not the micro-average over all test cases. Rechecked against the June 2026 paper; no new runs implied.",
      "results": [
        {
          "model": "GPT-5.4 (high reasoning)",
          "lab": "OpenAI",
          "scoreDisplay": "0.65",
          "scoreNumeric": 0.65,
          "date": "2026-06",
          "config": "average over 12 intent columns; run as human-validation configuration, not in the headline 12-model table",
          "alias": "GPT-5.4",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "0.65",
          "variant": "average over 12 intent columns; run as human-validation configuration, not in the headline 12-model table",
          "measured": "2026-06",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 175,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 15, Table 10 (Strong commercial model results), Avg. column"
          },
          "corroborating": [
            {
              "id": 27,
              "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301)",
              "url": "https://arxiv.org/abs/2606.23301",
              "kind": "paper",
              "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.65",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "0.63",
          "scoreNumeric": 0.63,
          "date": "2026-06",
          "config": "validation configuration",
          "modelSlug": "gemini-3.1-pro",
          "scorePrinted": "0.63",
          "variant": "validation configuration",
          "measured": "2026-06",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 176,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 15, Table 10 (Strong commercial model results), Avg. column"
          },
          "corroborating": [
            {
              "id": 27,
              "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301)",
              "url": "https://arxiv.org/abs/2606.23301",
              "kind": "paper",
              "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.63",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "Kimi-K2.5",
          "lab": "Moonshot AI",
          "scoreDisplay": "0.62",
          "scoreNumeric": 0.62,
          "date": "2026-06",
          "config": "headline 12-model evaluation, top open-weight",
          "alias": "Kimi K2.5",
          "modelSlug": "kimi-k2.5",
          "scorePrinted": "0.62",
          "variant": "headline 12-model evaluation, top open-weight",
          "measured": "2026-06",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 177,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3 (Evaluation Results on the EHR-Complex Test Set), Avg. column"
          },
          "corroborating": [
            {
              "id": 27,
              "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301)",
              "url": "https://arxiv.org/abs/2606.23301",
              "kind": "paper",
              "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.62",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "Qwen3.5-397B",
          "lab": "Alibaba",
          "scoreDisplay": "0.62",
          "scoreNumeric": 0.62,
          "date": "2026-06",
          "config": "headline evaluation",
          "alias": "Qwen3.5 397B A17B",
          "modelSlug": "qwen3.5-397b-a17b",
          "scorePrinted": "0.62",
          "variant": "headline evaluation",
          "measured": "2026-06",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 178,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3 (Evaluation Results on the EHR-Complex Test Set), Avg. column"
          },
          "corroborating": [
            {
              "id": 27,
              "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301)",
              "url": "https://arxiv.org/abs/2606.23301",
              "kind": "paper",
              "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.62",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "DeepSeek-V3.2-Exp",
          "lab": "DeepSeek",
          "scoreDisplay": "0.59",
          "scoreNumeric": 0.59,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "deepseek-v3.2-exp",
          "scorePrinted": "0.59",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20226,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-5.4 (low reasoning)",
          "lab": "OpenAI",
          "scoreDisplay": "0.58",
          "scoreNumeric": 0.58,
          "date": "2026-06",
          "config": "validation configuration",
          "alias": "GPT-5.4",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "0.58",
          "variant": "validation configuration",
          "measured": "2026-06",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 179,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 15, Table 10 (Strong commercial model results), Avg. column"
          },
          "corroborating": [
            {
              "id": 27,
              "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301)",
              "url": "https://arxiv.org/abs/2606.23301",
              "kind": "paper",
              "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.58",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "DeepSeek-V3.1",
          "lab": "DeepSeek",
          "scoreDisplay": "0.56",
          "scoreNumeric": 0.56,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "deepseek-v3.1",
          "scorePrinted": "0.56",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20225,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3-32B-SFT",
          "lab": "Alibaba",
          "scoreDisplay": "0.55",
          "scoreNumeric": 0.55,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns; fine-tuned on trajectories from EHR-Complex training set",
          "modelSlug": "qwen3-32b-sft",
          "scorePrinted": "0.55",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns; fine-tuned on trajectories from EHR-Complex training set",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20231,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3-235B",
          "lab": "Alibaba",
          "scoreDisplay": "0.53",
          "scoreNumeric": 0.53,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "qwen3-235b",
          "scorePrinted": "0.53",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20224,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-4.1 mini",
          "lab": "OpenAI",
          "scoreDisplay": "0.49",
          "scoreNumeric": 0.49,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "gpt-4.1-mini",
          "scorePrinted": "0.49",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20221,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-4.1",
          "lab": "OpenAI",
          "scoreDisplay": "0.47",
          "scoreNumeric": 0.47,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "gpt-4.1",
          "scorePrinted": "0.47",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20222,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3-14B-SFT",
          "lab": "Alibaba",
          "scoreDisplay": "0.45",
          "scoreNumeric": 0.45,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns; fine-tuned on trajectories from EHR-Complex training set",
          "modelSlug": "qwen3-14b-sft",
          "scorePrinted": "0.45",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns; fine-tuned on trajectories from EHR-Complex training set",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20229,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "0.36",
          "scoreNumeric": 0.36,
          "date": "2026-06",
          "config": "validation configuration",
          "modelSlug": "claude-sonnet-4.6",
          "scorePrinted": "0.36",
          "variant": "validation configuration",
          "measured": "2026-06",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 180,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 15, Table 10 (Strong commercial model results), Avg. column"
          },
          "corroborating": [
            {
              "id": 27,
              "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301)",
              "url": "https://arxiv.org/abs/2606.23301",
              "kind": "paper",
              "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "0.36",
              "quote": null,
              "locator": "arXiv abs page (landing page for the PDF)"
            }
          ]
        },
        {
          "model": "Qwen3-32B",
          "lab": "Alibaba",
          "scoreDisplay": "0.36",
          "scoreNumeric": 0.36,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "qwen3-32b",
          "scorePrinted": "0.36",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20230,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "GPT-4o",
          "lab": "OpenAI",
          "scoreDisplay": "0.31",
          "scoreNumeric": 0.31,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "gpt-4o",
          "scorePrinted": "0.31",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20220,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Pro",
          "lab": "Google",
          "scoreDisplay": "0.31",
          "scoreNumeric": 0.31,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "gemini-2.5-pro",
          "scorePrinted": "0.31",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20223,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3-14B",
          "lab": "Alibaba",
          "scoreDisplay": "0.30",
          "scoreNumeric": 0.3,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "qwen3-14b",
          "scorePrinted": "0.30",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20228,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        },
        {
          "model": "Qwen3-4B",
          "lab": "Alibaba",
          "scoreDisplay": "0.16",
          "scoreNumeric": 0.16,
          "date": "2026-06",
          "config": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "modelSlug": "qwen3-4b",
          "scorePrinted": "0.16",
          "variant": "Table 3; macro-average across 12 intent/scope columns; temperature 0; up to 50 SQL/Python interaction turns",
          "measured": null,
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20227,
          "source": {
            "id": 109,
            "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
            "url": "https://arxiv.org/pdf/2606.23301",
            "kind": "paper",
            "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
            "firstParty": true,
            "publishedAt": "2026-06-22",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 6, Table 3, Avg. column"
          },
          "corroborating": []
        }
      ],
      "unitShort": "3,915-task test set",
      "publisherShort": "academic team",
      "sourceShort": "EHR-Complex paper",
      "fieldSize": 18,
      "category": "agentic",
      "confidence": "verified",
      "officialUrl": "https://arxiv.org/abs/2606.23301",
      "paperUrl": "https://arxiv.org/abs/2606.23301",
      "scaleKind": "fraction",
      "summary": "Agentic clinical reasoning over MIMIC-IV records through SQL and Python, at patient and population level.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "Tables 3 and 10 verified: 18 configurations, including both benchmark-specific SFT variants. Current arXiv version remains v1 dated 2026-06-22.",
        "urls": [
          "https://arxiv.org/pdf/2606.23301",
          "https://arxiv.org/abs/2606.23301"
        ]
      }
    },
    {
      "slug": "whbench",
      "name": "WHBench",
      "publisher": "Independent researchers (Maurya, Govindgari, Kumar)",
      "released": "2026-04",
      "unit": "47 scenarios / 3,100 scored responses across 22 models",
      "scale": "mean normalized percentage 0-100, higher better",
      "description": "Women's health: 47 expert-crafted scenarios across 10 topics graded on a 23-criterion rubric for clinical accuracy, safety, equity, and guideline adherence; targets failure modes like outdated guidelines, unsafe omissions, dosing errors, equity blind spots.",
      "basis": "independent-run",
      "sourceName": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
      "sourceUrl": "https://arxiv.org/pdf/2604.00024v2",
      "ours": null,
      "lastUpdate": "2026-03",
      "notes": "An academic study with expert validation rather than a live leaderboard; the model set was frozen in March 2026, before GPT-5.6 and the Claude 5 family shipped.",
      "results": [
        {
          "model": "Claude Opus 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "72.1%",
          "scoreNumeric": 72.1,
          "date": "2026-03",
          "config": "95% CI 69.6-74.4; evaluations run March 2026",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "72.1%",
          "variant": "95% CI 69.6-74.4; evaluations run March 2026",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 181,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": [
            {
              "id": 28,
              "title": "WHBench (arXiv 2604.00024 abstract page)",
              "url": "https://arxiv.org/abs/2604.00024",
              "kind": "paper",
              "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "72.1%",
              "quote": null,
              "locator": "same paper, abstract page"
            }
          ]
        },
        {
          "model": "Claude Sonnet 4.6",
          "lab": "Anthropic",
          "scoreDisplay": "67.1%",
          "scoreNumeric": 67.1,
          "date": "2026-03",
          "config": "95% CI 64.5-69.6",
          "modelSlug": "claude-sonnet-4.6",
          "scorePrinted": "67.1%",
          "variant": "95% CI 64.5-69.6",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 182,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": [
            {
              "id": 28,
              "title": "WHBench (arXiv 2604.00024 abstract page)",
              "url": "https://arxiv.org/abs/2604.00024",
              "kind": "paper",
              "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "67.1%",
              "quote": null,
              "locator": "same paper, abstract page"
            }
          ]
        },
        {
          "model": "GPT-5.4",
          "lab": "OpenAI",
          "scoreDisplay": "66.8%",
          "scoreNumeric": 66.8,
          "date": "2026-03",
          "config": "95% CI 64.5-69.2",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "66.8%",
          "variant": "95% CI 64.5-69.2",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 183,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": [
            {
              "id": 28,
              "title": "WHBench (arXiv 2604.00024 abstract page)",
              "url": "https://arxiv.org/abs/2604.00024",
              "kind": "paper",
              "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "66.8%",
              "quote": null,
              "locator": "same paper, abstract page"
            }
          ]
        },
        {
          "model": "Gemini 3 Flash Preview",
          "lab": "Google",
          "scoreDisplay": "64.7%",
          "scoreNumeric": 64.7,
          "date": "2026-03",
          "config": "",
          "alias": "Gemini 3 Flash",
          "modelSlug": "gemini-3-flash",
          "scorePrinted": "64.7%",
          "variant": "",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 184,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": [
            {
              "id": 28,
              "title": "WHBench (arXiv 2604.00024 abstract page)",
              "url": "https://arxiv.org/abs/2604.00024",
              "kind": "paper",
              "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "64.7%",
              "quote": null,
              "locator": "same paper, abstract page"
            }
          ]
        },
        {
          "model": "OpenAI o3",
          "lab": "OpenAI",
          "scoreDisplay": "63.6%",
          "scoreNumeric": 63.6,
          "date": "2026-03",
          "config": "95% bootstrap CI 61.3–65.9; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "openai-o3",
          "scorePrinted": "63.6%",
          "variant": "95% bootstrap CI 61.3–65.9; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20232,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek V3.2",
          "lab": "DeepSeek",
          "scoreDisplay": "61.3%",
          "scoreNumeric": 61.3,
          "date": "2026-03",
          "config": "95% bootstrap CI 58.6–63.9; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "deepseek-v3.2",
          "scorePrinted": "61.3%",
          "variant": "95% bootstrap CI 58.6–63.9; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20233,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Grok 3",
          "lab": "SpaceX AI",
          "scoreDisplay": "60.7%",
          "scoreNumeric": 60.7,
          "date": "2026-03",
          "config": "95% bootstrap CI 58.0–63.4; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "grok-3",
          "scorePrinted": "60.7%",
          "variant": "95% bootstrap CI 58.0–63.4; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20234,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Mistral Large",
          "lab": "Mistral AI",
          "scoreDisplay": "60.2%",
          "scoreNumeric": 60.2,
          "date": "2026-03",
          "config": "95% bootstrap CI 57.4–63.0; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "mistral-large",
          "scorePrinted": "60.2%",
          "variant": "95% bootstrap CI 57.4–63.0; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20235,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Grok 4",
          "lab": "SpaceX AI",
          "scoreDisplay": "57.9%",
          "scoreNumeric": 57.9,
          "date": "2026-03",
          "config": "95% bootstrap CI 54.9–60.8; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "grok-4",
          "scorePrinted": "57.9%",
          "variant": "95% bootstrap CI 54.9–60.8; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20236,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "DeepSeek-R1",
          "lab": "DeepSeek",
          "scoreDisplay": "52.9%",
          "scoreNumeric": 52.9,
          "date": "2026-03",
          "config": "95% bootstrap CI 50.5–55.3; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "deepseek-r1",
          "scorePrinted": "52.9%",
          "variant": "95% bootstrap CI 50.5–55.3; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20237,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-4.1",
          "lab": "OpenAI",
          "scoreDisplay": "51.8%",
          "scoreNumeric": 51.8,
          "date": "2026-03",
          "config": "",
          "modelSlug": "gpt-4.1",
          "scorePrinted": "51.8%",
          "variant": "",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 185,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": [
            {
              "id": 28,
              "title": "WHBench (arXiv 2604.00024 abstract page)",
              "url": "https://arxiv.org/abs/2604.00024",
              "kind": "paper",
              "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "51.8%",
              "quote": null,
              "locator": "same paper, abstract page"
            }
          ]
        },
        {
          "model": "Grok 3 Mini",
          "lab": "SpaceX AI",
          "scoreDisplay": "50.0%",
          "scoreNumeric": 50,
          "date": "2026-03",
          "config": "95% bootstrap CI 47.5–52.5; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "grok-3-mini",
          "scorePrinted": "50.0%",
          "variant": "95% bootstrap CI 47.5–52.5; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20238,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Flash",
          "lab": "Google",
          "scoreDisplay": "49.5%",
          "scoreNumeric": 49.5,
          "date": "2026-03",
          "config": "95% bootstrap CI 47.0–52.0; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "gemini-2.5-flash",
          "scorePrinted": "49.5%",
          "variant": "95% bootstrap CI 47.0–52.0; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20239,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Claude Opus 4",
          "lab": "Anthropic",
          "scoreDisplay": "49.1%",
          "scoreNumeric": 49.1,
          "date": "2026-03",
          "config": "95% bootstrap CI 46.4–51.7; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "claude-opus-4",
          "scorePrinted": "49.1%",
          "variant": "95% bootstrap CI 46.4–51.7; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20240,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Claude Sonnet 4",
          "lab": "Anthropic",
          "scoreDisplay": "48.1%",
          "scoreNumeric": 48.1,
          "date": "2026-03",
          "config": "95% bootstrap CI 45.5–50.6; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "claude-sonnet-4",
          "scorePrinted": "48.1%",
          "variant": "95% bootstrap CI 45.5–50.6; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20241,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "GPT-4o",
          "lab": "OpenAI",
          "scoreDisplay": "44.6%",
          "scoreNumeric": 44.6,
          "date": "2026-03",
          "config": "",
          "modelSlug": "gpt-4o",
          "scorePrinted": "44.6%",
          "variant": "",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 186,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": [
            {
              "id": 28,
              "title": "WHBench (arXiv 2604.00024 abstract page)",
              "url": "https://arxiv.org/abs/2604.00024",
              "kind": "paper",
              "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
              "firstParty": true,
              "role": "corroborating",
              "scoreDisplay": "44.6%",
              "quote": null,
              "locator": "same paper, abstract page"
            }
          ]
        },
        {
          "model": "Llama 4 Maverick",
          "lab": "Meta",
          "scoreDisplay": "42.1%",
          "scoreNumeric": 42.1,
          "date": "2026-03",
          "config": "95% bootstrap CI 39.6–44.6; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "llama-4-maverick",
          "scorePrinted": "42.1%",
          "variant": "95% bootstrap CI 39.6–44.6; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20242,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Nemotron 70B",
          "lab": "NVIDIA",
          "scoreDisplay": "39.3%",
          "scoreNumeric": 39.3,
          "date": "2026-03",
          "config": "95% bootstrap CI 37.3–41.3; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "nemotron-70b",
          "scorePrinted": "39.3%",
          "variant": "95% bootstrap CI 37.3–41.3; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20243,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Llama 3.3 70B",
          "lab": "Meta",
          "scoreDisplay": "37.8%",
          "scoreNumeric": 37.8,
          "date": "2026-03",
          "config": "95% bootstrap CI 35.2–40.5; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "llama-3.3-70b",
          "scorePrinted": "37.8%",
          "variant": "95% bootstrap CI 35.2–40.5; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20244,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Llama 3.1 405B",
          "lab": "Meta",
          "scoreDisplay": "36.1%",
          "scoreNumeric": 36.1,
          "date": "2026-03",
          "config": "95% bootstrap CI 33.9–38.3; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "llama-3.1-405b",
          "scorePrinted": "36.1%",
          "variant": "95% bootstrap CI 33.9–38.3; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20245,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Gemini 2.5 Pro",
          "lab": "Google",
          "scoreDisplay": "35.3%",
          "scoreNumeric": 35.3,
          "date": "2026-03",
          "config": "95% bootstrap CI 32.7–38.1; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "gemini-2.5-pro",
          "scorePrinted": "35.3%",
          "variant": "95% bootstrap CI 32.7–38.1; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20246,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        },
        {
          "model": "Llama 4 Scout",
          "lab": "Meta",
          "scoreDisplay": "35.2%",
          "scoreNumeric": 35.2,
          "date": "2026-03",
          "config": "95% bootstrap CI 33.2–37.3; 3 runs; temperature 0; zero-shot, closed-book",
          "modelSlug": "llama-4-scout",
          "scorePrinted": "35.2%",
          "variant": "95% bootstrap CI 33.2–37.3; 3 runs; temperature 0; zero-shot, closed-book",
          "measured": "2026-03",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 20247,
          "source": {
            "id": 45,
            "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
            "url": "https://arxiv.org/pdf/2604.00024v2",
            "kind": "paper",
            "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
            "firstParty": true,
            "publishedAt": "2026-07-23",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)"
          },
          "corroborating": []
        }
      ],
      "unitShort": "47 scenarios",
      "publisherShort": "academic team",
      "sourceShort": "WHBench paper",
      "fieldSize": 22,
      "category": "rubric",
      "confidence": "verified",
      "officialUrl": "https://arxiv.org/abs/2604.00024",
      "paperUrl": "https://arxiv.org/abs/2604.00024",
      "scaleKind": "percent",
      "summary": "Women's health scenarios graded on a 23-criterion rubric for clinical accuracy, safety, equity, and guideline adherence.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "All 22 model scores in Table 3 checked against v2, revised 2026-07-23. Historical March 2026 evaluation set retained; no live refresh claimed.",
        "urls": [
          "https://arxiv.org/pdf/2604.00024v2",
          "https://arxiv.org/abs/2604.00024"
        ]
      }
    },
    {
      "slug": "healthadminbench",
      "name": "HealthAdminBench",
      "publisher": "Kinetic Systems (with Stanford Hospital domain experts)",
      "released": "2026-04",
      "unit": "135 tasks / 1,698 rubric-scored subtasks",
      "scale": "percentage end-to-end task success 0-100, higher better",
      "description": "End-to-end task success of computer-use LLM agents on healthcare administration workflows: prior authorizations, denial appeals, and DME ordering; success requires completing every subtask in a task.",
      "basis": "independent-run",
      "sourceName": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
      "sourceUrl": "https://arxiv.org/pdf/2604.09937",
      "ours": null,
      "lastUpdate": "2026-04",
      "notes": "Seven agent configurations are compared using screenshots plus task description and portal guidance. Native computer-use agents and the standardized harness have separate rows. The paper also reports accessibility-tree settings, which must not be compared directly with these screenshot-only scores. No revised paper or newer scored configuration found on the official site.",
      "results": [
        {
          "model": "Claude Opus 4.6 (computer-use agent)",
          "lab": "Anthropic",
          "scoreDisplay": "36.3%",
          "scoreNumeric": 36.3,
          "date": "2026-04",
          "config": "screenshot-only, task description + portal guidance; native CUA harness; subtask rate 78.4%",
          "alias": "Claude Opus 4.6",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "36.3%",
          "variant": "screenshot-only, task description + portal guidance; native CUA harness; subtask rate 78.4%",
          "measured": "2026-04",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 328,
          "source": {
            "id": 138,
            "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
            "url": "https://arxiv.org/pdf/2604.09937",
            "kind": "paper",
            "publisher": "Bedi, Welch, Steinberg et al. (Stanford University / Kinetic Systems / Stanford Health Care)",
            "firstParty": true,
            "publishedAt": "2026-04-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)"
          },
          "corroborating": [
            {
              "id": 29,
              "title": "Introducing HealthAdminBench: AI Agents Can Diagnose Rare Diseases, But Can They Handle Your Insurance? (Kinetic Systems blog)",
              "url": "https://kineticsystems.ai/blog/healthadminbench-automating-healthcare-administration-with-computer-use-agents",
              "kind": "blog",
              "publisher": "Kinetic Systems",
              "firstParty": false,
              "role": "corroborating",
              "scoreDisplay": "36.3%",
              "quote": null,
              "locator": "Section 'LLMs struggle with long-horizon tasks'"
            }
          ]
        },
        {
          "model": "GPT-5.4 (computer-use agent)",
          "lab": "OpenAI",
          "scoreDisplay": "26.7%",
          "scoreNumeric": 26.7,
          "date": "2026-04",
          "config": "screenshot-only, task description + portal guidance; subtask rate 82.8%",
          "alias": "GPT-5.4",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "26.7%",
          "variant": "screenshot-only, task description + portal guidance; subtask rate 82.8%",
          "measured": "2026-04",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 188,
          "source": {
            "id": 138,
            "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
            "url": "https://arxiv.org/pdf/2604.09937",
            "kind": "paper",
            "publisher": "Bedi, Welch, Steinberg et al. (Stanford University / Kinetic Systems / Stanford Health Care)",
            "firstParty": true,
            "publishedAt": "2026-04-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)"
          },
          "corroborating": [
            {
              "id": 29,
              "title": "Introducing HealthAdminBench: AI Agents Can Diagnose Rare Diseases, But Can They Handle Your Insurance? (Kinetic Systems blog)",
              "url": "https://kineticsystems.ai/blog/healthadminbench-automating-healthcare-administration-with-computer-use-agents",
              "kind": "blog",
              "publisher": "Kinetic Systems",
              "firstParty": false,
              "role": "corroborating",
              "scoreDisplay": "26.7%",
              "quote": null,
              "locator": "Section 'LLMs struggle with long-horizon tasks'"
            }
          ]
        },
        {
          "model": "Kimi K2.5",
          "lab": "Moonshot AI",
          "scoreDisplay": "15.6%",
          "scoreNumeric": 15.6,
          "date": "2026-04",
          "config": "screenshot-only, task description + portal guidance",
          "modelSlug": "kimi-k2.5",
          "scorePrinted": "15.6%",
          "variant": "screenshot-only, task description + portal guidance",
          "measured": "2026-04",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 189,
          "source": {
            "id": 138,
            "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
            "url": "https://arxiv.org/pdf/2604.09937",
            "kind": "paper",
            "publisher": "Bedi, Welch, Steinberg et al. (Stanford University / Kinetic Systems / Stanford Health Care)",
            "firstParty": true,
            "publishedAt": "2026-04-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)"
          },
          "corroborating": [
            {
              "id": 29,
              "title": "Introducing HealthAdminBench: AI Agents Can Diagnose Rare Diseases, But Can They Handle Your Insurance? (Kinetic Systems blog)",
              "url": "https://kineticsystems.ai/blog/healthadminbench-automating-healthcare-administration-with-computer-use-agents",
              "kind": "blog",
              "publisher": "Kinetic Systems",
              "firstParty": false,
              "role": "corroborating",
              "scoreDisplay": "15.6%",
              "quote": null,
              "locator": "Section 'LLMs struggle with long-horizon tasks'"
            }
          ]
        },
        {
          "model": "Claude Opus 4.6 (standardized harness)",
          "lab": "Anthropic",
          "scoreDisplay": "14.8%",
          "scoreNumeric": 14.8,
          "date": "2026-04",
          "config": "screenshot-only, task description + portal guidance; authors' standardized harness, no native CUA",
          "alias": "Claude Opus 4.6",
          "modelSlug": "claude-opus-4.6",
          "scorePrinted": "14.8%",
          "variant": "screenshot-only, task description + portal guidance; authors' standardized harness, no native CUA",
          "measured": "2026-04",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 330,
          "source": {
            "id": 138,
            "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
            "url": "https://arxiv.org/pdf/2604.09937",
            "kind": "paper",
            "publisher": "Bedi, Welch, Steinberg et al. (Stanford University / Kinetic Systems / Stanford Health Care)",
            "firstParty": true,
            "publishedAt": "2026-04-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)"
          },
          "corroborating": []
        },
        {
          "model": "Qwen 3.5",
          "lab": "Alibaba",
          "scoreDisplay": "13.3%",
          "scoreNumeric": 13.3,
          "date": "2026-04",
          "config": "screenshot-only, task description + portal guidance",
          "modelSlug": "qwen-3.5",
          "scorePrinted": "13.3%",
          "variant": "screenshot-only, task description + portal guidance",
          "measured": "2026-04",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 190,
          "source": {
            "id": 138,
            "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
            "url": "https://arxiv.org/pdf/2604.09937",
            "kind": "paper",
            "publisher": "Bedi, Welch, Steinberg et al. (Stanford University / Kinetic Systems / Stanford Health Care)",
            "firstParty": true,
            "publishedAt": "2026-04-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)"
          },
          "corroborating": [
            {
              "id": 29,
              "title": "Introducing HealthAdminBench: AI Agents Can Diagnose Rare Diseases, But Can They Handle Your Insurance? (Kinetic Systems blog)",
              "url": "https://kineticsystems.ai/blog/healthadminbench-automating-healthcare-administration-with-computer-use-agents",
              "kind": "blog",
              "publisher": "Kinetic Systems",
              "firstParty": false,
              "role": "corroborating",
              "scoreDisplay": "13.3%",
              "quote": null,
              "locator": "Section 'LLMs struggle with long-horizon tasks'"
            }
          ]
        },
        {
          "model": "Gemini 3.1 Pro",
          "lab": "Google",
          "scoreDisplay": "11.9%",
          "scoreNumeric": 11.9,
          "date": "2026-04",
          "config": "screenshot-only, task description + portal guidance",
          "modelSlug": "gemini-3.1-pro",
          "scorePrinted": "11.9%",
          "variant": "screenshot-only, task description + portal guidance",
          "measured": "2026-04",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 191,
          "source": {
            "id": 138,
            "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
            "url": "https://arxiv.org/pdf/2604.09937",
            "kind": "paper",
            "publisher": "Bedi, Welch, Steinberg et al. (Stanford University / Kinetic Systems / Stanford Health Care)",
            "firstParty": true,
            "publishedAt": "2026-04-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)"
          },
          "corroborating": [
            {
              "id": 29,
              "title": "Introducing HealthAdminBench: AI Agents Can Diagnose Rare Diseases, But Can They Handle Your Insurance? (Kinetic Systems blog)",
              "url": "https://kineticsystems.ai/blog/healthadminbench-automating-healthcare-administration-with-computer-use-agents",
              "kind": "blog",
              "publisher": "Kinetic Systems",
              "firstParty": false,
              "role": "corroborating",
              "scoreDisplay": "11.9%",
              "quote": null,
              "locator": "Section 'LLMs struggle with long-horizon tasks'"
            }
          ]
        },
        {
          "model": "GPT-5.4 (standardized harness)",
          "lab": "OpenAI",
          "scoreDisplay": "5.9%",
          "scoreNumeric": 5.9,
          "date": "2026-04",
          "config": "screenshot-only, task description + portal guidance; authors' standardized harness, no native CUA",
          "alias": "GPT-5.4",
          "modelSlug": "gpt-5.4",
          "scorePrinted": "5.9%",
          "variant": "screenshot-only, task description + portal guidance; authors' standardized harness, no native CUA",
          "measured": "2026-04",
          "reportedBy": "benchmark_owner",
          "reportedByLabel": "benchmark publisher",
          "confidence": "verified",
          "resultId": 332,
          "source": {
            "id": 138,
            "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
            "url": "https://arxiv.org/pdf/2604.09937",
            "kind": "paper",
            "publisher": "Bedi, Welch, Steinberg et al. (Stanford University / Kinetic Systems / Stanford Health Care)",
            "firstParty": true,
            "publishedAt": "2026-04-10",
            "retrieved": "2026-09-28",
            "quote": null,
            "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)"
          },
          "corroborating": []
        }
      ],
      "unitShort": "135 admin tasks",
      "publisherShort": "Kinetic Systems",
      "sourceShort": "HealthAdminBench paper",
      "fieldSize": 7,
      "category": "agentic",
      "confidence": "verified",
      "officialUrl": "https://healthadminbench.stanford.edu/",
      "paperUrl": "https://arxiv.org/abs/2604.09937",
      "scaleKind": "percent",
      "summary": "Computer-use agents completing healthcare administration workflows: prior authorizations, denial appeals, and DME ordering.",
      "verification": {
        "checkedAt": "2026-09-28",
        "status": "verified",
        "summary": "All seven Figure 3(a) task-success rates verified against v1 and official project page. The paper remains dated 2026-04-10; source checks are not new evaluations.",
        "urls": [
          "https://arxiv.org/pdf/2604.09937",
          "https://arxiv.org/abs/2604.09937",
          "https://healthadminbench.stanford.edu/"
        ]
      }
    }
  ],
  "retired": [
    {
      "name": "HealthBench Consensus",
      "reason": "near-saturated physician-consensus baseline; frontier runs stopped reporting it separately"
    },
    {
      "name": "MedQA / MultiMedQA",
      "reason": "exam-style multiple choice, saturated above 95 percent since 2025; archived by its trackers"
    },
    {
      "name": "AgentClinic",
      "reason": "no public frontier-model results since 2025"
    },
    {
      "name": "CRAFT-MD",
      "reason": "no public frontier-model results since 2025"
    },
    {
      "name": "MedAgentBench",
      "reason": "v2 lives on inside the MAST composite; the standalone board has no current frontier rows"
    },
    {
      "name": "SDBench / MAI-DxO",
      "reason": "Microsoft's 2025 sequential-diagnosis study was not re-run on current models"
    },
    {
      "name": "Open Medical-LLM Leaderboard (Hugging Face)",
      "reason": "built on saturated exam sets; no frontier submissions in 2026"
    },
    {
      "name": "MedArena",
      "reason": "clinician preference arena; ratings pool too thin on current frontier models to quote"
    },
    {
      "name": "AMIE evaluations",
      "reason": "Google DeepMind research prototypes, never opened to cross-vendor comparison"
    },
    {
      "name": "LiveClin, PrIME-LLM, MedMCP-Calc",
      "reason": "single studies with two or fewer current-frontier rows; tracked for a future qualifying update"
    }
  ],
  "sources": [
    {
      "id": 16,
      "title": "GPT-5.6 Preview System Card",
      "url": "https://deploymentsafety.openai.com/gpt-5-6-preview/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "healthbench-professional:GPT-5.6 Sol",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-SOL",
          "scoreDisplay": "60.5"
        },
        {
          "key": "healthbench-professional:GPT-5.6 Terra",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-TERRA",
          "scoreDisplay": "57.7"
        },
        {
          "key": "healthbench-professional:GPT-5.6 Luna",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-LUNA",
          "scoreDisplay": "55.7"
        },
        {
          "key": "healthbench-professional:GPT-5.5",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.5",
          "scoreDisplay": "51.8"
        },
        {
          "key": "healthbench-professional:GPT-5.4",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.4",
          "scoreDisplay": "48.1"
        },
        {
          "key": "healthbench-professional:GPT-5",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5",
          "scoreDisplay": "46.2"
        },
        {
          "key": "healthbench-professional:GPT-5.2",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.2",
          "scoreDisplay": "45.9"
        },
        {
          "key": "healthbench-professional:GPT-5.1",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.1",
          "scoreDisplay": "39.6"
        },
        {
          "key": "healthbench-hard:GPT-5",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5",
          "scoreDisplay": "34.7"
        },
        {
          "key": "healthbench-hard:GPT-5.2",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.2",
          "scoreDisplay": "34.3"
        },
        {
          "key": "healthbench-hard:GPT-5.6 Sol",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-SOL",
          "scoreDisplay": "33.1"
        },
        {
          "key": "healthbench-hard:GPT-5.6 Terra",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-TERRA",
          "scoreDisplay": "32.7"
        },
        {
          "key": "healthbench-hard:GPT-5.6 Luna",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-LUNA",
          "scoreDisplay": "32.0"
        },
        {
          "key": "healthbench-hard:GPT-5.5",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.5",
          "scoreDisplay": "31.5"
        },
        {
          "key": "healthbench-hard:GPT-5.4",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.4",
          "scoreDisplay": "29.1"
        },
        {
          "key": "healthbench-hard:GPT-5.1",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.1",
          "scoreDisplay": "25.4"
        },
        {
          "key": "healthbench:GPT-5.6 Sol",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-SOL",
          "scoreDisplay": "57.0"
        },
        {
          "key": "healthbench:GPT-5.6 Terra",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-TERRA",
          "scoreDisplay": "57.0"
        },
        {
          "key": "healthbench:GPT-5.5",
          "role": "corroborating",
          "quote": null,
          "locator": null,
          "scoreDisplay": "56.5"
        },
        {
          "key": "healthbench:GPT-5.6 Luna",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)) (preview card, published 2026-06-26), column GPT-5.6-LUNA",
          "scoreDisplay": "55.8"
        },
        {
          "key": "healthbench:GPT-5.6 Sol (August)",
          "role": "corroborating",
          "quote": null,
          "locator": null,
          "scoreDisplay": "55.0"
        }
      ]
    },
    {
      "id": 17,
      "title": "MAST: Medical AI Superintelligence Test leaderboard (General board)",
      "url": "https://arise-ai.org/mast",
      "kind": "official_leaderboard",
      "publisher": "ARISE AI Research Network",
      "firstParty": true,
      "publishedAt": "2026-08-15",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "mast:GPT-5.6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')",
          "scoreDisplay": "60.2%"
        },
        {
          "key": "mast:Kimi K3",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')",
          "scoreDisplay": "60.1%"
        },
        {
          "key": "mast:Gemini 3.6 Flash",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')",
          "scoreDisplay": "59.3%"
        },
        {
          "key": "mast:Gemini 3.1 Pro",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')",
          "scoreDisplay": "58.9%"
        },
        {
          "key": "mast:Qwen3.5 397B A17B",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')",
          "scoreDisplay": "57.9%"
        },
        {
          "key": "mast:Claude Opus 5",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')",
          "scoreDisplay": "57.1%"
        },
        {
          "key": "mast:Claude Sonnet 5",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')",
          "scoreDisplay": "56.6%"
        },
        {
          "key": "mast:Grok 4.3",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast, 'Which AI can you trust for medical questions?' General tab, composite score table (8 of 11 models shown; 'Last updated August 15, 2026')",
          "scoreDisplay": "53.7%"
        }
      ]
    },
    {
      "id": 18,
      "title": "MedHELM leaderboard (medhelm.org), v5.0.0",
      "url": "https://medhelm.org/",
      "kind": "official_leaderboard",
      "publisher": "Stanford CRFM (MedHELM)",
      "firstParty": true,
      "publishedAt": "2026-05-14",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "medhelm:Gemini 3.1 Pro (Preview)",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.652"
        },
        {
          "key": "medhelm:Gemini 3.5 Flash",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.642"
        },
        {
          "key": "medhelm:Muse Spark (2026-04-08)",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.621"
        },
        {
          "key": "medhelm:GPT-5.4 mini",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.552"
        },
        {
          "key": "medhelm:GPT-5.4 (2026-03-05)",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.538"
        },
        {
          "key": "medhelm:Gemini 2.5 Pro",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.529"
        },
        {
          "key": "medhelm:DeepSeek R1",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.485"
        },
        {
          "key": "medhelm:Claude 4.6 Opus",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.456"
        },
        {
          "key": "medhelm:Claude 3.7 Sonnet",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.45"
        },
        {
          "key": "medhelm:Gemini 2.0 Flash",
          "role": "primary",
          "quote": null,
          "locator": "medhelm.org home, 'Current leaders  Mean win rate  v5.0.0' table ('10 of 11 models · Updated 14 May 2026')",
          "scoreDisplay": "0.342"
        }
      ]
    },
    {
      "id": 19,
      "title": "MAST technical leaderboard (First Do NOHARM v2 and per-benchmark results)",
      "url": "https://arise-ai.org/mast/technical",
      "kind": "official_leaderboard",
      "publisher": "ARISE AI Research Network",
      "firstParty": true,
      "publishedAt": "2026-08-15",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "first-do-noharm:Muse Spark 1.1",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)",
          "scoreDisplay": "79.7%"
        },
        {
          "key": "first-do-noharm:Claude Opus 5",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column",
          "scoreDisplay": "74.6%"
        },
        {
          "key": "first-do-noharm:Kimi K3",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column",
          "scoreDisplay": "74.0%"
        },
        {
          "key": "first-do-noharm:GPT-5.6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column",
          "scoreDisplay": "70.1%"
        },
        {
          "key": "first-do-noharm:GPT-5.5",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column",
          "scoreDisplay": "70.0%"
        },
        {
          "key": "first-do-noharm:GPT-5",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, Model Leaderboard (Top 10 shown), row 10, SAFETY column",
          "scoreDisplay": "68.6%"
        },
        {
          "key": "first-do-noharm:Claude Fable 5",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)",
          "scoreDisplay": "65.0%"
        },
        {
          "key": "first-do-noharm:Gemini 3.1 Pro",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view), plus Model Leaderboard SAFETY column",
          "scoreDisplay": "62.6%"
        },
        {
          "key": "first-do-noharm:Gemini 2.5 Pro",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)",
          "scoreDisplay": "61.9%"
        },
        {
          "key": "first-do-noharm:Qwen3.5 397B A17B",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)",
          "scoreDisplay": "61.1%"
        },
        {
          "key": "first-do-noharm:Kimi K2.6",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)",
          "scoreDisplay": "59.1%"
        },
        {
          "key": "first-do-noharm:DeepSeek R1",
          "role": "primary",
          "quote": null,
          "locator": "arise-ai.org/mast/technical, 'First Do NOHARM v2 overall metric across 19 models' ranking (Latest Flagships view)",
          "scoreDisplay": "55.8%"
        }
      ]
    },
    {
      "id": 20,
      "title": "HealthAgentBench leaderboard",
      "url": "https://microsoft.github.io/HealthAgentBench/",
      "kind": "official_leaderboard",
      "publisher": "Microsoft Research (HealthAgentBench)",
      "firstParty": true,
      "publishedAt": "2026-07-27",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthagentbench:Claude Code (Opus 5)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 1, Success Rate column",
          "scoreDisplay": "55%"
        },
        {
          "key": "healthagentbench:Codex (GPT-5.6-sol)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 2, Success Rate column",
          "scoreDisplay": "45%"
        },
        {
          "key": "healthagentbench:Codex (GPT 5.5)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 3, Success Rate column",
          "scoreDisplay": "42%"
        },
        {
          "key": "healthagentbench:Copilot (Opus 4.8)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 4, Success Rate column",
          "scoreDisplay": "36%"
        },
        {
          "key": "healthagentbench:Copilot (GPT 5.5)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 5, Success Rate column",
          "scoreDisplay": "35%"
        },
        {
          "key": "healthagentbench:Claude Code (Opus 4.8)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 6, Success Rate column",
          "scoreDisplay": "32%"
        },
        {
          "key": "healthagentbench:Codex (GPT 5.4)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 7, Success Rate column",
          "scoreDisplay": "28%"
        },
        {
          "key": "healthagentbench:Claude Code (Opus 4.7)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 8, Success Rate column",
          "scoreDisplay": "27%"
        },
        {
          "key": "healthagentbench:Codex (GPT 5.3)",
          "role": "primary",
          "quote": null,
          "locator": "Homepage leaderboard, rank 9, success rate and cost/task",
          "scoreDisplay": "22%"
        },
        {
          "key": "healthagentbench:Claude Code (Opus 4.6)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 10, Success Rate column",
          "scoreDisplay": "19%"
        },
        {
          "key": "healthagentbench:Claude Code (Sonnet 4.6)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 11, Success Rate column",
          "scoreDisplay": "17%"
        },
        {
          "key": "healthagentbench:Codex (GPT 5.4 Mini)",
          "role": "primary",
          "quote": null,
          "locator": "Leaderboard table (homepage), rank 12, Success Rate column",
          "scoreDisplay": "16%"
        }
      ]
    },
    {
      "id": 21,
      "title": "CHI-Bench leaderboard (actAVA)",
      "url": "https://actava.ai/benchmarks/leaderboards",
      "kind": "official_leaderboard",
      "publisher": "actAVA",
      "firstParty": true,
      "publishedAt": "2026-08-12",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "chi-bench:erius + claude-opus-5",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 1, Accuracy column; submission date 2026-07-26",
          "scoreDisplay": "54.7%"
        },
        {
          "key": "chi-bench:erius + claude-opus-4-8",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 2, Accuracy column; submission date 2026-06-05",
          "scoreDisplay": "37.3%"
        },
        {
          "key": "chi-bench:claude-code + claude-opus-5",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 3, Accuracy column; submission date 2026-07-24",
          "scoreDisplay": "37.3%"
        },
        {
          "key": "chi-bench:claude-code + claude-opus-4-8",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 4, Accuracy column; submission date 2026-05-28",
          "scoreDisplay": "33.3%"
        },
        {
          "key": "chi-bench:claude-code + claude-opus-4-6",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 5, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "28.0%"
        },
        {
          "key": "chi-bench:claude-code + claude-sonnet-4-6",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 6, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "26.2%"
        },
        {
          "key": "chi-bench:codex + gpt-5.6-sol",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 7, Accuracy column; submission date 2026-07-24",
          "scoreDisplay": "25.3%"
        },
        {
          "key": "chi-bench:openai-agents + kimi-k3",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 8, Accuracy column; submission date 2026-07-24",
          "scoreDisplay": "25.3%"
        },
        {
          "key": "chi-bench:claude-code + claude-opus-4-7",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 9, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "24.4%"
        },
        {
          "key": "chi-bench:claude-code + claude-fable-5",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 10, Accuracy column; submission date 2026-07-22",
          "scoreDisplay": "24.0%"
        },
        {
          "key": "chi-bench:hermes + MedGuard",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 11, Accuracy column; submission date 2026-07-06",
          "scoreDisplay": "22.7%"
        },
        {
          "key": "chi-bench:codex + gpt-5.5",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 12, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "20.9%"
        },
        {
          "key": "chi-bench:claude-code + claude-sonnet-5",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 13, Accuracy column; submission date 2026-07-06",
          "scoreDisplay": "20.0%"
        },
        {
          "key": "chi-bench:openai-agents + glm-5.1",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 14, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "18.7%"
        },
        {
          "key": "chi-bench:hermes + glm-5.1",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 15, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "18.7%"
        },
        {
          "key": "chi-bench:openai-agents + glm-5.2",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 16, Accuracy column; submission date 2026-07-06",
          "scoreDisplay": "18.7%"
        },
        {
          "key": "chi-bench:openclaw + claude-opus-4-7",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 17, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "17.3%"
        },
        {
          "key": "chi-bench:openclaw + glm-5.1",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 18, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "16.9%"
        },
        {
          "key": "chi-bench:hermes + qwen-3.6-max",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 19, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "16.4%"
        },
        {
          "key": "chi-bench:codex + gpt-5.4",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 20, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "16.0%"
        },
        {
          "key": "chi-bench:openai-agents + qwen-3.6-max",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 21, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "15.6%"
        },
        {
          "key": "chi-bench:hermes + kimi-k2.6",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 22, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "15.6%"
        },
        {
          "key": "chi-bench:openai-agents + kimi-k2.6",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 23, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "15.1%"
        },
        {
          "key": "chi-bench:openai-agents + deepseek-v4-pro",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 24, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "14.2%"
        },
        {
          "key": "chi-bench:hermes + deepseek-v4-pro",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 25, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "13.8%"
        },
        {
          "key": "chi-bench:codex + gpt-5.6-terra",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 26, Accuracy column; submission date 2026-07-24",
          "scoreDisplay": "13.3%"
        },
        {
          "key": "chi-bench:codex + gpt-5.6-luna",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 27, Accuracy column; submission date 2026-07-24",
          "scoreDisplay": "13.3%"
        },
        {
          "key": "chi-bench:gemini-cli + gemini-3-flash",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 28, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "12.5%"
        },
        {
          "key": "chi-bench:openclaw + deepseek-v4-pro",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 29, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "11.1%"
        },
        {
          "key": "chi-bench:deepagents + glm-5.1",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 30, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "11.1%"
        },
        {
          "key": "chi-bench:deepagents + deepseek-v4-pro",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 31, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "10.7%"
        },
        {
          "key": "chi-bench:openclaw + kimi-k2.6",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 32, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "10.2%"
        },
        {
          "key": "chi-bench:deepagents + qwen-3.6-max",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 33, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "9.3%"
        },
        {
          "key": "chi-bench:codex + gpt-5.4-mini",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 34, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "8.4%"
        },
        {
          "key": "chi-bench:openai-agents + TML Inkling 256K",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 35, Accuracy column; submission date 2026-07-24",
          "scoreDisplay": "8.0%"
        },
        {
          "key": "chi-bench:gemini-cli + gemini-3.1-pro",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 36, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "7.1%"
        },
        {
          "key": "chi-bench:claude-code + claude-haiku-4-5",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 37, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "6.2%"
        },
        {
          "key": "chi-bench:openai-agents + grok-4.3",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 38, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "5.8%"
        },
        {
          "key": "chi-bench:openclaw + qwen-3.6-max",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 39, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "4.9%"
        },
        {
          "key": "chi-bench:hermes + grok-4.3",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 40, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "4.4%"
        },
        {
          "key": "chi-bench:deepagents + kimi-k2.6",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 41, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "3.1%"
        },
        {
          "key": "chi-bench:deepagents + grok-4.3",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 42, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "2.2%"
        },
        {
          "key": "chi-bench:openclaw + grok-4.3",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 43, Accuracy column; submission date 2026-05-01",
          "scoreDisplay": "0.4%"
        },
        {
          "key": "chi-bench:openai-agents + Nemotron 3 Ultra 256K",
          "role": "primary",
          "quote": null,
          "locator": "All Domains leaderboard, rank 44, Accuracy column; submission date 2026-07-24",
          "scoreDisplay": "0.0%"
        }
      ]
    },
    {
      "id": 22,
      "title": "Vals AI MedCode leaderboard",
      "url": "https://www.vals.ai/benchmarks/medcode",
      "kind": "official_leaderboard",
      "publisher": "Vals AI",
      "firstParty": true,
      "publishedAt": "2026-09-26",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "medcode:Claude Opus 5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-5\"].accuracy; Overall leaderboard rank 1",
          "scoreDisplay": "63.57%"
        },
        {
          "key": "medcode:Gemini 3.1 Pro Preview (02/26)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.1-pro-preview\"].accuracy; Overall leaderboard rank 2",
          "scoreDisplay": "59.06%"
        },
        {
          "key": "medcode:Claude Fable 5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-fable-5\"].accuracy; Overall leaderboard rank 3",
          "scoreDisplay": "56.07%"
        },
        {
          "key": "medcode:Gemini 3 Flash (12/25)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3-flash-preview\"].accuracy; Overall leaderboard rank 4",
          "scoreDisplay": "55.92%"
        },
        {
          "key": "medcode:Gemini 3.5 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.5-flash\"].accuracy; Overall leaderboard rank 5",
          "scoreDisplay": "55.83%"
        },
        {
          "key": "medcode:Claude Opus 4.7",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-7\"].accuracy; Overall leaderboard rank 6",
          "scoreDisplay": "54.86%"
        },
        {
          "key": "medcode:Claude Fable 5.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-fable-5-1\"].accuracy; Overall leaderboard rank 7",
          "scoreDisplay": "53.51%"
        },
        {
          "key": "medcode:Gemini 3.7 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.7-flash\"].accuracy; Overall leaderboard rank 8",
          "scoreDisplay": "53.39%"
        },
        {
          "key": "medcode:Claude Opus 4.8",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-8\"].accuracy; Overall leaderboard rank 9",
          "scoreDisplay": "53.22%"
        },
        {
          "key": "medcode:Gemini 3.6 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.6-flash\"].accuracy; Overall leaderboard rank 10",
          "scoreDisplay": "53.15%"
        },
        {
          "key": "medcode:Claude Sonnet 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-5-5\"].accuracy; Overall leaderboard rank 11",
          "scoreDisplay": "52.92%"
        },
        {
          "key": "medcode:GPT 5.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.1-2025-11-13\"].accuracy; Overall leaderboard rank 12",
          "scoreDisplay": "52.73%"
        },
        {
          "key": "medcode:Gemini 3 Pro (11/25)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3-pro-preview\"].accuracy; Overall leaderboard rank 13",
          "scoreDisplay": "52.20%"
        },
        {
          "key": "medcode:Muse Spark",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark\"].accuracy; Overall leaderboard rank 14",
          "scoreDisplay": "51.31%"
        },
        {
          "key": "medcode:Gemini 2.5 Pro",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-pro\"].accuracy; Overall leaderboard rank 15",
          "scoreDisplay": "50.59%"
        },
        {
          "key": "medcode:Claude Opus 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-5-5\"].accuracy; Overall leaderboard rank 16",
          "scoreDisplay": "49.80%"
        },
        {
          "key": "medcode:GPT 5.2",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.2-2025-12-11\"].accuracy; Overall leaderboard rank 17",
          "scoreDisplay": "49.75%"
        },
        {
          "key": "medcode:GPT 5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-2025-08-07\"].accuracy; Overall leaderboard rank 18",
          "scoreDisplay": "49.63%"
        },
        {
          "key": "medcode:Grok 4.7",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.7\"].accuracy; Overall leaderboard rank 19",
          "scoreDisplay": "49.55%"
        },
        {
          "key": "medcode:Muse Spark 1.2",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark_1_2\"].accuracy; Overall leaderboard rank 20",
          "scoreDisplay": "49.35%"
        },
        {
          "key": "medcode:Claude Opus 4.5 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-5-20251101-thinking\"].accuracy; Overall leaderboard rank 21",
          "scoreDisplay": "49.16%"
        },
        {
          "key": "medcode:Claude Opus 4.6 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-6-thinking\"].accuracy; Overall leaderboard rank 22",
          "scoreDisplay": "49.13%"
        },
        {
          "key": "medcode:GPT 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.5\"].accuracy; Overall leaderboard rank 23",
          "scoreDisplay": "49.10%"
        },
        {
          "key": "medcode:Kimi K3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k3\"].accuracy; Overall leaderboard rank 24",
          "scoreDisplay": "48.88%"
        },
        {
          "key": "medcode:GPT-6 Astra",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-astra\"].accuracy; Overall leaderboard rank 25",
          "scoreDisplay": "48.49%"
        },
        {
          "key": "medcode:Claude Opus 4.6 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-6\"].accuracy; Overall leaderboard rank 26",
          "scoreDisplay": "48.24%"
        },
        {
          "key": "medcode:Gemini 3.8 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.8-flash\"].accuracy; Overall leaderboard rank 27",
          "scoreDisplay": "48.13%"
        },
        {
          "key": "medcode:Gemini 3.1 Flash Lite Preview",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.1-flash-lite-preview\"].accuracy; Overall leaderboard rank 28",
          "scoreDisplay": "47.60%"
        },
        {
          "key": "medcode:Claude Sonnet 5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-5\"].accuracy; Overall leaderboard rank 29",
          "scoreDisplay": "47.54%"
        },
        {
          "key": "medcode:o3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/o3-2025-04-16\"].accuracy; Overall leaderboard rank 30",
          "scoreDisplay": "47.29%"
        },
        {
          "key": "medcode:Claude Opus 4.1 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-1-20250805-thinking\"].accuracy; Overall leaderboard rank 31",
          "scoreDisplay": "47.23%"
        },
        {
          "key": "medcode:GPT-6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-sol\"].accuracy; Overall leaderboard rank 32",
          "scoreDisplay": "47.07%"
        },
        {
          "key": "medcode:MiniMax-M3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M3\"].accuracy; Overall leaderboard rank 33",
          "scoreDisplay": "46.29%"
        },
        {
          "key": "medcode:Claude Opus 4.5 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-5-20251101\"].accuracy; Overall leaderboard rank 34",
          "scoreDisplay": "45.17%"
        },
        {
          "key": "medcode:MiMo V2.6 Pro",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.6-pro\"].accuracy; Overall leaderboard rank 35",
          "scoreDisplay": "44.97%"
        },
        {
          "key": "medcode:Grok 4.6",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.6\"].accuracy; Overall leaderboard rank 36",
          "scoreDisplay": "44.71%"
        },
        {
          "key": "medcode:GPT-6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-luna\"].accuracy; Overall leaderboard rank 37",
          "scoreDisplay": "44.69%"
        },
        {
          "key": "medcode:Claude Sonnet 4.5 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-5-20250929-thinking\"].accuracy; Overall leaderboard rank 38",
          "scoreDisplay": "44.13%"
        },
        {
          "key": "medcode:GPT-5.6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-sol\"].accuracy; Overall leaderboard rank 39",
          "scoreDisplay": "43.97%"
        },
        {
          "key": "medcode:Gemini 3.5 Flash Lite",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.5-flash-lite\"].accuracy; Overall leaderboard rank 40",
          "scoreDisplay": "43.49%"
        },
        {
          "key": "medcode:GPT-5.6 Terra",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-terra\"].accuracy; Overall leaderboard rank 41",
          "scoreDisplay": "43.41%"
        },
        {
          "key": "medcode:Grok 4.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.5\"].accuracy; Overall leaderboard rank 42",
          "scoreDisplay": "43.29%"
        },
        {
          "key": "medcode:Hy4 Preview",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"tencent/hy4-preview\"].accuracy; Overall leaderboard rank 43",
          "scoreDisplay": "43.25%"
        },
        {
          "key": "medcode:GPT 5 Mini",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-mini-2025-08-07\"].accuracy; Overall leaderboard rank 44",
          "scoreDisplay": "43.05%"
        },
        {
          "key": "medcode:GLM 5.3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.3\"].accuracy; Overall leaderboard rank 45",
          "scoreDisplay": "42.86%"
        },
        {
          "key": "medcode:DeepSeek V4 Pro 0813",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-pro-0813\"].accuracy; Overall leaderboard rank 46",
          "scoreDisplay": "42.47%"
        },
        {
          "key": "medcode:GPT-5.6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-luna\"].accuracy; Overall leaderboard rank 47",
          "scoreDisplay": "42.39%"
        },
        {
          "key": "medcode:GLM 5.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.1\"].accuracy; Overall leaderboard rank 48",
          "scoreDisplay": "41.60%"
        },
        {
          "key": "medcode:DeepSeek V4 Flash 0731",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-flash-0731\"].accuracy; Overall leaderboard rank 49",
          "scoreDisplay": "41.41%"
        },
        {
          "key": "medcode:Claude Opus 4.1 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-1-20250805\"].accuracy; Overall leaderboard rank 50",
          "scoreDisplay": "41.37%"
        },
        {
          "key": "medcode:GPT 5.4 (xhigh)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.4-2026-03-05\"].accuracy; Overall leaderboard rank 51",
          "scoreDisplay": "41.29%"
        },
        {
          "key": "medcode:Inkling",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"thinkingmachines/inkling\"].accuracy; Overall leaderboard rank 52",
          "scoreDisplay": "41.19%"
        },
        {
          "key": "medcode:DeepSeek V4.1 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4.1-flash\"].accuracy; Overall leaderboard rank 53",
          "scoreDisplay": "41.17%"
        },
        {
          "key": "medcode:MiMo V2.6 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.6-flash\"].accuracy; Overall leaderboard rank 54",
          "scoreDisplay": "41.06%"
        },
        {
          "key": "medcode:GPT 5.4 Nano",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.4-nano-2026-03-17\"].accuracy; Overall leaderboard rank 55",
          "scoreDisplay": "41.03%"
        },
        {
          "key": "medcode:GLM 5.2",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.2\"].accuracy; Overall leaderboard rank 56",
          "scoreDisplay": "40.77%"
        },
        {
          "key": "medcode:Qwen 3.8 Max",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.8-max\"].accuracy; Overall leaderboard rank 57",
          "scoreDisplay": "40.67%"
        },
        {
          "key": "medcode:Claude Sonnet 4.5 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-5-20250929\"].accuracy; Overall leaderboard rank 58",
          "scoreDisplay": "40.57%"
        },
        {
          "key": "medcode:Gemini 2.5 Flash Preview (9/25) (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-preview-09-2025\"].accuracy; Overall leaderboard rank 59",
          "scoreDisplay": "40.54%"
        },
        {
          "key": "medcode:DeepSeek V4",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-pro\"].accuracy; Overall leaderboard rank 60",
          "scoreDisplay": "40.45%"
        },
        {
          "key": "medcode:Gemini 2.5 Flash (7/17) (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-thinking\"].accuracy; Overall leaderboard rank 61",
          "scoreDisplay": "40.36%"
        },
        {
          "key": "medcode:Gemini 2.5 Flash Preview (9/25) (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-preview-09-2025-thinking\"].accuracy; Overall leaderboard rank 62",
          "scoreDisplay": "40.33%"
        },
        {
          "key": "medcode:Kimi K2.6",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k2.6\"].accuracy; Overall leaderboard rank 63",
          "scoreDisplay": "40.14%"
        },
        {
          "key": "medcode:Kimi K2.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k2.5-thinking\"].accuracy; Overall leaderboard rank 64",
          "scoreDisplay": "39.32%"
        },
        {
          "key": "medcode:Qwen 3.7 Max",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.7-max\"].accuracy; Overall leaderboard rank 65",
          "scoreDisplay": "38.75%"
        },
        {
          "key": "medcode:Nemotron 3 Ultra",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"nvidia/nemotron-3-ultra-550b-a55b\"].accuracy; Overall leaderboard rank 66",
          "scoreDisplay": "38.62%"
        },
        {
          "key": "medcode:Gemini 2.5 Flash (7/17) (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash\"].accuracy; Overall leaderboard rank 67",
          "scoreDisplay": "38.42%"
        },
        {
          "key": "medcode:Grok 4",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-0709\"].accuracy; Overall leaderboard rank 68",
          "scoreDisplay": "38.08%"
        },
        {
          "key": "medcode:Grok 4.3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.3\"].accuracy; Overall leaderboard rank 69",
          "scoreDisplay": "38.07%"
        },
        {
          "key": "medcode:Inkling Small",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"thinkingmachines/inkling-small\"].accuracy; Overall leaderboard rank 70",
          "scoreDisplay": "37.89%"
        },
        {
          "key": "medcode:Grok 4 Fast (Reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-fast-reasoning\"].accuracy; Overall leaderboard rank 71",
          "scoreDisplay": "37.38%"
        },
        {
          "key": "medcode:Qwen 3.6 Plus",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.6-plus\"].accuracy; Overall leaderboard rank 72",
          "scoreDisplay": "36.89%"
        },
        {
          "key": "medcode:Llama 4 Maverick",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"fireworks/llama4-maverick-instruct-basic\"].accuracy; Overall leaderboard rank 73",
          "scoreDisplay": "36.51%"
        },
        {
          "key": "medcode:Claude Sonnet 4 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-20250514-thinking\"].accuracy; Overall leaderboard rank 74",
          "scoreDisplay": "34.96%"
        },
        {
          "key": "medcode:MiniMax-M2.7",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M2.7\"].accuracy; Overall leaderboard rank 75",
          "scoreDisplay": "34.44%"
        },
        {
          "key": "medcode:Gemini 2.5 Flash Lite (9/25) (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite-preview-09-2025-thinking\"].accuracy; Overall leaderboard rank 76",
          "scoreDisplay": "34.19%"
        },
        {
          "key": "medcode:MiniMax-M2.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M2.1\"].accuracy; Overall leaderboard rank 77",
          "scoreDisplay": "34.08%"
        },
        {
          "key": "medcode:Claude Sonnet 4 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-20250514\"].accuracy; Overall leaderboard rank 78",
          "scoreDisplay": "33.94%"
        },
        {
          "key": "medcode:o4 Mini",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/o4-mini-2025-04-16\"].accuracy; Overall leaderboard rank 79",
          "scoreDisplay": "33.79%"
        },
        {
          "key": "medcode:Mistral Medium 3.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"mistralai/mistral-medium-3.5\"].accuracy; Overall leaderboard rank 80",
          "scoreDisplay": "33.75%"
        },
        {
          "key": "medcode:Qwen 3.5 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.5-flash\"].accuracy; Overall leaderboard rank 81",
          "scoreDisplay": "33.00%"
        },
        {
          "key": "medcode:GLM 4.7",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-4.7\"].accuracy; Overall leaderboard rank 82",
          "scoreDisplay": "32.77%"
        },
        {
          "key": "medcode:Claude Haiku 4.5 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-haiku-4-5-20251001-thinking\"].accuracy; Overall leaderboard rank 83",
          "scoreDisplay": "32.68%"
        },
        {
          "key": "medcode:MiMo V2.5 Pro",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.5-pro\"].accuracy; Overall leaderboard rank 84",
          "scoreDisplay": "32.48%"
        },
        {
          "key": "medcode:Ling 3.0 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"ant/ling-3.0-flash-2607\"].accuracy; Overall leaderboard rank 85",
          "scoreDisplay": "32.27%"
        },
        {
          "key": "medcode:Grok 4.20 (Reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.20-0309-reasoning\"].accuracy; Overall leaderboard rank 86",
          "scoreDisplay": "32.16%"
        },
        {
          "key": "medcode:MiMo V2.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.5\"].accuracy; Overall leaderboard rank 87",
          "scoreDisplay": "31.89%"
        },
        {
          "key": "medcode:Qwen 3 VL Plus",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3-vl-plus-2025-09-23\"].accuracy; Overall leaderboard rank 88",
          "scoreDisplay": "31.65%"
        },
        {
          "key": "medcode:Qwen 3 Max Thinking",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3-max-2026-01-23\"].accuracy; Overall leaderboard rank 89",
          "scoreDisplay": "31.37%"
        },
        {
          "key": "medcode:Mercury 2.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"inception/mercury-2.5\"].accuracy; Overall leaderboard rank 90",
          "scoreDisplay": "31.33%"
        },
        {
          "key": "medcode:GPT 5 Nano",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-nano-2025-08-07\"].accuracy; Overall leaderboard rank 91",
          "scoreDisplay": "30.44%"
        },
        {
          "key": "medcode:Grok 4 Fast (Non-Reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-fast-non-reasoning\"].accuracy; Overall leaderboard rank 92",
          "scoreDisplay": "30.04%"
        },
        {
          "key": "medcode:Ling 3.0 Flash Fin",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"ant/ling-3.0-flash-af-rc3\"].accuracy; Overall leaderboard rank 93",
          "scoreDisplay": "29.30%"
        },
        {
          "key": "medcode:Qwen 3.8 27B",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.8-27b\"].accuracy; Overall leaderboard rank 94",
          "scoreDisplay": "28.70%"
        },
        {
          "key": "medcode:Grok 4.1 Fast Non-Reasoning",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-1-fast-non-reasoning\"].accuracy; Overall leaderboard rank 95",
          "scoreDisplay": "28.35%"
        },
        {
          "key": "medcode:Grok 4.1 Fast (Reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-1-fast-reasoning\"].accuracy; Overall leaderboard rank 96",
          "scoreDisplay": "28.08%"
        },
        {
          "key": "medcode:Gemini 2.5 Flash Lite (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite\"].accuracy; Overall leaderboard rank 97",
          "scoreDisplay": "27.11%"
        },
        {
          "key": "medcode:Gemini 2.5 Flash Lite (9/25) (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite-preview-09-2025\"].accuracy; Overall leaderboard rank 98",
          "scoreDisplay": "27.08%"
        },
        {
          "key": "medcode:Llama 4 Scout",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"together/meta-llama/Llama-4-Scout-17B-16E-Instruct\"].accuracy; Overall leaderboard rank 99",
          "scoreDisplay": "23.31%"
        },
        {
          "key": "medcode:Laguna M.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"poolside/laguna-m.1\"].accuracy; Overall leaderboard rank 100",
          "scoreDisplay": "23.11%"
        },
        {
          "key": "medcode:Laguna XS.2",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"poolside/laguna-xs.2\"].accuracy; Overall leaderboard rank 101",
          "scoreDisplay": "21.25%"
        },
        {
          "key": "medcode:Command A+",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"cohere/command-a-plus-05-2026\"].accuracy; Overall leaderboard rank 102",
          "scoreDisplay": "19.72%"
        }
      ]
    },
    {
      "id": 23,
      "title": "Vals AI MedScribe leaderboard",
      "url": "https://www.vals.ai/benchmarks/medscribe",
      "kind": "official_leaderboard",
      "publisher": "Vals AI",
      "firstParty": true,
      "publishedAt": "2026-09-26",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "medscribe:Claude Opus 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-5-5\"].accuracy; Overall leaderboard rank 1",
          "scoreDisplay": "91.43%"
        },
        {
          "key": "medscribe:Claude Fable 5.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-fable-5-1\"].accuracy; Overall leaderboard rank 2",
          "scoreDisplay": "91.29%"
        },
        {
          "key": "medscribe:Claude Sonnet 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-5-5\"].accuracy; Overall leaderboard rank 3",
          "scoreDisplay": "91.10%"
        },
        {
          "key": "medscribe:Claude Opus 5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-5\"].accuracy; Overall leaderboard rank 4",
          "scoreDisplay": "90.98%"
        },
        {
          "key": "medscribe:Muse Spark 1.2",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark_1_2\"].accuracy; Overall leaderboard rank 5",
          "scoreDisplay": "90.06%"
        },
        {
          "key": "medscribe:Grok 4.7",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.7\"].accuracy; Overall leaderboard rank 6",
          "scoreDisplay": "89.38%"
        },
        {
          "key": "medscribe:GLM 5.3 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.3-flash\"].accuracy; Overall leaderboard rank 7",
          "scoreDisplay": "88.94%"
        },
        {
          "key": "medscribe:Muse Spark 1.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark_1_1\"].accuracy; Overall leaderboard rank 8",
          "scoreDisplay": "88.89%"
        },
        {
          "key": "medscribe:GLM 5.3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.3\"].accuracy; Overall leaderboard rank 9",
          "scoreDisplay": "88.81%"
        },
        {
          "key": "medscribe:Claude Fable 5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-fable-5\"].accuracy; Overall leaderboard rank 10",
          "scoreDisplay": "88.52%"
        },
        {
          "key": "medscribe:MiMo V2.6 Pro",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.6-pro\"].accuracy; Overall leaderboard rank 11",
          "scoreDisplay": "88.31%"
        },
        {
          "key": "medscribe:GPT 5.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.1-2025-11-13\"].accuracy; Overall leaderboard rank 12",
          "scoreDisplay": "88.09%"
        },
        {
          "key": "medscribe:Kimi K3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k3\"].accuracy; Overall leaderboard rank 13",
          "scoreDisplay": "87.96%"
        },
        {
          "key": "medscribe:GPT-6 Astra",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-astra\"].accuracy; Overall leaderboard rank 14",
          "scoreDisplay": "87.91%"
        },
        {
          "key": "medscribe:MiniMax-M3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M3\"].accuracy; Overall leaderboard rank 15",
          "scoreDisplay": "87.25%"
        },
        {
          "key": "medscribe:Grok 4.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.5\"].accuracy; Overall leaderboard rank 16",
          "scoreDisplay": "86.88%"
        },
        {
          "key": "medscribe:GPT 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.5\"].accuracy; Overall leaderboard rank 17",
          "scoreDisplay": "86.87%"
        },
        {
          "key": "medscribe:Claude Opus 4.6 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-6\"].accuracy; Overall leaderboard rank 18",
          "scoreDisplay": "86.74%"
        },
        {
          "key": "medscribe:Grok 4.6",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.6\"].accuracy; Overall leaderboard rank 19",
          "scoreDisplay": "86.53%"
        },
        {
          "key": "medscribe:Claude Opus 4.6 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-6-thinking\"].accuracy; Overall leaderboard rank 20",
          "scoreDisplay": "86.13%"
        },
        {
          "key": "medscribe:Muse Spark",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"meta/muse_spark\"].accuracy; Overall leaderboard rank 21",
          "scoreDisplay": "85.90%"
        },
        {
          "key": "medscribe:Claude Opus 4.8",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-8\"].accuracy; Overall leaderboard rank 22",
          "scoreDisplay": "85.75%"
        },
        {
          "key": "medscribe:DeepSeek V4.1 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4.1-flash\"].accuracy; Overall leaderboard rank 23",
          "scoreDisplay": "85.50%"
        },
        {
          "key": "medscribe:Inkling",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"thinkingmachines/inkling\"].accuracy; Overall leaderboard rank 24",
          "scoreDisplay": "85.41%"
        },
        {
          "key": "medscribe:Claude Opus 4.5 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-5-20251101-thinking\"].accuracy; Overall leaderboard rank 25",
          "scoreDisplay": "85.32%"
        },
        {
          "key": "medscribe:MiMo V2.6 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.6-flash\"].accuracy; Overall leaderboard rank 26",
          "scoreDisplay": "85.28%"
        },
        {
          "key": "medscribe:GPT-5.6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-sol\"].accuracy; Overall leaderboard rank 27",
          "scoreDisplay": "85.23%"
        },
        {
          "key": "medscribe:Claude Haiku 4.5 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-haiku-4-5-20251001-thinking\"].accuracy; Overall leaderboard rank 28",
          "scoreDisplay": "85.23%"
        },
        {
          "key": "medscribe:Qwen 3.8 Max",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.8-max\"].accuracy; Overall leaderboard rank 29",
          "scoreDisplay": "84.95%"
        },
        {
          "key": "medscribe:Claude Sonnet 4.5 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-5-20250929\"].accuracy; Overall leaderboard rank 30",
          "scoreDisplay": "84.52%"
        },
        {
          "key": "medscribe:Gemini 3.8 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.8-flash\"].accuracy; Overall leaderboard rank 31",
          "scoreDisplay": "84.50%"
        },
        {
          "key": "medscribe:GPT-5.6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-luna\"].accuracy; Overall leaderboard rank 32",
          "scoreDisplay": "84.39%"
        },
        {
          "key": "medscribe:GPT 5.2",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.2-2025-12-11\"].accuracy; Overall leaderboard rank 33",
          "scoreDisplay": "84.39%"
        },
        {
          "key": "medscribe:Inkling Small",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"thinkingmachines/inkling-small\"].accuracy; Overall leaderboard rank 34",
          "scoreDisplay": "84.11%"
        },
        {
          "key": "medscribe:Claude Sonnet 4.5 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-5-20250929-thinking\"].accuracy; Overall leaderboard rank 35",
          "scoreDisplay": "84.10%"
        },
        {
          "key": "medscribe:Gemini 3.7 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.7-flash\"].accuracy; Overall leaderboard rank 36",
          "scoreDisplay": "83.94%"
        },
        {
          "key": "medscribe:Qwen 3.8 27B",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.8-27b\"].accuracy; Overall leaderboard rank 37",
          "scoreDisplay": "83.85%"
        },
        {
          "key": "medscribe:MiMo V2.5 Pro",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.5-pro\"].accuracy; Overall leaderboard rank 38",
          "scoreDisplay": "83.73%"
        },
        {
          "key": "medscribe:GPT-6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-luna\"].accuracy; Overall leaderboard rank 39",
          "scoreDisplay": "83.71%"
        },
        {
          "key": "medscribe:GPT 5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-2025-08-07\"].accuracy; Overall leaderboard rank 40",
          "scoreDisplay": "83.65%"
        },
        {
          "key": "medscribe:Hy4 Preview",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"tencent/hy4-preview\"].accuracy; Overall leaderboard rank 41",
          "scoreDisplay": "83.60%"
        },
        {
          "key": "medscribe:GLM 5.2",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.2\"].accuracy; Overall leaderboard rank 42",
          "scoreDisplay": "83.53%"
        },
        {
          "key": "medscribe:Claude Opus 4.5 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-5-20251101\"].accuracy; Overall leaderboard rank 43",
          "scoreDisplay": "83.25%"
        },
        {
          "key": "medscribe:Gemini 2.5 Flash (7/17) (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-thinking\"].accuracy; Overall leaderboard rank 44",
          "scoreDisplay": "82.98%"
        },
        {
          "key": "medscribe:Claude Opus 4.7",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-7\"].accuracy; Overall leaderboard rank 45",
          "scoreDisplay": "82.95%"
        },
        {
          "key": "medscribe:Gemini 2.5 Flash (7/17) (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash\"].accuracy; Overall leaderboard rank 46",
          "scoreDisplay": "82.87%"
        },
        {
          "key": "medscribe:GPT-5.6 Terra",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.6-terra\"].accuracy; Overall leaderboard rank 47",
          "scoreDisplay": "82.87%"
        },
        {
          "key": "medscribe:GPT-6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-6-sol\"].accuracy; Overall leaderboard rank 48",
          "scoreDisplay": "82.03%"
        },
        {
          "key": "medscribe:Grok 4 Fast (Reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-fast-reasoning\"].accuracy; Overall leaderboard rank 49",
          "scoreDisplay": "81.63%"
        },
        {
          "key": "medscribe:Ling 3.0 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"ant/ling-3.0-flash-2607\"].accuracy; Overall leaderboard rank 50",
          "scoreDisplay": "80.90%"
        },
        {
          "key": "medscribe:MiniMax-M2.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M2.1\"].accuracy; Overall leaderboard rank 51",
          "scoreDisplay": "80.78%"
        },
        {
          "key": "medscribe:GPT 5 Mini",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-mini-2025-08-07\"].accuracy; Overall leaderboard rank 52",
          "scoreDisplay": "80.58%"
        },
        {
          "key": "medscribe:DeepSeek V4 Flash 0731",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-flash-0731\"].accuracy; Overall leaderboard rank 53",
          "scoreDisplay": "80.36%"
        },
        {
          "key": "medscribe:DeepSeek V4 Pro 0813",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-pro-0813\"].accuracy; Overall leaderboard rank 54",
          "scoreDisplay": "80.17%"
        },
        {
          "key": "medscribe:MiniMax-M2.7",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"minimax/MiniMax-M2.7\"].accuracy; Overall leaderboard rank 55",
          "scoreDisplay": "79.87%"
        },
        {
          "key": "medscribe:Grok 4 Fast (Non-Reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-fast-non-reasoning\"].accuracy; Overall leaderboard rank 56",
          "scoreDisplay": "79.72%"
        },
        {
          "key": "medscribe:Gemini 3.6 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.6-flash\"].accuracy; Overall leaderboard rank 57",
          "scoreDisplay": "79.66%"
        },
        {
          "key": "medscribe:Qwen 3.7 Max",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.7-max\"].accuracy; Overall leaderboard rank 58",
          "scoreDisplay": "79.40%"
        },
        {
          "key": "medscribe:Grok 4.1 Fast (Reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-1-fast-reasoning\"].accuracy; Overall leaderboard rank 59",
          "scoreDisplay": "78.73%"
        },
        {
          "key": "medscribe:Gemini 2.5 Flash Preview (9/25) (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-preview-09-2025-thinking\"].accuracy; Overall leaderboard rank 60",
          "scoreDisplay": "78.50%"
        },
        {
          "key": "medscribe:Grok 4",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-0709\"].accuracy; Overall leaderboard rank 61",
          "scoreDisplay": "78.15%"
        },
        {
          "key": "medscribe:Kimi K2.6",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k2.6\"].accuracy; Overall leaderboard rank 62",
          "scoreDisplay": "78.15%"
        },
        {
          "key": "medscribe:Gemini 2.5 Flash Preview (9/25) (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-preview-09-2025\"].accuracy; Overall leaderboard rank 63",
          "scoreDisplay": "77.95%"
        },
        {
          "key": "medscribe:GPT 5.4 (xhigh)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.4-2026-03-05\"].accuracy; Overall leaderboard rank 64",
          "scoreDisplay": "77.55%"
        },
        {
          "key": "medscribe:Grok 4.1 Fast Non-Reasoning",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4-1-fast-non-reasoning\"].accuracy; Overall leaderboard rank 65",
          "scoreDisplay": "77.46%"
        },
        {
          "key": "medscribe:Qwen 3 VL Plus",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3-vl-plus-2025-09-23\"].accuracy; Overall leaderboard rank 66",
          "scoreDisplay": "77.13%"
        },
        {
          "key": "medscribe:GPT 5.4 Nano",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5.4-nano-2026-03-17\"].accuracy; Overall leaderboard rank 67",
          "scoreDisplay": "77.09%"
        },
        {
          "key": "medscribe:Qwen 3.6 Plus",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.6-plus\"].accuracy; Overall leaderboard rank 68",
          "scoreDisplay": "76.96%"
        },
        {
          "key": "medscribe:o3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/o3-2025-04-16\"].accuracy; Overall leaderboard rank 69",
          "scoreDisplay": "76.65%"
        },
        {
          "key": "medscribe:Gemini 3.5 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.5-flash\"].accuracy; Overall leaderboard rank 70",
          "scoreDisplay": "76.57%"
        },
        {
          "key": "medscribe:Kimi K2.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"kimi/kimi-k2.5-thinking\"].accuracy; Overall leaderboard rank 71",
          "scoreDisplay": "76.44%"
        },
        {
          "key": "medscribe:Gemini 3.1 Pro Preview (02/26)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.1-pro-preview\"].accuracy; Overall leaderboard rank 72",
          "scoreDisplay": "76.11%"
        },
        {
          "key": "medscribe:Claude Sonnet 5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-5\"].accuracy; Overall leaderboard rank 73",
          "scoreDisplay": "76.05%"
        },
        {
          "key": "medscribe:Gemini 2.5 Flash Lite (9/25) (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite-preview-09-2025\"].accuracy; Overall leaderboard rank 74",
          "scoreDisplay": "75.82%"
        },
        {
          "key": "medscribe:Ling 3.0 Flash Fin",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"ant/ling-3.0-flash-af-rc3\"].accuracy; Overall leaderboard rank 75",
          "scoreDisplay": "75.59%"
        },
        {
          "key": "medscribe:DeepSeek V4",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"deepseek/deepseek-v4-pro\"].accuracy; Overall leaderboard rank 76",
          "scoreDisplay": "75.14%"
        },
        {
          "key": "medscribe:Grok 4.3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.3\"].accuracy; Overall leaderboard rank 77",
          "scoreDisplay": "74.40%"
        },
        {
          "key": "medscribe:Claude Opus 4.1 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-1-20250805-thinking\"].accuracy; Overall leaderboard rank 78",
          "scoreDisplay": "73.90%"
        },
        {
          "key": "medscribe:Gemini 2.5 Pro",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-pro\"].accuracy; Overall leaderboard rank 79",
          "scoreDisplay": "73.55%"
        },
        {
          "key": "medscribe:GPT 5 Nano",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/gpt-5-nano-2025-08-07\"].accuracy; Overall leaderboard rank 80",
          "scoreDisplay": "72.86%"
        },
        {
          "key": "medscribe:Gemini 2.5 Flash Lite (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite\"].accuracy; Overall leaderboard rank 81",
          "scoreDisplay": "72.83%"
        },
        {
          "key": "medscribe:Qwen 3 Max Thinking",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3-max-2026-01-23\"].accuracy; Overall leaderboard rank 82",
          "scoreDisplay": "72.71%"
        },
        {
          "key": "medscribe:Claude Sonnet 4 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-20250514\"].accuracy; Overall leaderboard rank 83",
          "scoreDisplay": "72.41%"
        },
        {
          "key": "medscribe:GLM 5.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-5.1\"].accuracy; Overall leaderboard rank 84",
          "scoreDisplay": "72.27%"
        },
        {
          "key": "medscribe:MiMo V2.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"xiaomi/mimo-v2.5\"].accuracy; Overall leaderboard rank 85",
          "scoreDisplay": "72.15%"
        },
        {
          "key": "medscribe:Gemini 3 Pro (11/25)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3-pro-preview\"].accuracy; Overall leaderboard rank 86",
          "scoreDisplay": "72.04%"
        },
        {
          "key": "medscribe:Claude Opus 4.1 (Nonthinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-opus-4-1-20250805\"].accuracy; Overall leaderboard rank 87",
          "scoreDisplay": "71.75%"
        },
        {
          "key": "medscribe:Gemini 3.5 Flash Lite",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.5-flash-lite\"].accuracy; Overall leaderboard rank 88",
          "scoreDisplay": "70.89%"
        },
        {
          "key": "medscribe:Qwen 3.5 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"alibaba/qwen3.5-flash\"].accuracy; Overall leaderboard rank 89",
          "scoreDisplay": "70.62%"
        },
        {
          "key": "medscribe:Gemini 3 Flash (12/25)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3-flash-preview\"].accuracy; Overall leaderboard rank 90",
          "scoreDisplay": "69.92%"
        },
        {
          "key": "medscribe:Claude Sonnet 4 (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"anthropic/claude-sonnet-4-20250514-thinking\"].accuracy; Overall leaderboard rank 91",
          "scoreDisplay": "69.35%"
        },
        {
          "key": "medscribe:o4 Mini",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"openai/o4-mini-2025-04-16\"].accuracy; Overall leaderboard rank 92",
          "scoreDisplay": "69.14%"
        },
        {
          "key": "medscribe:GLM 4.7",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"zai/glm-4.7\"].accuracy; Overall leaderboard rank 93",
          "scoreDisplay": "68.63%"
        },
        {
          "key": "medscribe:Mistral Medium 3.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"mistralai/mistral-medium-3.5\"].accuracy; Overall leaderboard rank 94",
          "scoreDisplay": "67.73%"
        },
        {
          "key": "medscribe:Gemini 2.5 Flash Lite (9/25) (Thinking)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-2.5-flash-lite-preview-09-2025-thinking\"].accuracy; Overall leaderboard rank 95",
          "scoreDisplay": "66.88%"
        },
        {
          "key": "medscribe:Laguna M.1",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"poolside/laguna-m.1\"].accuracy; Overall leaderboard rank 96",
          "scoreDisplay": "65.91%"
        },
        {
          "key": "medscribe:Gemini 3.1 Flash Lite Preview",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"google/gemini-3.1-flash-lite-preview\"].accuracy; Overall leaderboard rank 97",
          "scoreDisplay": "63.90%"
        },
        {
          "key": "medscribe:Grok 4.20 (Reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"grok/grok-4.20-0309-reasoning\"].accuracy; Overall leaderboard rank 98",
          "scoreDisplay": "63.41%"
        },
        {
          "key": "medscribe:Laguna XS.2",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"poolside/laguna-xs.2\"].accuracy; Overall leaderboard rank 99",
          "scoreDisplay": "61.43%"
        },
        {
          "key": "medscribe:Command A+",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"cohere/command-a-plus-05-2026\"].accuracy; Overall leaderboard rank 100",
          "scoreDisplay": "55.68%"
        },
        {
          "key": "medscribe:Mercury 2.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"inception/mercury-2.5\"].accuracy; Overall leaderboard rank 101",
          "scoreDisplay": "55.09%"
        },
        {
          "key": "medscribe:Llama 4 Maverick",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"fireworks/llama4-maverick-instruct-basic\"].accuracy; Overall leaderboard rank 102",
          "scoreDisplay": "54.22%"
        },
        {
          "key": "medscribe:Llama 4 Scout",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"together/meta-llama/Llama-4-Scout-17B-16E-Instruct\"].accuracy; Overall leaderboard rank 103",
          "scoreDisplay": "50.59%"
        },
        {
          "key": "medscribe:Nemotron 3.5 Lightning",
          "role": "primary",
          "quote": null,
          "locator": "Embedded BenchmarkView data, tasks.overall[\"fireworks/nemotron-lightning-3p5-30b-a3b\"].accuracy; Overall leaderboard rank 104",
          "scoreDisplay": "4.27%"
        }
      ]
    },
    {
      "id": 24,
      "title": "MedXpertQA (MM) Leaderboard & Scores - September 2026 | BenchLM.ai",
      "url": "https://benchlm.ai/benchmarks/medxpertqamm",
      "kind": "mirror",
      "publisher": "benchlm.ai",
      "firstParty": false,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "medxpertqa-mm:Gemini 3.1 Pro",
          "role": "mirror",
          "quote": null,
          "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 1; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology",
          "scoreDisplay": "81.3%"
        },
        {
          "key": "medxpertqa-mm:Qwen3.8 Max",
          "role": "mirror",
          "quote": null,
          "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 2; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology",
          "scoreDisplay": "80.4%"
        },
        {
          "key": "medxpertqa-mm:Muse Spark",
          "role": "mirror",
          "quote": null,
          "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 3; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology",
          "scoreDisplay": "78.4%"
        },
        {
          "key": "medxpertqa-mm:GPT-5.4",
          "role": "mirror",
          "quote": null,
          "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 4; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology",
          "scoreDisplay": "77.1%"
        },
        {
          "key": "medxpertqa-mm:Qwen3.7 Plus",
          "role": "mirror",
          "quote": null,
          "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 5; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology",
          "scoreDisplay": "71.0%"
        },
        {
          "key": "medxpertqa-mm:Grok 4.20",
          "role": "mirror",
          "quote": null,
          "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 6; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology",
          "scoreDisplay": "65.8%"
        },
        {
          "key": "medxpertqa-mm:Claude Opus 4.6",
          "role": "mirror",
          "quote": null,
          "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 7; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology",
          "scoreDisplay": "64.8%"
        },
        {
          "key": "medxpertqa-mm:Gemma 4 12B",
          "role": "mirror",
          "quote": null,
          "locator": "BENCHMARK SCORE TABLE (8 MODELS), rank 8; the page says 'We mirror the published score view for MedXpertQA (MM)' and 'Updated September 4, 2026', citing Meta's Muse Spark Eval Methodology",
          "scoreDisplay": "48.7%"
        }
      ]
    },
    {
      "id": 26,
      "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240)",
      "url": "https://arxiv.org/abs/2605.02240",
      "kind": "paper",
      "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "physicianbench:GPT-5.5",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "46.3 ± 1.2"
        },
        {
          "key": "physicianbench:Claude Opus 4.6",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "31.7 ± 2.3"
        },
        {
          "key": "physicianbench:Claude Opus 4.7",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "29.3 ± 2.5"
        },
        {
          "key": "physicianbench:GPT-5.4",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "27.7 ± 1.5"
        },
        {
          "key": "physicianbench:Claude Sonnet 4.6",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "23.0 ± 2.6"
        },
        {
          "key": "physicianbench:Kimi-K2.6",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "17.0 ± 2.6"
        },
        {
          "key": "physicianbench:Qwen3.6-Plus",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "13.7 ± 4.0"
        },
        {
          "key": "physicianbench:Gemini Pro 3.1",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "6.0 ± 1.0"
        }
      ]
    },
    {
      "id": 27,
      "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301)",
      "url": "https://arxiv.org/abs/2606.23301",
      "kind": "paper",
      "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "ehr-complex:GPT-5.4 (high reasoning)",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "0.65"
        },
        {
          "key": "ehr-complex:Gemini 3.1 Pro",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "0.63"
        },
        {
          "key": "ehr-complex:Kimi-K2.5",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "0.62"
        },
        {
          "key": "ehr-complex:Qwen3.5-397B",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "0.62"
        },
        {
          "key": "ehr-complex:GPT-5.4 (low reasoning)",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "0.58"
        },
        {
          "key": "ehr-complex:Claude Sonnet 4.6",
          "role": "corroborating",
          "quote": null,
          "locator": "arXiv abs page (landing page for the PDF)",
          "scoreDisplay": "0.36"
        }
      ]
    },
    {
      "id": 28,
      "title": "WHBench (arXiv 2604.00024 abstract page)",
      "url": "https://arxiv.org/abs/2604.00024",
      "kind": "paper",
      "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "whbench:Claude Opus 4.6",
          "role": "corroborating",
          "quote": null,
          "locator": "same paper, abstract page",
          "scoreDisplay": "72.1%"
        },
        {
          "key": "whbench:Claude Sonnet 4.6",
          "role": "corroborating",
          "quote": null,
          "locator": "same paper, abstract page",
          "scoreDisplay": "67.1%"
        },
        {
          "key": "whbench:GPT-5.4",
          "role": "corroborating",
          "quote": null,
          "locator": "same paper, abstract page",
          "scoreDisplay": "66.8%"
        },
        {
          "key": "whbench:Gemini 3 Flash Preview",
          "role": "corroborating",
          "quote": null,
          "locator": "same paper, abstract page",
          "scoreDisplay": "64.7%"
        },
        {
          "key": "whbench:GPT-4.1",
          "role": "corroborating",
          "quote": null,
          "locator": "same paper, abstract page",
          "scoreDisplay": "51.8%"
        },
        {
          "key": "whbench:GPT-4o",
          "role": "corroborating",
          "quote": null,
          "locator": "same paper, abstract page",
          "scoreDisplay": "44.6%"
        }
      ]
    },
    {
      "id": 29,
      "title": "Introducing HealthAdminBench: AI Agents Can Diagnose Rare Diseases, But Can They Handle Your Insurance? (Kinetic Systems blog)",
      "url": "https://kineticsystems.ai/blog/healthadminbench-automating-healthcare-administration-with-computer-use-agents",
      "kind": "blog",
      "publisher": "Kinetic Systems",
      "firstParty": false,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "healthadminbench:Claude Opus 4.6 (computer-use agent)",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 'LLMs struggle with long-horizon tasks'",
          "scoreDisplay": "36.3%"
        },
        {
          "key": "healthadminbench:GPT-5.4 (computer-use agent)",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 'LLMs struggle with long-horizon tasks'",
          "scoreDisplay": "26.7%"
        },
        {
          "key": "healthadminbench:Kimi K2.5",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 'LLMs struggle with long-horizon tasks'",
          "scoreDisplay": "15.6%"
        },
        {
          "key": "healthadminbench:Qwen 3.5",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 'LLMs struggle with long-horizon tasks'",
          "scoreDisplay": "13.3%"
        },
        {
          "key": "healthadminbench:Gemini 3.1 Pro",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 'LLMs struggle with long-horizon tasks'",
          "scoreDisplay": "11.9%"
        }
      ]
    },
    {
      "id": 30,
      "title": "GPT-5.6 - August Updates (system card addendum)",
      "url": "https://cdn.openai.com/pdf/GPT_5_6_August_Updates.pdf",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-08-06",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:GPT-5.6 Sol (August)",
          "role": "primary",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Sol (August)",
          "scoreDisplay": "0.540"
        },
        {
          "key": "healthbench-professional:GPT-5.6 Luna (August)",
          "role": "primary",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Luna (August)",
          "scoreDisplay": "0.441"
        },
        {
          "key": "healthbench-professional:GPT-5.5 Instant",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.5 Instant",
          "scoreDisplay": "38.4"
        },
        {
          "key": "healthbench-hard:GPT-5.6 Sol (August)",
          "role": "primary",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Sol (August)",
          "scoreDisplay": "0.314"
        },
        {
          "key": "healthbench-hard:GPT-5.6 Luna (August)",
          "role": "primary",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Luna (August)",
          "scoreDisplay": "0.287"
        },
        {
          "key": "healthbench-hard:GPT-5.3 Chat",
          "role": "conflicting",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.3 Instant",
          "scoreDisplay": "20.2"
        },
        {
          "key": "healthbench-hard:GPT-5.5 Instant",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.5 Instant",
          "scoreDisplay": "22.9"
        },
        {
          "key": "healthbench:GPT-5.6 Sol (August)",
          "role": "primary",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Sol (August)",
          "scoreDisplay": "55.0"
        },
        {
          "key": "healthbench:GPT-5.6 Luna (August)",
          "role": "primary",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.6 Luna (August)",
          "scoreDisplay": "53.3"
        },
        {
          "key": "healthbench:GPT-5.5 Instant",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, section 5.1 HealthBench, table \"Reported as length-adjusted score (unadjusted, mean response length in characters)\", column GPT-5.5 Instant",
          "scoreDisplay": "51.4"
        }
      ]
    },
    {
      "id": 45,
      "title": "WHBench: A Women's Health Benchmark for Evaluating Frontier LLMs with Expert-in-the-Loop Validation (arXiv 2604.00024v2)",
      "url": "https://arxiv.org/pdf/2604.00024v2",
      "kind": "paper",
      "publisher": "Maurya, Govindgari, Kumar (Columbia University / Akhara AI)",
      "firstParty": true,
      "publishedAt": "2026-07-23",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "whbench:Claude Opus 4.6",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "72.1%"
        },
        {
          "key": "whbench:Claude Sonnet 4.6",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "67.1%"
        },
        {
          "key": "whbench:GPT-5.4",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "66.8%"
        },
        {
          "key": "whbench:Gemini 3 Flash Preview",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "64.7%"
        },
        {
          "key": "whbench:OpenAI o3",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "63.6%"
        },
        {
          "key": "whbench:DeepSeek V3.2",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "61.3%"
        },
        {
          "key": "whbench:Grok 3",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "60.7%"
        },
        {
          "key": "whbench:Mistral Large",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "60.2%"
        },
        {
          "key": "whbench:Grok 4",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "57.9%"
        },
        {
          "key": "whbench:DeepSeek-R1",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "52.9%"
        },
        {
          "key": "whbench:GPT-4.1",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "51.8%"
        },
        {
          "key": "whbench:Grok 3 Mini",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "50.0%"
        },
        {
          "key": "whbench:Gemini 2.5 Flash",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "49.5%"
        },
        {
          "key": "whbench:Claude Opus 4",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "49.1%"
        },
        {
          "key": "whbench:Claude Sonnet 4",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "48.1%"
        },
        {
          "key": "whbench:GPT-4o",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "44.6%"
        },
        {
          "key": "whbench:Llama 4 Maverick",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "42.1%"
        },
        {
          "key": "whbench:Nemotron 70B",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "39.3%"
        },
        {
          "key": "whbench:Llama 3.3 70B",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "37.8%"
        },
        {
          "key": "whbench:Llama 3.1 405B",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "36.1%"
        },
        {
          "key": "whbench:Gemini 2.5 Pro",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "35.3%"
        },
        {
          "key": "whbench:Llama 4 Scout",
          "role": "primary",
          "quote": null,
          "locator": "p. 5, Table 3 (WHBench v3.0 leaderboard)",
          "scoreDisplay": "35.2%"
        }
      ]
    },
    {
      "id": 46,
      "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents (arXiv 2606.31179v1)",
      "url": "https://arxiv.org/pdf/2606.31179",
      "kind": "paper",
      "publisher": "Microsoft Research (Liu, Zhang, Qin et al.)",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "healthagentbench:Codex (GPT 5.5)",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)",
          "scoreDisplay": "42%"
        },
        {
          "key": "healthagentbench:Copilot (Opus 4.8)",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)",
          "scoreDisplay": "36%"
        },
        {
          "key": "healthagentbench:Copilot (GPT 5.5)",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)",
          "scoreDisplay": "35%"
        },
        {
          "key": "healthagentbench:Claude Code (Opus 4.8)",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)",
          "scoreDisplay": "32%"
        },
        {
          "key": "healthagentbench:Codex (GPT 5.4)",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)",
          "scoreDisplay": "28%"
        },
        {
          "key": "healthagentbench:Claude Code (Opus 4.7)",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)",
          "scoreDisplay": "27%"
        },
        {
          "key": "healthagentbench:Claude Code (Opus 4.6)",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)",
          "scoreDisplay": "19%"
        },
        {
          "key": "healthagentbench:Claude Code (Sonnet 4.6)",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)",
          "scoreDisplay": "17%"
        },
        {
          "key": "healthagentbench:Codex (GPT 5.4 Mini)",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 11, Figure 4 (pooled task success rate, ten agents)",
          "scoreDisplay": "16%"
        }
      ]
    },
    {
      "id": 50,
      "title": "GPT-5.6 System Card",
      "url": "https://deploymentsafety.openai.com/gpt-5-6/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-07-09",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:GPT-5.6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-SOL",
          "scoreDisplay": "0.605"
        },
        {
          "key": "healthbench-professional:GPT-5.6 Terra",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-TERRA",
          "scoreDisplay": "0.577"
        },
        {
          "key": "healthbench-professional:GPT-5.6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-LUNA",
          "scoreDisplay": "0.557"
        },
        {
          "key": "healthbench-professional:GPT-5.5",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5",
          "scoreDisplay": "0.518"
        },
        {
          "key": "healthbench-professional:GPT-5.4",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.4",
          "scoreDisplay": "0.481"
        },
        {
          "key": "healthbench-professional:GPT-5",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5",
          "scoreDisplay": "0.462"
        },
        {
          "key": "healthbench-professional:GPT-5.2",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.2",
          "scoreDisplay": "0.459"
        },
        {
          "key": "healthbench-professional:GPT-5.1",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.1",
          "scoreDisplay": "0.396"
        },
        {
          "key": "healthbench-hard:GPT-5",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5",
          "scoreDisplay": "0.347"
        },
        {
          "key": "healthbench-hard:GPT-5.2",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.2",
          "scoreDisplay": "0.343"
        },
        {
          "key": "healthbench-hard:GPT-5.6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-SOL",
          "scoreDisplay": "0.331"
        },
        {
          "key": "healthbench-hard:GPT-5.6 Terra",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-TERRA",
          "scoreDisplay": "0.327"
        },
        {
          "key": "healthbench-hard:GPT-5.6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-LUNA",
          "scoreDisplay": "0.320"
        },
        {
          "key": "healthbench-hard:GPT-5.5",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5",
          "scoreDisplay": "0.315"
        },
        {
          "key": "healthbench-hard:GPT-5.4",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.4",
          "scoreDisplay": "0.291"
        },
        {
          "key": "healthbench-hard:GPT-5.1",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.1",
          "scoreDisplay": "0.254"
        },
        {
          "key": "healthbench:GPT-5.2-High",
          "role": "conflicting",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.2",
          "scoreDisplay": "56.8"
        },
        {
          "key": "healthbench:GPT-5",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1, Table 6, HealthBench length-adjusted row, GPT-5 column",
          "scoreDisplay": "57.7"
        },
        {
          "key": "healthbench:GPT-5.6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-SOL",
          "scoreDisplay": "57.0"
        },
        {
          "key": "healthbench:GPT-5.6 Terra",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-TERRA",
          "scoreDisplay": "57.0"
        },
        {
          "key": "healthbench:GPT-5.2",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1, Table 6, HealthBench length-adjusted row, GPT-5.2 column",
          "scoreDisplay": "56.8"
        },
        {
          "key": "healthbench:GPT-5.5",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5",
          "scoreDisplay": "56.5"
        },
        {
          "key": "healthbench:GPT-5.6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1 HealthBench, Table 6 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.6-LUNA",
          "scoreDisplay": "55.8"
        },
        {
          "key": "healthbench:GPT-5.4",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1, Table 6, HealthBench length-adjusted row, GPT-5.4 column",
          "scoreDisplay": "54.0"
        },
        {
          "key": "healthbench:GPT-5.1",
          "role": "primary",
          "quote": null,
          "locator": "Section 5.1, Table 6, HealthBench length-adjusted row, GPT-5.1 column",
          "scoreDisplay": "50.9"
        }
      ]
    },
    {
      "id": 51,
      "title": "GPT-5.5 Instant System Card",
      "url": "https://deploymentsafety.openai.com/gpt-5-5-instant/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-05-05",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:GPT-5.5 Instant",
          "role": "primary",
          "quote": null,
          "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5 INSTANT",
          "scoreDisplay": "0.384"
        },
        {
          "key": "healthbench-hard:GPT-5.3 Chat",
          "role": "conflicting",
          "quote": null,
          "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.3 INSTANT",
          "scoreDisplay": "20.2"
        },
        {
          "key": "healthbench-hard:GPT-5.5 Instant",
          "role": "primary",
          "quote": null,
          "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5 INSTANT",
          "scoreDisplay": "0.229"
        },
        {
          "key": "healthbench:GPT-5.3 Chat",
          "role": "conflicting",
          "quote": null,
          "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.3 INSTANT",
          "scoreDisplay": "49.6"
        },
        {
          "key": "healthbench:GPT-5.5 Instant",
          "role": "primary",
          "quote": null,
          "locator": "Section 4.1 HealthBench, Table 5 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5 INSTANT",
          "scoreDisplay": "51.4"
        }
      ]
    },
    {
      "id": 54,
      "title": "GPT-5.3 Instant System Card",
      "url": "https://deploymentsafety.openai.com/gpt-5-3-instant/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-03-02",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-hard:GPT-5.3 Chat",
          "role": "primary",
          "quote": null,
          "locator": "Section 4.1 HealthBench, Table 3: HealthBench, row Hard, column GPT-5.3-INSTANT",
          "scoreDisplay": "0.259"
        },
        {
          "key": "healthbench:GPT-5.3 Chat",
          "role": "primary",
          "quote": null,
          "locator": "Section 4.1 HealthBench, Table 3: HealthBench, row HealthBench, column GPT-5.3-INSTANT",
          "scoreDisplay": "54.1%"
        }
      ]
    },
    {
      "id": 55,
      "title": "GPT-5.5 System Card",
      "url": "https://deploymentsafety.openai.com/gpt-5-5/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-hard:GPT-5",
          "role": "corroborating",
          "quote": null,
          "locator": "Section 5 Health, Table 7, column GPT-5",
          "scoreDisplay": "34.7"
        },
        {
          "key": "healthbench:GPT-5.5",
          "role": "primary",
          "quote": null,
          "locator": "Section 5 Health, Table 7 (reported as length-adjusted score (unadjusted, mean response length in characters)), column GPT-5.5",
          "scoreDisplay": "56.5"
        }
      ]
    },
    {
      "id": 61,
      "title": "Claude Fable 5 and Claude Mythos 5 System Card",
      "url": "https://anthropic.com/claude-fable-5-mythos-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-06-09",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench:Claude Opus 4.8",
          "role": "primary",
          "quote": null,
          "locator": "p. 252, Table 8.1.A, row HealthBench, column Opus 4.8",
          "scoreDisplay": "59.3"
        }
      ]
    },
    {
      "id": 63,
      "title": "gpt-oss-120b & gpt-oss-20b Model Card",
      "url": "https://deploymentsafety.openai.com/gpt-oss/healthbench",
      "kind": "model_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2025-08-05",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-hard:GPT OSS 120B",
          "role": "primary",
          "quote": null,
          "locator": "Section 2, Table 3: Evaluations across multiple benchmarks and reasoning levels, row HealthBench Hard, column gpt-oss-120b high",
          "scoreDisplay": "0.300"
        },
        {
          "key": "healthbench-hard:GPT OSS 20B",
          "role": "primary",
          "quote": null,
          "locator": "Section 2, Table 3: Evaluations across multiple benchmarks and reasoning levels, row HealthBench Hard, column gpt-oss-20b high",
          "scoreDisplay": "0.108"
        },
        {
          "key": "healthbench:GPT OSS 120B",
          "role": "primary",
          "quote": null,
          "locator": "Section 2, Table 3: Evaluations across multiple benchmarks and reasoning levels, row HealthBench, column gpt-oss-120b high",
          "scoreDisplay": "57.6"
        },
        {
          "key": "healthbench:GPT OSS 20B",
          "role": "primary",
          "quote": null,
          "locator": "Section 2, Table 3: Evaluations across multiple benchmarks and reasoning levels, row HealthBench, column gpt-oss-20b high",
          "scoreDisplay": "42.5"
        }
      ]
    },
    {
      "id": 64,
      "title": "HealthAgentBench detailed results",
      "url": "https://microsoft.github.io/HealthAgentBench/results",
      "kind": "official_leaderboard",
      "publisher": "Microsoft Research (HealthAgentBench)",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "healthagentbench:Claude Code (Opus 5)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 1",
          "scoreDisplay": "55%"
        },
        {
          "key": "healthagentbench:Codex (GPT-5.6-sol)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 2",
          "scoreDisplay": "45%"
        },
        {
          "key": "healthagentbench:Codex (GPT 5.5)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 3",
          "scoreDisplay": "42%"
        },
        {
          "key": "healthagentbench:Copilot (Opus 4.8)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 4",
          "scoreDisplay": "36%"
        },
        {
          "key": "healthagentbench:Copilot (GPT 5.5)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 5",
          "scoreDisplay": "35%"
        },
        {
          "key": "healthagentbench:Claude Code (Opus 4.8)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 6",
          "scoreDisplay": "32%"
        },
        {
          "key": "healthagentbench:Codex (GPT 5.4)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 7",
          "scoreDisplay": "28%"
        },
        {
          "key": "healthagentbench:Claude Code (Opus 4.7)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 8",
          "scoreDisplay": "27%"
        },
        {
          "key": "healthagentbench:Claude Code (Opus 4.6)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 10",
          "scoreDisplay": "19%"
        },
        {
          "key": "healthagentbench:Claude Code (Sonnet 4.6)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 11",
          "scoreDisplay": "17%"
        },
        {
          "key": "healthagentbench:Codex (GPT 5.4 Mini)",
          "role": "corroborating",
          "quote": null,
          "locator": "Detailed results table, rank 12",
          "scoreDisplay": "16%"
        }
      ]
    },
    {
      "id": 71,
      "title": "Introducing Muse Spark: Scaling Towards Personal Superintelligence",
      "url": "https://ai.meta.com/blog/introducing-muse-spark-msl/",
      "kind": "launch_post",
      "publisher": "Meta",
      "firstParty": true,
      "publishedAt": "2026-04-08",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-hard:Muse Spark",
          "role": "primary",
          "quote": null,
          "locator": "Launch-post benchmark table image, HEALTH section, row HealthBench Hard, column Muse Spark Thinking; identical table in the Eval Methodology PDF p. 5; protocol p. 2: HealthBench Hard: This is a subset of OpenAI's HealthBench benchmark, containing 1000 prompts. We used the same implementation as in the OpenAI’s official simple-evals repo, with GPT-4.1-genai as the LLM-as-judge model.",
          "scoreDisplay": "0.428"
        },
        {
          "key": "medxpertqa-mm:Gemini 3.1 Pro",
          "role": "primary",
          "quote": null,
          "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column Gemini 3.1 Pro High; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'",
          "scoreDisplay": "81.3%"
        },
        {
          "key": "medxpertqa-mm:Muse Spark",
          "role": "primary",
          "quote": null,
          "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column Muse Spark Thinking; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'",
          "scoreDisplay": "78.4%"
        },
        {
          "key": "medxpertqa-mm:GPT-5.4",
          "role": "primary",
          "quote": null,
          "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column GPT 5.4 Xhigh; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'",
          "scoreDisplay": "77.1%"
        },
        {
          "key": "medxpertqa-mm:Grok 4.20",
          "role": "primary",
          "quote": null,
          "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column Grok 4.2 Reasoning; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'",
          "scoreDisplay": "65.8%"
        },
        {
          "key": "medxpertqa-mm:Claude Opus 4.6",
          "role": "primary",
          "quote": null,
          "locator": "Launch-post benchmark table image, HEALTH section, row MedXpertQA (MM), column Opus 4.6 Max; the same table is page 5 of the Eval Methodology PDF; protocol p. 2: 'MedXpertQA Text/Multimodal: ... The multimodal variant contains 2,000 multimodal medical questions with clinical images (X-rays, histology, dermatology, etc.) and 5 answer choices (A-E). For grading, we use gpt-oss-120b to parse the predicted answer letter from free-form text.'",
          "scoreDisplay": "64.8%"
        }
      ]
    },
    {
      "id": 74,
      "title": "Muse Spark Eval Methodology",
      "url": "https://ai.meta.com/static-resource/muse-spark-eval-methodology",
      "kind": "model_card",
      "publisher": "Meta",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "healthbench-hard:Muse Spark",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 5 results table image, row HealthBench Hard; protocol p. 2",
          "scoreDisplay": "42.8"
        },
        {
          "key": "medxpertqa-mm:Gemini 3.1 Pro",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28",
          "scoreDisplay": "81.3"
        },
        {
          "key": "medxpertqa-mm:Muse Spark",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28",
          "scoreDisplay": "78.4"
        },
        {
          "key": "medxpertqa-mm:GPT-5.4",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28",
          "scoreDisplay": "77.1"
        },
        {
          "key": "medxpertqa-mm:Grok 4.20",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28",
          "scoreDisplay": "65.8"
        },
        {
          "key": "medxpertqa-mm:Claude Opus 4.6",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 5 benchmark image, Health section, MedXpertQA (MM); visually checked 2026-09-28",
          "scoreDisplay": "64.8"
        }
      ]
    },
    {
      "id": 75,
      "title": "MAI-Thinking-1: Building a Hill-Climbing Machine",
      "url": "https://microsoft.ai/pdf/mai-thinking-1.pdf",
      "kind": "model_card",
      "publisher": "Microsoft AI",
      "firstParty": true,
      "publishedAt": "2026-08-12",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:MAI-Thinking-1",
          "role": "primary",
          "quote": null,
          "locator": "p. 54, Table 12 'Post-trained model evaluation results on various public benchmarks', Health group, column HealthBench Prof.; protocol Appendix K.6 p. 106: HealthBench Professional introduces a length penalty for the primary metric, to correct for a well-observed correlation between lengthy responses and artificially increased LLM-grader scores. For all reported scores, we use the standard GPT-5.4 grader and rubrics provided by OpenAI.",
          "scoreDisplay": "0.350"
        }
      ]
    },
    {
      "id": 79,
      "title": "GPT-5 System Card",
      "url": "https://cdn.openai.com/gpt-5-system-card.pdf",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "healthbench-hard:GPT-5",
          "role": "conflicting",
          "quote": null,
          "locator": "p. 18, section 3.10 Health, Figure 6 (HealthBench Hard, raw score %)",
          "scoreDisplay": "46.2"
        }
      ]
    },
    {
      "id": 83,
      "title": "MedHELM v5.0.0 release data: medhelm_scenarios group table (mean win rate)",
      "url": "https://leaderboard.medhelm.org/benchmark_output/releases/v5.0.0/groups/medhelm_scenarios.json",
      "kind": "official_leaderboard",
      "publisher": "Stanford CRFM (MedHELM)",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "medhelm:Gemini 3.1 Pro (Preview)",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.6520833333333333"
        },
        {
          "key": "medhelm:Gemini 3.5 Flash",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.6416666666666667"
        },
        {
          "key": "medhelm:Muse Spark (2026-04-08)",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.6208333333333333"
        },
        {
          "key": "medhelm:GPT-5.4 mini",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.5520833333333334"
        },
        {
          "key": "medhelm:GPT-5.4 (2026-03-05)",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.5375"
        },
        {
          "key": "medhelm:Gemini 2.5 Pro",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.5291666666666667"
        },
        {
          "key": "medhelm:DeepSeek R1",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.48541666666666666"
        },
        {
          "key": "medhelm:Claude 4.6 Opus",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.45625"
        },
        {
          "key": "medhelm:Claude 3.7 Sonnet",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.45"
        },
        {
          "key": "medhelm:Gemini 2.0 Flash",
          "role": "corroborating",
          "quote": null,
          "locator": "releases/v5.0.0/groups/medhelm_scenarios.json, table 'Accuracy', column 'Mean win rate'",
          "scoreDisplay": "0.3416666666666667"
        }
      ]
    },
    {
      "id": 94,
      "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments (arXiv 2605.02240v1 PDF)",
      "url": "https://arxiv.org/pdf/2605.02240",
      "kind": "paper",
      "publisher": "Stanford University (HealthRex; Liu, Chen et al.)",
      "firstParty": true,
      "publishedAt": "2026-05-04",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "physicianbench:GPT-5.5",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "46.3 ± 1.2"
        },
        {
          "key": "physicianbench:Claude Opus 4.6",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "31.7 ± 2.3"
        },
        {
          "key": "physicianbench:Claude Opus 4.7",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "29.3 ± 2.5"
        },
        {
          "key": "physicianbench:GPT-5.4",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "27.7 ± 1.5"
        },
        {
          "key": "physicianbench:Claude Sonnet 4.6",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "23.0 ± 2.6"
        },
        {
          "key": "physicianbench:DeepSeek V4-Pro",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "18.7 ± 2.9"
        },
        {
          "key": "physicianbench:Kimi-K2.6",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "17.0 ± 2.6"
        },
        {
          "key": "physicianbench:MiMo-v2.5-Pro",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "16.7 ± 4.0"
        },
        {
          "key": "physicianbench:Qwen3.6-Plus",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "13.7 ± 4.0"
        },
        {
          "key": "physicianbench:MiniMax M2.7",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "8.7 ± 1.2"
        },
        {
          "key": "physicianbench:Gemini Pro 3.1",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (columns Pass@1, Pass@3, Pass^3, #Turns)",
          "scoreDisplay": "6.0 ± 1.0"
        },
        {
          "key": "physicianbench:Grok-4.20",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Table 2 (Proprietary Models block)",
          "scoreDisplay": "5.3 ± 3.2"
        }
      ]
    },
    {
      "id": 98,
      "title": "System Card: Claude Opus 5",
      "url": "https://www.anthropic.com/claude-opus-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-07-24",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:Claude Opus 5",
          "role": "primary",
          "quote": null,
          "locator": "p. 189, section 8.15.2 HealthBench Professional results; also Table 8.1.A p. 152 ('HealthBench Professional 59.8 ...')",
          "scoreDisplay": "0.598"
        },
        {
          "key": "healthbench-professional:Claude Sonnet 5",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 189, section 8.15.2, Figure 8.15.2.A",
          "scoreDisplay": "57.8%"
        },
        {
          "key": "healthbench:Claude Opus 4.8",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 188, section 8.15.1, Figure 8.15.1.A",
          "scoreDisplay": "59.3%"
        },
        {
          "key": "healthbench:Claude Sonnet 5",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 188, section 8.15.1, Figure 8.15.1.A",
          "scoreDisplay": "58.7%"
        },
        {
          "key": "healthbench:Claude Opus 5",
          "role": "primary",
          "quote": null,
          "locator": "p. 188, section 8.15.1 and Figure 8.15.1.A; adjusted 57.8%, raw 67.1%",
          "scoreDisplay": "57.8"
        }
      ]
    },
    {
      "id": 99,
      "title": "System Card: Claude Sonnet 5",
      "url": "https://www.anthropic.com/claude-sonnet-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-06-30",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:Claude Sonnet 5",
          "role": "primary",
          "quote": null,
          "locator": "p. 115, Table 8.1.A, row HealthBench Professional, column Claude Sonnet 5; Figure 8.12.2.A p. 139",
          "scoreDisplay": "0.578"
        },
        {
          "key": "healthbench-professional:Claude Opus 4.8",
          "role": "primary",
          "quote": null,
          "locator": "p. 139, Figure 8.12.2.A; Opus 4.8 bar 57.4%",
          "scoreDisplay": "0.574"
        },
        {
          "key": "healthbench-professional:Claude Sonnet 4.6",
          "role": "primary",
          "quote": null,
          "locator": "p. 115, Table 8.1.A, row HealthBench Professional, column Claude Sonnet 4.6",
          "scoreDisplay": "0.442"
        },
        {
          "key": "healthbench:Claude Sonnet 5",
          "role": "primary",
          "quote": null,
          "locator": "p. 138, section 8.12.1 HealthBench results, Figure 8.12.1.A bar label (no prose or table number)",
          "scoreDisplay": "58.7%"
        }
      ]
    },
    {
      "id": 101,
      "title": "System Card: Claude Opus 4.8",
      "url": "https://www.anthropic.com/claude-opus-4-8-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-05-28",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:Claude Opus 4.7",
          "role": "primary",
          "quote": null,
          "locator": "p. 228, section 8.14.1 HealthBench Professional; Figure 8.14.A p. 229",
          "scoreDisplay": "0.519"
        },
        {
          "key": "healthbench-professional:Claude Sonnet 4.6",
          "role": "conflicting",
          "quote": null,
          "locator": "p. 228, section 8.14.1",
          "scoreDisplay": "41.7%"
        }
      ]
    },
    {
      "id": 105,
      "title": "Muse Spark 1.1 Evaluation Report",
      "url": "https://research.meta.ai/static/muse-spark-1-1-evaluation-report",
      "kind": "model_card",
      "publisher": "Meta",
      "firstParty": true,
      "publishedAt": "2026-07-09",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:Muse Spark 1.1",
          "role": "primary",
          "quote": null,
          "locator": "p. 101, Figure 44 'General capability benchmark results' (image), row HealthBench Professional, column Muse Spark 1.1; protocol p. 104 (printed 103): HealthBench Pro comprises 525 evaluation data points graded by rubrics. We use GPT-5.4 with low reasoning effort as the grader and report the length-normalized rubric score as done in their paper.",
          "scoreDisplay": "0.593"
        },
        {
          "key": "healthbench-professional:Muse Spark",
          "role": "primary",
          "quote": null,
          "locator": "p. 101, Figure 44 (image), row HealthBench Professional, column Muse Spark; protocol p. 104 (printed 103)",
          "scoreDisplay": "0.541"
        }
      ]
    },
    {
      "id": 109,
      "title": "EHR-Complex: Benchmarking Medical Agents for Complex Clinical Reasoning (arXiv 2606.23301v1 PDF)",
      "url": "https://arxiv.org/pdf/2606.23301",
      "kind": "paper",
      "publisher": "Zhejiang University / Ant Group (Qiao, Liu, Chu et al.)",
      "firstParty": true,
      "publishedAt": "2026-06-22",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "ehr-complex:GPT-5.4 (high reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "p. 15, Table 10 (Strong commercial model results), Avg. column",
          "scoreDisplay": "0.65"
        },
        {
          "key": "ehr-complex:Gemini 3.1 Pro",
          "role": "primary",
          "quote": null,
          "locator": "p. 15, Table 10 (Strong commercial model results), Avg. column",
          "scoreDisplay": "0.63"
        },
        {
          "key": "ehr-complex:Kimi-K2.5",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3 (Evaluation Results on the EHR-Complex Test Set), Avg. column",
          "scoreDisplay": "0.62"
        },
        {
          "key": "ehr-complex:Qwen3.5-397B",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3 (Evaluation Results on the EHR-Complex Test Set), Avg. column",
          "scoreDisplay": "0.62"
        },
        {
          "key": "ehr-complex:DeepSeek-V3.2-Exp",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.59"
        },
        {
          "key": "ehr-complex:GPT-5.4 (low reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "p. 15, Table 10 (Strong commercial model results), Avg. column",
          "scoreDisplay": "0.58"
        },
        {
          "key": "ehr-complex:DeepSeek-V3.1",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.56"
        },
        {
          "key": "ehr-complex:Qwen3-32B-SFT",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.55"
        },
        {
          "key": "ehr-complex:Qwen3-235B",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.53"
        },
        {
          "key": "ehr-complex:GPT-4.1 mini",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.49"
        },
        {
          "key": "ehr-complex:GPT-4.1",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.47"
        },
        {
          "key": "ehr-complex:Qwen3-14B-SFT",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.45"
        },
        {
          "key": "ehr-complex:Claude Sonnet 4.6",
          "role": "primary",
          "quote": null,
          "locator": "p. 15, Table 10 (Strong commercial model results), Avg. column",
          "scoreDisplay": "0.36"
        },
        {
          "key": "ehr-complex:Qwen3-32B",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.36"
        },
        {
          "key": "ehr-complex:GPT-4o",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.31"
        },
        {
          "key": "ehr-complex:Gemini 2.5 Pro",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.31"
        },
        {
          "key": "ehr-complex:Qwen3-14B",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.30"
        },
        {
          "key": "ehr-complex:Qwen3-4B",
          "role": "primary",
          "quote": null,
          "locator": "p. 6, Table 3, Avg. column",
          "scoreDisplay": "0.16"
        }
      ]
    },
    {
      "id": 138,
      "title": "HealthAdminBench: Evaluating Computer-Use Agents on Healthcare Administration Tasks (arXiv 2604.09937v1)",
      "url": "https://arxiv.org/pdf/2604.09937",
      "kind": "paper",
      "publisher": "Bedi, Welch, Steinberg et al. (Stanford University / Kinetic Systems / Stanford Health Care)",
      "firstParty": true,
      "publishedAt": "2026-04-10",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthadminbench:Claude Opus 4.6 (computer-use agent)",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)",
          "scoreDisplay": "36.3%"
        },
        {
          "key": "healthadminbench:GPT-5.4 (computer-use agent)",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)",
          "scoreDisplay": "26.7%"
        },
        {
          "key": "healthadminbench:Kimi K2.5",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)",
          "scoreDisplay": "15.6%"
        },
        {
          "key": "healthadminbench:Claude Opus 4.6 (standardized harness)",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)",
          "scoreDisplay": "14.8%"
        },
        {
          "key": "healthadminbench:Qwen 3.5",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)",
          "scoreDisplay": "13.3%"
        },
        {
          "key": "healthadminbench:Gemini 3.1 Pro",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)",
          "scoreDisplay": "11.9%"
        },
        {
          "key": "healthadminbench:GPT-5.4 (standardized harness)",
          "role": "primary",
          "quote": null,
          "locator": "p. 8, Figure 3(a) Task Success Rate (bar labels and values, extracted in order)",
          "scoreDisplay": "5.9%"
        }
      ]
    },
    {
      "id": 152,
      "title": "Qwen3.8-Max: A New Bar for Coding and Cowork",
      "url": "https://qwen.ai/blog?id=qwen3.8",
      "kind": "launch_post",
      "publisher": "Alibaba",
      "firstParty": true,
      "publishedAt": "2026-08-02",
      "retrieved": "2026-09-07",
      "results": [
        {
          "key": "medxpertqa-mm:GPT-5.6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Full Benchmark Table, second (multimodal) table, row MedXpertQA-MM; columns Opus4.8 / Fable5 / Gemini3.1-Pro / GPT5.6-Sol / Qwen3.7-Plus / Qwen3.8-Max",
          "scoreDisplay": "81.5"
        },
        {
          "key": "medxpertqa-mm:Qwen3.8 Max",
          "role": "primary",
          "quote": null,
          "locator": "Multimodal Benchmarks table, Multimodal Reasoning section, row MedXpertQA-MM, column Qwen3.8-Max; header row: |     | Opus4.8 | Fable5 | Gemini3.1-Pro | GPT5.6-Sol | Qwen3.7-Plus | Qwen3.8-Max |",
          "scoreDisplay": "80.4%"
        },
        {
          "key": "medxpertqa-mm:Claude Fable 5",
          "role": "primary",
          "quote": null,
          "locator": "Full Benchmark Table, second (multimodal) table, row MedXpertQA-MM; columns Opus4.8 / Fable5 / Gemini3.1-Pro / GPT5.6-Sol / Qwen3.7-Plus / Qwen3.8-Max",
          "scoreDisplay": "80.0"
        },
        {
          "key": "medxpertqa-mm:Claude Opus 4.8",
          "role": "primary",
          "quote": null,
          "locator": "Full Benchmark Table, second (multimodal) table, row MedXpertQA-MM; columns Opus4.8 / Fable5 / Gemini3.1-Pro / GPT5.6-Sol / Qwen3.7-Plus / Qwen3.8-Max",
          "scoreDisplay": "71.7"
        },
        {
          "key": "medxpertqa-mm:Qwen3.7 Plus",
          "role": "corroborating",
          "quote": null,
          "locator": "Qwen3.8 launch post, Multimodal Benchmarks table, row MedXpertQA-MM, column Qwen3.7-Plus (same value carried forward)",
          "scoreDisplay": "71.0"
        }
      ]
    },
    {
      "id": 202,
      "title": "Gemma 4 model card",
      "url": "https://ai.google.dev/gemma/docs/core/model_card_4",
      "kind": "model_card",
      "publisher": "Google",
      "firstParty": true,
      "publishedAt": "2026-04-02",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "medxpertqa-mm:Gemma 4 31B",
          "role": "primary",
          "quote": null,
          "locator": "Evaluation Results table, MedXPertQA MM row, Gemma 4 31B column",
          "scoreDisplay": "61.3%"
        },
        {
          "key": "medxpertqa-mm:Gemma 4 26B A4B",
          "role": "primary",
          "quote": null,
          "locator": "Evaluation Results table, MedXPertQA MM row, Gemma 4 26B A4B column",
          "scoreDisplay": "58.1%"
        },
        {
          "key": "medxpertqa-mm:Gemma 4 12B",
          "role": "primary",
          "quote": null,
          "locator": "Benchmark Results table, Vision section, row MedXPertQA MM, column Gemma 4 12B Unified; header row: |     | Gemma 4 31B | Gemma 4 26B A4B | Gemma 4 12B Unified | Gemma 4 E4B | Gemma 4 E2B | Gemma 3 27B (no think) |",
          "scoreDisplay": "48.7%"
        },
        {
          "key": "medxpertqa-mm:Gemma 4 E4B",
          "role": "primary",
          "quote": null,
          "locator": "Evaluation Results table, MedXPertQA MM row, Gemma 4 E4B column",
          "scoreDisplay": "28.7%"
        },
        {
          "key": "medxpertqa-mm:Gemma 4 E2B",
          "role": "primary",
          "quote": null,
          "locator": "Evaluation Results table, MedXPertQA MM row, Gemma 4 E2B column",
          "scoreDisplay": "23.5%"
        }
      ]
    },
    {
      "id": 205,
      "title": "Qwen3.7-Plus: Multimodal Agent Intelligence",
      "url": "https://qwen.ai/blog?id=qwen3.7-plus",
      "kind": "launch_post",
      "publisher": "Alibaba",
      "firstParty": true,
      "publishedAt": "2026-05-31",
      "retrieved": "2026-09-07",
      "results": [
        {
          "key": "medxpertqa-mm:Qwen3.7 Plus",
          "role": "primary",
          "quote": null,
          "locator": "Multimodal Benchmarks table, Multimodal Reasoning section, row MedXpertQA-MM, column Qwen3.7-Plus; header row: |     | GPT-5.4 (xhigh) | Opus-4.6 Max | Gemini-3.1 Pro | Qwen3.6-Plus | Qwen3.7-Plus |",
          "scoreDisplay": "71.0%"
        },
        {
          "key": "medxpertqa-mm:Qwen3.6 Plus",
          "role": "primary",
          "quote": null,
          "locator": "Multimodal Benchmarks table, row MedXpertQA-MM; columns GPT-5.4 (xhigh) / Opus-4.6 Max / Gemini-3.1 Pro / Qwen3.6-Plus / Qwen3.7-Plus",
          "scoreDisplay": "68.7"
        }
      ]
    },
    {
      "id": 212,
      "title": "Gemma 4 Technical Report",
      "url": "https://arxiv.org/pdf/2607.02770",
      "kind": "paper",
      "publisher": "Google DeepMind",
      "firstParty": false,
      "publishedAt": null,
      "retrieved": null,
      "results": [
        {
          "key": "medxpertqa-mm:Gemma 4 12B",
          "role": "corroborating",
          "quote": null,
          "locator": "Gemma 4 Technical Report p. 6, Table 6 (vision benchmarks, thinking), row MedXPertQA MM, column 12B",
          "scoreDisplay": "48.7"
        }
      ]
    },
    {
      "id": 318,
      "title": "Qwen/Qwen3.5-397B-A17B model card",
      "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B",
      "kind": "model_card",
      "publisher": "Alibaba / Qwen",
      "firstParty": true,
      "publishedAt": "2026-02-16",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "medxpertqa-mm:Gemini 3 Pro",
          "role": "primary",
          "quote": null,
          "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B",
          "scoreDisplay": "76.0%"
        },
        {
          "key": "medxpertqa-mm:GPT-5.2",
          "role": "primary",
          "quote": null,
          "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B",
          "scoreDisplay": "73.3"
        },
        {
          "key": "medxpertqa-mm:Qwen3.5 397B A17B",
          "role": "primary",
          "quote": null,
          "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B",
          "scoreDisplay": "70.0"
        },
        {
          "key": "medxpertqa-mm:Kimi K2.5",
          "role": "primary",
          "quote": null,
          "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B",
          "scoreDisplay": "65.3"
        },
        {
          "key": "medxpertqa-mm:Claude Opus 4.5",
          "role": "primary",
          "quote": null,
          "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B",
          "scoreDisplay": "63.6%"
        },
        {
          "key": "medxpertqa-mm:Qwen3-VL-235B-A22B",
          "role": "primary",
          "quote": null,
          "locator": "Benchmark Results > Vision Language > Medical VQA, row MedXpertQA-MM; columns GPT5.2 / Claude 4.5 Opus / Gemini-3 Pro / Qwen3-VL-235B-A22B / K2.5-1T-A32B / Qwen3.5-397B-A17B",
          "scoreDisplay": "47.6%"
        }
      ]
    },
    {
      "id": 10501,
      "title": "GPT-6 Astra System Card — September 22 revision",
      "url": "https://deploymentsafety.openai.com/gpt-6-astra/healthbench",
      "kind": "system_card",
      "publisher": "OpenAI",
      "firstParty": true,
      "publishedAt": "2026-09-03",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:GPT-6 Astra",
          "role": "primary",
          "quote": null,
          "locator": "Section 11.4.1, Table 29; HealthBench Professional length-adjusted row, GPT-6 Astra column; value 64.7 (68.2, 3185)",
          "scoreDisplay": "0.647"
        },
        {
          "key": "healthbench-professional:GPT-6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Section 11.4.1, Table 29; HealthBench Professional length-adjusted row, GPT-6 Sol column; value 60.8 (59.5, 1573)",
          "scoreDisplay": "0.608"
        },
        {
          "key": "healthbench-professional:GPT-6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Section 11.4.1, Table 29; HealthBench Professional length-adjusted row, GPT-6 Luna column; value 60.8 (61.2, 2119)",
          "scoreDisplay": "0.608"
        },
        {
          "key": "healthbench-hard:GPT-6 Astra",
          "role": "primary",
          "quote": null,
          "locator": "Section 11.4.1, Table 29; HealthBench Hard length-adjusted row, GPT-6 Astra column; value 36.6 (34.2, 1697)",
          "scoreDisplay": "0.366"
        },
        {
          "key": "healthbench-hard:GPT-6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Section 11.4.1, Table 29; HealthBench Hard length-adjusted row, GPT-6 Luna column; value 31.4 (25.4, 1241)",
          "scoreDisplay": "0.314"
        },
        {
          "key": "healthbench-hard:GPT-6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Section 11.4.1, Table 29; HealthBench Hard length-adjusted row, GPT-6 Sol column; value 30.1 (22.1, 974)",
          "scoreDisplay": "0.301"
        },
        {
          "key": "healthbench:GPT-6 Astra",
          "role": "primary",
          "quote": null,
          "locator": "Section 11.4.1, Table 29; HealthBench length-adjusted row, GPT-6 Astra column; value 58.3 (56.9, 1760)",
          "scoreDisplay": "58.3"
        },
        {
          "key": "healthbench:GPT-6 Luna",
          "role": "primary",
          "quote": null,
          "locator": "Section 11.4.1, Table 29; HealthBench length-adjusted row, GPT-6 Luna column; value 54.5 (50, 1255)",
          "scoreDisplay": "54.5"
        },
        {
          "key": "healthbench:GPT-6 Sol",
          "role": "primary",
          "quote": null,
          "locator": "Section 11.4.1, Table 29; HealthBench length-adjusted row, GPT-6 Sol column; value 53.2 (47.1, 977)",
          "scoreDisplay": "53.2"
        }
      ]
    },
    {
      "id": 10510,
      "title": "Claude Opus 5.5 System Card",
      "url": "https://www.anthropic.com/claude-opus-5-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-09-22",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:Claude Opus 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Section 8.15.2; pp. 213–214, Figures 8.15.1.A and 8.15.2.A",
          "scoreDisplay": "0.656"
        },
        {
          "key": "healthbench:Claude Opus 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Section 8.15.1; pp. 213–214, Figures 8.15.1.A and 8.15.2.A",
          "scoreDisplay": "60.6"
        }
      ]
    },
    {
      "id": 10514,
      "title": "Claude Sonnet 5.5 System Card",
      "url": "https://www.anthropic.com/claude-sonnet-5-5-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-09-28",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:GPT-6 Astra (Anthropic run)",
          "role": "primary",
          "quote": null,
          "locator": "Section 8.15, pp. 137–138; text above Figure 8.15.B: max-effort Astra 70.3 adjusted and 74.0 raw",
          "scoreDisplay": "0.703"
        },
        {
          "key": "healthbench-professional:Claude Sonnet 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Section 8.15.2; pp. 137–139, Figure 8.15.B (max effort)",
          "scoreDisplay": "0.692"
        },
        {
          "key": "healthbench:Claude Sonnet 5.5",
          "role": "primary",
          "quote": null,
          "locator": "Section 8.15.1; pp. 137–139, Figure 8.15.B (max effort)",
          "scoreDisplay": "65.4"
        },
        {
          "key": "physicianbench:Claude Opus 5.5 (max)",
          "role": "primary",
          "quote": null,
          "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1",
          "scoreDisplay": "68.4%"
        },
        {
          "key": "physicianbench:Claude Sonnet 5.5 (max)",
          "role": "primary",
          "quote": null,
          "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1",
          "scoreDisplay": "63.2%"
        },
        {
          "key": "physicianbench:Claude Fable 5.1 (max)",
          "role": "primary",
          "quote": null,
          "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1",
          "scoreDisplay": "61.0%"
        },
        {
          "key": "physicianbench:Claude Opus 5 (max)",
          "role": "primary",
          "quote": null,
          "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1",
          "scoreDisplay": "57.6%"
        },
        {
          "key": "physicianbench:Claude Sonnet 5.5 (xhigh)",
          "role": "primary",
          "quote": null,
          "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1",
          "scoreDisplay": "56.4%"
        },
        {
          "key": "physicianbench:Claude Sonnet 5.5 (high)",
          "role": "primary",
          "quote": null,
          "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1",
          "scoreDisplay": "47.6%"
        },
        {
          "key": "physicianbench:Claude Sonnet 5 (max)",
          "role": "primary",
          "quote": null,
          "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1",
          "scoreDisplay": "37.4%"
        },
        {
          "key": "physicianbench:Claude Sonnet 5.5 (medium)",
          "role": "primary",
          "quote": null,
          "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1",
          "scoreDisplay": "30.0%"
        },
        {
          "key": "physicianbench:Claude Sonnet 5.5 (low)",
          "role": "primary",
          "quote": null,
          "locator": "pp. 137–139, Section 8.15.3, Figure 8.15.B, PhysicianBench pass@1",
          "scoreDisplay": "27.2%"
        }
      ]
    },
    {
      "id": 10515,
      "title": "Claude Fable 5.1 and Claude Mythos 5.1 System Card",
      "url": "https://www.anthropic.com/claude-fable-5-1-mythos-5-1-system-card",
      "kind": "system_card",
      "publisher": "Anthropic",
      "firstParty": true,
      "publishedAt": "2026-09-01",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:Claude Fable 5",
          "role": "primary",
          "quote": null,
          "locator": "p. 199, Figure 8.17.2.A; Claude Fable 5, adjusted bar 63.3%; also p. 167 Table 8.1.A",
          "scoreDisplay": "0.633"
        },
        {
          "key": "healthbench-professional:Claude Fable 5.1",
          "role": "primary",
          "quote": null,
          "locator": "p. 199, sec. 8.17.2 (same figure printed as 62.1% in Table 8.1.A, p. 167)",
          "scoreDisplay": "0.621"
        },
        {
          "key": "healthbench-professional:Claude Opus 5",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 167, Table 8.1.A, column 'Claude Opus 5'; raw 73.4% also reprinted on p. 199, sec. 8.17.2",
          "scoreDisplay": "59.8%"
        },
        {
          "key": "healthbench:Claude Fable 5",
          "role": "primary",
          "quote": null,
          "locator": "p. 198, Figure 8.17.1.A; Claude Fable 5, adjusted bar 60.4%",
          "scoreDisplay": "60.4"
        },
        {
          "key": "healthbench:Claude Fable 5.1",
          "role": "primary",
          "quote": null,
          "locator": "p. 198, sec. 8.17.1 / Figure 8.17.1.A",
          "scoreDisplay": "60%"
        }
      ]
    },
    {
      "id": 10518,
      "title": "Introducing Grok 4.7",
      "url": "https://x.ai/news/grok-4-7",
      "kind": "launch_post",
      "publisher": "SpaceXAI",
      "firstParty": true,
      "publishedAt": "2026-09-21",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-professional:Grok 4.7",
          "role": "primary",
          "quote": null,
          "locator": "Model Improvements table, Clinical reasoning / HealthBench Professional row, Grok 4.7 xhigh column",
          "scoreDisplay": "0.567"
        },
        {
          "key": "healthbench-professional:Grok 4.6",
          "role": "primary",
          "quote": null,
          "locator": "Model Improvements table, Clinical reasoning / HealthBench Professional row, Grok 4.6 high column",
          "scoreDisplay": "0.485"
        }
      ]
    },
    {
      "id": 10524,
      "title": "Baichuan-M3 Technical Report",
      "url": "https://arxiv.org/pdf/2602.06570",
      "kind": "paper",
      "publisher": "Baichuan",
      "firstParty": true,
      "publishedAt": "2026-02-06",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "healthbench-hard:Baichuan-M3",
          "role": "primary",
          "quote": null,
          "locator": "p. 23 (PDF page index 22), section 4.2.1 and Figure 7; HealthBench Hard",
          "scoreDisplay": "0.444"
        },
        {
          "key": "healthbench-hard:GPT-5.2-High (Baichuan run)",
          "role": "primary",
          "quote": null,
          "locator": "p. 23 (PDF page index 22), section 4.2.1 and Figure 7; HealthBench Hard",
          "scoreDisplay": "0.420"
        },
        {
          "key": "healthbench:Baichuan-M3",
          "role": "primary",
          "quote": null,
          "locator": "p. 23, section 4.2.1 HealthBench-Main; Table on p. 25 (Model / HealthBench Score)",
          "scoreDisplay": "65.1"
        },
        {
          "key": "healthbench:Baichuan-M3",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 23, section 4.2.1 HealthBench-Main (Figure 7); also Table 2, p. 25 (Baichuan-M3-235B, HealthBench Score 65.1)",
          "scoreDisplay": "65.1"
        },
        {
          "key": "healthbench:GPT-5.2-High",
          "role": "primary",
          "quote": null,
          "locator": "p. 25, HealthBench-Hallu table (Model / HealthBench Score column); also p. 23 prose",
          "scoreDisplay": "63.3"
        },
        {
          "key": "healthbench:GPT-5.2-High",
          "role": "corroborating",
          "quote": null,
          "locator": "p. 23, section 4.2.1 HealthBench-Main (Figure 7); also Table 2, p. 25 (GPT-5.2-High, HealthBench Score 63.3)",
          "scoreDisplay": "63.3"
        }
      ]
    },
    {
      "id": 10526,
      "title": "Health Optimization Bench — subject suites results",
      "url": "https://healthoptimizationbench.com/sources",
      "kind": "arcophos_run",
      "publisher": "Arcophos",
      "firstParty": true,
      "publishedAt": "2026-09-10",
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "health-optimization-bench:Claude Fable 5",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Claude Fable 5; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.709, n=257",
          "scoreDisplay": "70.9"
        },
        {
          "key": "health-optimization-bench:Claude Opus 5",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Claude Opus 5; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.693, n=257",
          "scoreDisplay": "69.3"
        },
        {
          "key": "health-optimization-bench:Grok 4.6",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Grok 4.6; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.668, n=257",
          "scoreDisplay": "66.8"
        },
        {
          "key": "health-optimization-bench:GPT-5.6 Sol (max)",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; GPT-5.6 Sol (max); tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.666, n=257",
          "scoreDisplay": "66.6"
        },
        {
          "key": "health-optimization-bench:GPT-5.6 Sol (high)",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; GPT-5.6 Sol (high); tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.646, n=257",
          "scoreDisplay": "64.6"
        },
        {
          "key": "health-optimization-bench:Kimi K3",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Kimi K3; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.599, n=257",
          "scoreDisplay": "59.9"
        },
        {
          "key": "health-optimization-bench:Muse Spark",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Muse Spark; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.572, n=257",
          "scoreDisplay": "57.2"
        },
        {
          "key": "health-optimization-bench:Claude Fable 5.1",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Claude Fable 5.1; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.473, n=257",
          "scoreDisplay": "47.3"
        },
        {
          "key": "health-optimization-bench:Gemini 3.6",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Gemini 3.6; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.397, n=257",
          "scoreDisplay": "39.7"
        },
        {
          "key": "health-optimization-bench:Inkling",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Inkling; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.356, n=257",
          "scoreDisplay": "35.6"
        },
        {
          "key": "health-optimization-bench:Claude Sonnet 5",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Claude Sonnet 5; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.346, n=257",
          "scoreDisplay": "34.6"
        },
        {
          "key": "health-optimization-bench:GLM 5.2",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; GLM 5.2; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.207, n=257",
          "scoreDisplay": "20.7"
        },
        {
          "key": "health-optimization-bench:MiniMax M3",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; MiniMax M3; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.183, n=257",
          "scoreDisplay": "18.3"
        },
        {
          "key": "health-optimization-bench:MAI Thinking",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; MAI Thinking; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.175, n=257",
          "scoreDisplay": "17.5"
        },
        {
          "key": "health-optimization-bench:Mistral Medium 3.5",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Mistral Medium 3.5; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.092, n=257",
          "scoreDisplay": "9.2"
        },
        {
          "key": "health-optimization-bench:Nemotron 3.5 Lightning",
          "role": "primary",
          "quote": null,
          "locator": "Subject suites table; Nemotron 3.5 Lightning; tasksets/mb2ev/analysis.json snapshot 2026-09-10; mean 0.049, n=257",
          "scoreDisplay": "4.9"
        }
      ]
    },
    {
      "id": 10542,
      "title": "ARISE MAST technical leaderboard",
      "url": "https://www.arise-ai.org/mast/technical",
      "kind": "official_leaderboard",
      "publisher": "ARISE",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "first-do-noharm:LiSA 2.5",
          "role": "primary",
          "quote": null,
          "locator": "First Do NOHARM v2 overall leaderboard; LiSA 2.5; score 86.2%",
          "scoreDisplay": "86.2"
        },
        {
          "key": "first-do-noharm:Doximity Ask 6.1",
          "role": "primary",
          "quote": null,
          "locator": "First Do NOHARM v2 overall leaderboard; Doximity Ask 6.1; score 84.5%",
          "scoreDisplay": "84.5"
        },
        {
          "key": "first-do-noharm:OpenEvidence",
          "role": "primary",
          "quote": null,
          "locator": "First Do NOHARM v2 overall leaderboard; OpenEvidence; score 80%",
          "scoreDisplay": "80.0"
        },
        {
          "key": "first-do-noharm:Glass 5.6 Max",
          "role": "primary",
          "quote": null,
          "locator": "First Do NOHARM v2 overall leaderboard; Glass 5.6 Max; score 79.7%",
          "scoreDisplay": "79.7"
        },
        {
          "key": "first-do-noharm:GLM 5.1",
          "role": "primary",
          "quote": null,
          "locator": "First Do NOHARM v2 overall leaderboard; GLM 5.1; score 57.9%",
          "scoreDisplay": "57.9"
        }
      ]
    },
    {
      "id": 10547,
      "title": "Artificial Analysis Healthcare & Medical Index",
      "url": "https://artificialanalysis.ai/models/capabilities/healthcare-and-medical",
      "kind": "official_leaderboard",
      "publisher": "Artificial Analysis",
      "firstParty": true,
      "publishedAt": null,
      "retrieved": "2026-09-28",
      "results": [
        {
          "key": "artificial-analysis-healthcare:Claude Opus 5.5 (Adaptive Reasoning, Max Effort, Default Fallback)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=claude-opus-5-5, weightedIndex=60.5359874288733; round to nearest whole point",
          "scoreDisplay": "61"
        },
        {
          "key": "artificial-analysis-healthcare:Claude Fable 5.1 (Adaptive Reasoning, Max Effort, Default Fallback)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=claude-fable-5-1, weightedIndex=58.0046680298741; round to nearest whole point",
          "scoreDisplay": "58"
        },
        {
          "key": "artificial-analysis-healthcare:Claude Opus 5 (Adaptive Reasoning, Max Effort)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=claude-opus-5, weightedIndex=53.3389883833501; round to nearest whole point",
          "scoreDisplay": "53"
        },
        {
          "key": "artificial-analysis-healthcare:GPT-6 Astra (max)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-6-astra, weightedIndex=51.6668965494132; round to nearest whole point",
          "scoreDisplay": "52"
        },
        {
          "key": "artificial-analysis-healthcare:Muse Spark 1.3 (max)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=muse-spark-1-3, weightedIndex=49.6912519301107; round to nearest whole point",
          "scoreDisplay": "50"
        },
        {
          "key": "artificial-analysis-healthcare:Grok 4.7 (xhigh)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=grok-4-7, weightedIndex=47.1903068758764; round to nearest whole point",
          "scoreDisplay": "47"
        },
        {
          "key": "artificial-analysis-healthcare:GLM-5.3 (max)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=glm-5-3, weightedIndex=46.6708125611712; round to nearest whole point",
          "scoreDisplay": "47"
        },
        {
          "key": "artificial-analysis-healthcare:GPT-5.6 Sol (max)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-5-6-sol, weightedIndex=45.4043575131731; round to nearest whole point",
          "scoreDisplay": "45"
        },
        {
          "key": "artificial-analysis-healthcare:Kimi K3 (max)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=kimi-k3, weightedIndex=45.0434943845881; round to nearest whole point",
          "scoreDisplay": "45"
        },
        {
          "key": "artificial-analysis-healthcare:GLM 5.3 Flash",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=glm-5-3-flash, weightedIndex=44.8426192521491; round to nearest whole point",
          "scoreDisplay": "45"
        },
        {
          "key": "artificial-analysis-healthcare:GPT-6 Sol (max)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-6-sol, weightedIndex=43.4880669048048; round to nearest whole point",
          "scoreDisplay": "43"
        },
        {
          "key": "artificial-analysis-healthcare:MiMo-V2.6-Pro",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=mimo-v2-6-pro, weightedIndex=41.8204296860305; round to nearest whole point",
          "scoreDisplay": "42"
        },
        {
          "key": "artificial-analysis-healthcare:Gemini 3.8 Flash (high)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gemini-3-8-flash, weightedIndex=41.7830626872104; round to nearest whole point",
          "scoreDisplay": "42"
        },
        {
          "key": "artificial-analysis-healthcare:Qwen3.8 Max (0902)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=qwen3-8-max, weightedIndex=41.4279151085327; round to nearest whole point",
          "scoreDisplay": "41"
        },
        {
          "key": "artificial-analysis-healthcare:Step 5 Preview",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=step-5, weightedIndex=40.8171769208551; round to nearest whole point",
          "scoreDisplay": "41"
        },
        {
          "key": "artificial-analysis-healthcare:DeepSeek V4.1 Flash (Reasoning, Max Effort)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=deepseek-v4-1-flash, weightedIndex=40.6388737813166; round to nearest whole point",
          "scoreDisplay": "41"
        },
        {
          "key": "artificial-analysis-healthcare:GPT-6 Luna (max)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-6-luna, weightedIndex=36.7252603853009; round to nearest whole point",
          "scoreDisplay": "37"
        },
        {
          "key": "artificial-analysis-healthcare:GPT-5.6 Luna (max)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gpt-5-6-luna, weightedIndex=35.740444759121; round to nearest whole point",
          "scoreDisplay": "36"
        },
        {
          "key": "artificial-analysis-healthcare:Qwen3.8 27B (xhigh)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=qwen3-8-27b, weightedIndex=34.4310443115205; round to nearest whole point",
          "scoreDisplay": "34"
        },
        {
          "key": "artificial-analysis-healthcare:MiniMax-M3",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=minimax-m3, weightedIndex=29.7042505129902; round to nearest whole point",
          "scoreDisplay": "30"
        },
        {
          "key": "artificial-analysis-healthcare:Inkling (xhigh)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=inkling, weightedIndex=25.3933378261902; round to nearest whole point",
          "scoreDisplay": "25"
        },
        {
          "key": "artificial-analysis-healthcare:Nemotron 3 Ultra 550B A55B (Reasoning)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=nvidia-nemotron-3-ultra-550b-a55b, weightedIndex=23.0046725383136; round to nearest whole point",
          "scoreDisplay": "23"
        },
        {
          "key": "artificial-analysis-healthcare:Gemini 3.5 Flash-Lite",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=gemini-3-5-flash-lite, weightedIndex=23.074121756108; round to nearest whole point",
          "scoreDisplay": "23"
        },
        {
          "key": "artificial-analysis-healthcare:Muse Glimmer (high)",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=muse-glimmer, weightedIndex=17.5980321302316; round to nearest whole point",
          "scoreDisplay": "18"
        },
        {
          "key": "artificial-analysis-healthcare:Mistral Medium 3.5",
          "role": "primary",
          "quote": null,
          "locator": "Embedded chart data: capability=healthcareAndMedical, initialModels slug=mistral-medium-3-5, weightedIndex=13.5556785369912; round to nearest whole point",
          "scoreDisplay": "14"
        }
      ]
    }
  ]
}
