{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-26-011",
    "source_story_id": "tmp-story-brie-living-ehr-benchmark",
    "edition_id": "mp-2026-09-26-morning-0079",
    "edition_url": "https://themachinepress.com/edition/2026-09-26",
    "position": 11,
    "story_type": "dispatch",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "A Clinical Benchmark Learned to Refresh Itself",
    "slug": "a-clinical-benchmark-learned-to-refresh-itself",
    "dek": "Nineteen clinicians validated a generator for questions and answers drawn from longitudinal health records.",
    "summary": "Nineteen clinicians validated a generator for questions and answers drawn from longitudinal health records.",
    "body_text": "BRIE automatically turns longitudinal electronic-health-record notes into retrieval questions, allowing the evaluation set to be refreshed as systems and records evolve. Across nine language models and five inference strategies, the study found frequent omissions of clinically important information, especially when answers required synthesis across several documents and encounters. The generator can also produce multiple acceptable answers reflecting clinician variation. This is an evaluation framework, not a clinical deployment validation or evidence that generated answers are safe without review.",
    "why_it_matters": "Nineteen clinicians validated a generator for questions and answers drawn from longitudinal health records.",
    "limitations": [
      "This is an evaluation framework, not a clinical deployment validation or evidence that generated answers are safe without review."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-26-011/a-clinical-benchmark-learned-to-refresh-itself",
    "json_url": "https://themachinepress.com/story/mp-2026-09-26-011.json",
    "first_published_at": "2026-09-26T09:00:00.000-04:00",
    "modified_at": "2026-09-26T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-26-011-001",
        "text": "Nineteen clinicians validated a generator for questions and answers drawn from longitudinal health records.",
        "source_ids": [
          "source-2026-09-26-011"
        ],
        "qualification": "This is an evaluation framework, not a clinical deployment validation or evidence that generated answers are safe without review."
      }
    ],
    "source_ids": [
      "source-2026-09-26-011"
    ],
    "tags": [
      "clinical AI",
      "electronic health records",
      "benchmarking"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-26-011",
      "title": "arXiv preprint 2609.30205",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.30205",
      "canonical_url": "https://arxiv.org/abs/2609.30205",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-24T13:41:16.000-04:00",
      "accessed_at": "2026-09-26T08:20:31.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-26-011-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "A Clinical Benchmark Learned to Refresh Itself",
    "publisher": "The Machine Press",
    "published_at": "2026-09-26T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-26-011/a-clinical-benchmark-learned-to-refresh-itself"
  }
}
