{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-05-004",
    "source_story_id": "tmp-story-clinician-preference-safety-proxy",
    "edition_id": "mp-2026-08-05-morning-0027",
    "edition_url": "https://themachinepress.com/edition/2026-08-05",
    "position": 4,
    "story_type": "dispatch",
    "section": "safety-security",
    "editorial_classification": "editorial",
    "headline": "Clinicians Preferred Answers That Still Failed Safety Rubrics",
    "slug": "clinicians-preferred-answers-that-still-failed-safety-rubrics",
    "dek": "More than 26,000 judgments showed that pairwise preference could hide specialty-specific clinical failure rates.",
    "summary": "More than 26,000 judgments showed that pairwise preference could hide specialty-specific clinical failure rates.",
    "body_text": "Using 26,804 blinded pairwise judgments from more than 736 clinicians in over 28 countries, a preprint compared which model answer clinicians preferred with separate rubric scores for accuracy, harmlessness and other safety-critical qualities. Models that ranked well by preference still produced meaningful failures, and those failures varied across specialties. Surface features explained slightly more preference variation than differences in the safety rubrics. The authors propose reporting failure rates directly and adding clinically grounded adjustments rather than treating a single preference ranking as a safety measure.",
    "why_it_matters": "More than 26,000 judgments showed that pairwise preference could hide specialty-specific clinical failure rates.",
    "limitations": [],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-05-004/clinicians-preferred-answers-that-still-failed-safety-rubrics",
    "json_url": "https://themachinepress.com/story/mp-2026-08-05-004.json",
    "first_published_at": "2026-08-05T09:00:00.000-04:00",
    "modified_at": "2026-08-05T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-05-004-001",
        "text": "More than 26,000 judgments showed that pairwise preference could hide specialty-specific clinical failure rates.",
        "source_ids": [
          "source-2026-08-05-004"
        ],
        "qualification": null
      }
    ],
    "source_ids": [
      "source-2026-08-05-004"
    ],
    "tags": [
      "clinical AI",
      "evaluation",
      "preferences",
      "safety"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-05-004",
      "title": "arXiv preprint 2608.02617",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.02617",
      "canonical_url": "https://arxiv.org/abs/2608.02617",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": null,
      "accessed_at": "2026-08-05T08:26:06.151-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-05-004-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Clinicians Preferred Answers That Still Failed Safety Rubrics",
    "publisher": "The Machine Press",
    "published_at": "2026-08-05T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-05-004/clinicians-preferred-answers-that-still-failed-safety-rubrics"
  }
}
