{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-09-008",
    "source_story_id": "tmp-story-scirigor-evidence-chain-audit",
    "edition_id": "mp-2026-09-09-morning-0062",
    "edition_url": "https://themachinepress.com/edition/2026-09-09",
    "position": 8,
    "story_type": "dispatch",
    "section": "research",
    "editorial_classification": "editorial",
    "headline": "Correct-Looking Claims Survived Broken Evidence",
    "slug": "correct-looking-claims-survived-broken-evidence",
    "dek": "In SciRIGOR, claims agreed with faithful and unfaithful results at almost the same rate, while strict end-to-end evidence success stayed below 18 percent.",
    "summary": "In SciRIGOR, claims agreed with faithful and unfaithful results at almost the same rate, while strict end-to-end evidence success stayed below 18 percent.",
    "body_text": "SciRIGOR evaluates scientific coding agents as linked chains from executable analysis through results and figures to claims. Its 100 cases span six domains and 17 subfields, with typed evidence graphs that distinguish artifact fidelity from the validity of each supporting relation. Across 11 agent-model configurations, claims agreed with faithful results 91.8 percent of the time and with unfaithful results 91.0 percent of the time. No system exceeded 62.6 percent on the soft evidence-chain score or 18 percent on strict whole-chain success. Internal coherence therefore did not establish scientific correctness in this benchmark.",
    "why_it_matters": "In SciRIGOR, claims agreed with faithful and unfaithful results at almost the same rate, while strict end-to-end evidence success stayed below 18 percent.",
    "limitations": [
      "Internal coherence therefore did not establish scientific correctness in this benchmark."
    ],
    "importance": 9,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-09-008/correct-looking-claims-survived-broken-evidence",
    "json_url": "https://themachinepress.com/story/mp-2026-09-09-008.json",
    "first_published_at": "2026-09-09T09:00:00.000-04:00",
    "modified_at": "2026-09-09T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-09-008-001",
        "text": "In SciRIGOR, claims agreed with faithful and unfaithful results at almost the same rate, while strict end-to-end evidence success stayed below 18 percent.",
        "source_ids": [
          "source-2026-09-09-008"
        ],
        "qualification": "Internal coherence therefore did not establish scientific correctness in this benchmark."
      }
    ],
    "source_ids": [
      "source-2026-09-09-008"
    ],
    "tags": [
      "scientific agents",
      "evidence chains",
      "evaluation"
    ],
    "image_url": "https://themachinepress.com/issues/2026-09-09/scirigor-code-file.webp",
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-09-008",
      "title": "arXiv preprint 2609.06192",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.06192",
      "canonical_url": "https://arxiv.org/abs/2609.06192",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-05T13:29:15.000-04:00",
      "accessed_at": "2026-09-09T08:30:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-09-008-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Correct-Looking Claims Survived Broken Evidence",
    "publisher": "The Machine Press",
    "published_at": "2026-09-09T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-09-008/correct-looking-claims-survived-broken-evidence"
  }
}
