{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-18-019",
    "source_story_id": "tmp-sidebar-chive-counterfactual-explanations",
    "edition_id": "mp-2026-08-18-morning-0040",
    "edition_url": "https://themachinepress.com/edition/2026-08-18",
    "position": 21,
    "story_type": "ticker",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "The Explanation Had to Predict a Changed Prompt",
    "slug": "the-explanation-had-to-predict-a-changed-prompt",
    "dek": "CHIVE tests behavioral explanations with counterfactual edits and found no uplift from the interpretability techniques it studied.",
    "summary": "CHIVE tests behavioral explanations with counterfactual edits and found no uplift from the interpretability techniques it studied.",
    "body_text": "The agentic pipeline locates unexpected model behaviors, edits prompts and asks whether an explanation predicts the counterfactual outcome. Common interpretability techniques did not improve that prediction task, while training on CHIVE experiments generalized to reported out-of-distribution settings. The negative result applies to the evaluated methods and models.",
    "why_it_matters": "CHIVE tests behavioral explanations with counterfactual edits and found no uplift from the interpretability techniques it studied.",
    "limitations": [
      "Common interpretability techniques did not improve that prediction task, while training on CHIVE experiments generalized to reported out-of-distribution settings."
    ],
    "importance": 9,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-18-019/the-explanation-had-to-predict-a-changed-prompt",
    "json_url": "https://themachinepress.com/story/mp-2026-08-18-019.json",
    "first_published_at": "2026-08-18T09:00:00.000-04:00",
    "modified_at": "2026-08-18T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-18-019-001",
        "text": "CHIVE tests behavioral explanations with counterfactual edits and found no uplift from the interpretability techniques it studied.",
        "source_ids": [
          "source-2026-08-18-021"
        ],
        "qualification": "Common interpretability techniques did not improve that prediction task, while training on CHIVE experiments generalized to reported out-of-distribution settings."
      }
    ],
    "source_ids": [
      "source-2026-08-18-021"
    ],
    "tags": [
      "interpretability",
      "counterfactuals",
      "LLM behavior"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-18-021",
      "title": "arXiv preprint 2608.16747",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.16747",
      "canonical_url": "https://arxiv.org/abs/2608.16747",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-16T20:00:00.000-04:00",
      "accessed_at": "2026-08-18T08:25:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-18-019-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Explanation Had to Predict a Changed Prompt",
    "publisher": "The Machine Press",
    "published_at": "2026-08-18T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-18-019/the-explanation-had-to-predict-a-changed-prompt"
  }
}
