{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-10-026",
    "source_story_id": "tmp-story-diffusion-llm-safety-neurons",
    "edition_id": "mp-2026-08-10-morning-0032",
    "edition_url": "https://themachinepress.com/edition/2026-08-10",
    "position": 15,
    "story_type": "dispatch",
    "section": "safety-security",
    "editorial_classification": "editorial",
    "headline": "Diffusion Models Carried Safety in a Few Neurons",
    "slug": "diffusion-models-carried-safety-in-a-few-neurons",
    "dek": "Mechanistic attacks mapped and pruned sparse safety features inherited from autoregressive parent models.",
    "summary": "Mechanistic attacks mapped and pruned sparse safety features inherited from autoregressive parent models.",
    "body_text": "Researchers tested diffusion language models that generate by iterative denoising rather than next-token prediction. They report that pruning mapped safety neurons raised attack success rates from 2.6 to 73.8 percent on LLaDA and from 1.9 to 86.6 percent on Dream; a separate offline steering method transferred attacks to several targets with reported success as high as 86.9 percent. These figures come from the authors' threat model and preprint codebase and should not be generalized to every diffusion model.",
    "why_it_matters": "Mechanistic attacks mapped and pruned sparse safety features inherited from autoregressive parent models.",
    "limitations": [
      "These figures come from the authors' threat model and preprint codebase and should not be generalized to every diffusion model."
    ],
    "importance": 9,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-10-026/diffusion-models-carried-safety-in-a-few-neurons",
    "json_url": "https://themachinepress.com/story/mp-2026-08-10-026.json",
    "first_published_at": "2026-08-10T09:00:00.000-04:00",
    "modified_at": "2026-08-10T15:50:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-10-026-001",
        "text": "Mechanistic attacks mapped and pruned sparse safety features inherited from autoregressive parent models.",
        "source_ids": [
          "source-2026-08-10-015"
        ],
        "qualification": "These figures come from the authors' threat model and preprint codebase and should not be generalized to every diffusion model."
      }
    ],
    "source_ids": [
      "source-2026-08-10-015"
    ],
    "tags": [
      "diffusion language models",
      "jailbreaks",
      "mechanistic safety"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-10-015",
      "title": "arXiv preprint 2608.07430",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.07430",
      "canonical_url": "https://arxiv.org/abs/2608.07430",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-07T13:00:00.000-04:00",
      "accessed_at": "2026-08-10T15:30:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-10-026-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Diffusion Models Carried Safety in a Few Neurons",
    "publisher": "The Machine Press",
    "published_at": "2026-08-10T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-10-026/diffusion-models-carried-safety-in-a-few-neurons"
  }
}
