{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-27-008",
    "source_story_id": "tmp-story-alignment-auditors",
    "edition_id": "mp-2026-08-27-morning-0049",
    "edition_url": "https://themachinepress.com/edition/2026-08-27",
    "position": 8,
    "story_type": "dispatch",
    "section": "safety-security",
    "editorial_classification": "editorial",
    "headline": "The Auditor Trained Against Hidden Behaviors",
    "slug": "the-auditor-trained-against-hidden-behaviors",
    "dek": "Reinforcement learning improved model investigations while negative examples helped keep false positives below one percent.",
    "summary": "Reinforcement learning improved model investigations while negative examples helped keep false positives below one percent.",
    "body_text": "The training environment planted hidden behaviors through target system prompts and rewarded investigations by pairwise comparison with references. The authors report stronger investigations, more concerning behaviors surfaced in unmodified production models, improved realism and cross-scaffold generalization, with false positives below one percent in tested settings.",
    "why_it_matters": "Reinforcement learning improved model investigations while negative examples helped keep false positives below one percent.",
    "limitations": [],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-27-008/the-auditor-trained-against-hidden-behaviors",
    "json_url": "https://themachinepress.com/story/mp-2026-08-27-008.json",
    "first_published_at": "2026-08-27T09:00:00.000-04:00",
    "modified_at": "2026-08-27T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-27-008-001",
        "text": "Reinforcement learning improved model investigations while negative examples helped keep false positives below one percent.",
        "source_ids": [
          "source-2026-08-27-008"
        ],
        "qualification": null
      }
    ],
    "source_ids": [
      "source-2026-08-27-008"
    ],
    "tags": [
      "alignment auditing",
      "reinforcement learning",
      "hidden behaviors"
    ],
    "image_url": "https://themachinepress.com/issues/2026-08-27/alignment-auditor-file.webp",
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-27-008",
      "title": "arXiv preprint 2608.25460",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.25460",
      "canonical_url": "https://arxiv.org/abs/2608.25460",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-26T03:28:09.000-04:00",
      "accessed_at": "2026-08-27T08:18:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-27-008-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Auditor Trained Against Hidden Behaviors",
    "publisher": "The Machine Press",
    "published_at": "2026-08-27T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-27-008/the-auditor-trained-against-hidden-behaviors"
  }
}
