{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-24-007",
    "source_story_id": "tmp-story-security-agent-capability-exposure",
    "edition_id": "mp-2026-08-24-morning-0046",
    "edition_url": "https://themachinepress.com/edition/2026-08-24",
    "position": 7,
    "story_type": "dispatch",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "The Security Agent Failed Before the Tested Capability Appeared",
    "slug": "the-security-agent-failed-before-the-tested-capability-appeared",
    "dek": "Checkpoint instrumentation separates long-horizon failures that happen before and after an agent reaches the relevant state.",
    "summary": "Checkpoint instrumentation separates long-horizon failures that happen before and after an agent reaches the relevant state.",
    "body_text": "In one 92-seed study, protocol-disambiguation guidance raised state observation for Gemini 2.5 Flash from 65.5 to 95.4 percent, but repeating the design with Gemini 3.7 Flash produced the opposite effect. The shifting bottleneck shows why final success alone cannot reveal which capability failed or whether it was exercised at all.",
    "why_it_matters": "Checkpoint instrumentation separates long-horizon failures that happen before and after an agent reaches the relevant state.",
    "limitations": [
      "The shifting bottleneck shows why final success alone cannot reveal which capability failed or whether it was exercised at all."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-24-007/the-security-agent-failed-before-the-tested-capability-appeared",
    "json_url": "https://themachinepress.com/story/mp-2026-08-24-007.json",
    "first_published_at": "2026-08-24T09:00:00.000-04:00",
    "modified_at": "2026-08-24T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-24-007-001",
        "text": "Checkpoint instrumentation separates long-horizon failures that happen before and after an agent reaches the relevant state.",
        "source_ids": [
          "source-2026-08-24-007"
        ],
        "qualification": "The shifting bottleneck shows why final success alone cannot reveal which capability failed or whether it was exercised at all."
      }
    ],
    "source_ids": [
      "source-2026-08-24-007"
    ],
    "tags": [
      "security agents",
      "long-horizon tasks",
      "diagnostics"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-24-007",
      "title": "arXiv preprint 2608.20563",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.20563",
      "canonical_url": "https://arxiv.org/abs/2608.20563",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-20T16:55:22.000-04:00",
      "accessed_at": "2026-08-24T08:23:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-24-007-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Security Agent Failed Before the Tested Capability Appeared",
    "publisher": "The Machine Press",
    "published_at": "2026-08-24T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-24-007/the-security-agent-failed-before-the-tested-capability-appeared"
  }
}
