{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-24-011",
    "source_story_id": "tmp-story-dreambench-swe-memory-hygiene",
    "edition_id": "mp-2026-08-24-morning-0046",
    "edition_url": "https://themachinepress.com/edition/2026-08-24",
    "position": 11,
    "story_type": "dispatch",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "Literal Memory Won More Sessions but Proved No General Mechanism",
    "slug": "literal-memory-won-more-sessions-but-proved-no-general-mechanism",
    "dek": "DreamBench-SWE uses hidden executable oracles for software tasks that depend on non-inferable evidence from earlier sessions.",
    "summary": "DreamBench-SWE uses hidden executable oracles for software tasks that depend on non-inferable evidence from earlier sessions.",
    "body_text": "In the preregistered successor audit, no external memory passed 21 of 180 tasks, deterministic verbatim memory passed 82 and one pinned hosted configuration passed 97. The authors explicitly stop short of claiming mechanism, general product superiority or equivalence, making the benchmark a profile of exact conditions rather than a universal memory ranking.",
    "why_it_matters": "DreamBench-SWE uses hidden executable oracles for software tasks that depend on non-inferable evidence from earlier sessions.",
    "limitations": [],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-24-011/literal-memory-won-more-sessions-but-proved-no-general-mechanism",
    "json_url": "https://themachinepress.com/story/mp-2026-08-24-011.json",
    "first_published_at": "2026-08-24T09:00:00.000-04:00",
    "modified_at": "2026-08-24T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-24-011-001",
        "text": "DreamBench-SWE uses hidden executable oracles for software tasks that depend on non-inferable evidence from earlier sessions.",
        "source_ids": [
          "source-2026-08-24-011"
        ],
        "qualification": null
      }
    ],
    "source_ids": [
      "source-2026-08-24-011"
    ],
    "tags": [
      "software agents",
      "memory",
      "preregistration"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-24-011",
      "title": "arXiv preprint 2608.20664",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.20664",
      "canonical_url": "https://arxiv.org/abs/2608.20664",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-20T21:48:57.000-04:00",
      "accessed_at": "2026-08-24T08:23:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-24-011-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Literal Memory Won More Sessions but Proved No General Mechanism",
    "publisher": "The Machine Press",
    "published_at": "2026-08-24T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-24-011/literal-memory-won-more-sessions-but-proved-no-general-mechanism"
  }
}
