{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-25-027",
    "source_story_id": "tmp-story-explorationbench-alien-worlds-front-rail",
    "edition_id": "mp-2026-09-25-morning-0078",
    "edition_url": "https://themachinepress.com/edition/2026-09-25",
    "position": 16,
    "story_type": "dispatch",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "AI Exploration Sometimes Got Worse With More Searching",
    "slug": "ai-exploration-sometimes-got-worse-with-more-searching",
    "dek": "Two executable alien-world sandboxes separate learning unfamiliar rules from recalling familiar ones.",
    "summary": "Two executable alien-world sandboxes separate learning unfamiliar rules from recalling familiar ones.",
    "body_text": "ExplorationBench creates AlienCode and AlienLogic environments whose executable rules conflict with familiar knowledge. Together they contain 55 discovery targets and 140 tasks, each with a flawed manual, environmental feedback and a tool schema. Across ten systems, the strongest could acquire and apply unfamiliar rules, but results varied substantially by trajectory and continued exploration sometimes stalled or reversed earlier gains. The benchmark tests synthetic worlds, not open-ended scientific discovery.",
    "why_it_matters": "Two executable alien-world sandboxes separate learning unfamiliar rules from recalling familiar ones.",
    "limitations": [
      "The benchmark tests synthetic worlds, not open-ended scientific discovery."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-25-027/ai-exploration-sometimes-got-worse-with-more-searching",
    "json_url": "https://themachinepress.com/story/mp-2026-09-25-027.json",
    "first_published_at": "2026-09-25T09:00:00.000-04:00",
    "modified_at": "2026-09-25T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-25-027-001",
        "text": "Two executable alien-world sandboxes separate learning unfamiliar rules from recalling familiar ones.",
        "source_ids": [
          "source-2026-09-25-016"
        ],
        "qualification": "The benchmark tests synthetic worlds, not open-ended scientific discovery."
      }
    ],
    "source_ids": [
      "source-2026-09-25-016"
    ],
    "tags": [
      "AI exploration",
      "benchmark",
      "scientific discovery"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-25-016",
      "title": "arXiv preprint 2609.30199",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.30199",
      "canonical_url": "https://arxiv.org/abs/2609.30199",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-24T13:37:14.000-04:00",
      "accessed_at": "2026-09-25T08:29:03.434-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-25-027-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "AI Exploration Sometimes Got Worse With More Searching",
    "publisher": "The Machine Press",
    "published_at": "2026-09-25T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-25-027/ai-exploration-sometimes-got-worse-with-more-searching"
  }
}
