{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-19-017",
    "source_story_id": "tmp-sidebar-q-learning-world-model-search",
    "edition_id": "mp-2026-08-19-morning-0041",
    "edition_url": "https://themachinepress.com/edition/2026-08-19",
    "position": 19,
    "story_type": "ticker",
    "section": "robotics",
    "editorial_classification": "editorial",
    "headline": "The World Model Searched Without Training on Imagined Data",
    "slug": "the-world-model-searched-without-training-on-imagined-data",
    "dek": "QWM used predicted trajectories to select actions while keeping policy and value learning grounded in real transitions.",
    "summary": "QWM used predicted trajectories to select actions while keeping policy and value learning grounded in real transitions.",
    "body_text": "Test-time search over imagined rollouts improved sample efficiency and performance on Robomimic and LIBERO in the authors’ comparison. Avoiding direct training on imagined transitions reduces one source of compounding bias but does not eliminate world-model error during action selection.",
    "why_it_matters": "QWM used predicted trajectories to select actions while keeping policy and value learning grounded in real transitions.",
    "limitations": [
      "Avoiding direct training on imagined transitions reduces one source of compounding bias but does not eliminate world-model error during action selection."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-19-017/the-world-model-searched-without-training-on-imagined-data",
    "json_url": "https://themachinepress.com/story/mp-2026-08-19-017.json",
    "first_published_at": "2026-08-19T09:00:00.000-04:00",
    "modified_at": "2026-08-19T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-19-017-001",
        "text": "QWM used predicted trajectories to select actions while keeping policy and value learning grounded in real transitions.",
        "source_ids": [
          "source-2026-08-19-019"
        ],
        "qualification": "Avoiding direct training on imagined transitions reduces one source of compounding bias but does not eliminate world-model error during action selection."
      }
    ],
    "source_ids": [
      "source-2026-08-19-019"
    ],
    "tags": [
      "world models",
      "Q-learning",
      "robot manipulation"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-19-019",
      "title": "arXiv preprint 2608.17163",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.17163",
      "canonical_url": "https://arxiv.org/abs/2608.17163",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-16T20:00:00.000-04:00",
      "accessed_at": "2026-08-19T08:22:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-19-017-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The World Model Searched Without Training on Imagined Data",
    "publisher": "The Machine Press",
    "published_at": "2026-08-19T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-19-017/the-world-model-searched-without-training-on-imagined-data"
  }
}
