{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-18-015",
    "source_story_id": "tmp-sidebar-q-based-variational-irl",
    "edition_id": "mp-2026-08-18-morning-0040",
    "edition_url": "https://themachinepress.com/edition/2026-08-18",
    "position": 17,
    "story_type": "ticker",
    "section": "research",
    "editorial_classification": "editorial",
    "headline": "Reward Uncertainty Moved Into Q-Space",
    "slug": "reward-uncertainty-moved-into-q-space",
    "dek": "QVIRL learns a variational distribution over optimal Q-values and recovers a posterior over rewards, including from raw pixels.",
    "summary": "QVIRL learns a variational distribution over optimal Q-values and recovers a posterior over rewards, including from raw pixels.",
    "body_text": "The Bayesian inverse-reinforcement-learning method combines uncertainty estimates with experiments across grid worlds, Lunar Lander, highway driving and two Atari games. The authors call it the first Bayesian IRL method demonstrated from raw pixel observations; performance remains tied to the studied apprenticeship-learning settings.",
    "why_it_matters": "QVIRL learns a variational distribution over optimal Q-values and recovers a posterior over rewards, including from raw pixels.",
    "limitations": [],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-18-015/reward-uncertainty-moved-into-q-space",
    "json_url": "https://themachinepress.com/story/mp-2026-08-18-015.json",
    "first_published_at": "2026-08-18T09:00:00.000-04:00",
    "modified_at": "2026-08-18T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-18-015-001",
        "text": "QVIRL learns a variational distribution over optimal Q-values and recovers a posterior over rewards, including from raw pixels.",
        "source_ids": [
          "source-2026-08-18-017"
        ],
        "qualification": null
      }
    ],
    "source_ids": [
      "source-2026-08-18-017"
    ],
    "tags": [
      "inverse reinforcement learning",
      "uncertainty",
      "Q-values"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-18-017",
      "title": "arXiv preprint 2608.16888",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.16888",
      "canonical_url": "https://arxiv.org/abs/2608.16888",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-16T20:00:00.000-04:00",
      "accessed_at": "2026-08-18T08:25:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-18-015-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Reward Uncertainty Moved Into Q-Space",
    "publisher": "The Machine Press",
    "published_at": "2026-08-18T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-18-015/reward-uncertainty-moved-into-q-space"
  }
}
