{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-19-008",
    "source_story_id": "tmp-story-opted-render-free-driving-teacher",
    "edition_id": "mp-2026-09-19-morning-0072",
    "edition_url": "https://themachinepress.com/edition/2026-09-19",
    "position": 8,
    "story_type": "dispatch",
    "section": "robotics",
    "editorial_classification": "editorial",
    "headline": "Driving Policies Improved With Three Orders Fewer Simulator Interactions",
    "slug": "driving-policies-improved-with-three-orders-fewer-simulator-interactions",
    "dek": "A vector-input teacher lifted two camera-policy driving scores by 1.6× and 9.5× in AlpaSim.",
    "summary": "A vector-input teacher lifted two camera-policy driving scores by 1.6× and 9.5× in AlpaSim.",
    "body_text": "OPTED separates reinforcement learning from camera-policy post-training. A privileged teacher learns from vectorized maps and boxes, then supervises pretrained TransFuser and VaVAM students in closed loop on neural reconstructions of real driving logs. The reported driving scores increased 1.6 times and 9.5 times. In controlled experiments, the approach matched direct reinforcement-learning post-training with roughly one-thousandth as many simulator interactions while remaining closer to the human demonstration prior; this is simulation evidence, not public-road validation.",
    "why_it_matters": "A vector-input teacher lifted two camera-policy driving scores by 1.6× and 9.5× in AlpaSim.",
    "limitations": [
      "In controlled experiments, the approach matched direct reinforcement-learning post-training with roughly one-thousandth as many simulator interactions while remaining closer to the human demonstration prior; this is simulation evidence, not public-road validation."
    ],
    "importance": 9,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-19-008/driving-policies-improved-with-three-orders-fewer-simulator-interactions",
    "json_url": "https://themachinepress.com/story/mp-2026-09-19-008.json",
    "first_published_at": "2026-09-19T09:00:00.000-04:00",
    "modified_at": "2026-09-19T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-19-008-001",
        "text": "A vector-input teacher lifted two camera-policy driving scores by 1.6× and 9.5× in AlpaSim.",
        "source_ids": [
          "source-2026-09-19-008"
        ],
        "qualification": "In controlled experiments, the approach matched direct reinforcement-learning post-training with roughly one-thousandth as many simulator interactions while remaining closer to the human demonstration prior; this is simulation evidence, not public-road validation."
      }
    ],
    "source_ids": [
      "source-2026-09-19-008"
    ],
    "tags": [
      "autonomous driving",
      "closed-loop training",
      "simulation"
    ],
    "image_url": "https://themachinepress.com/issues/2026-09-19/opted-bridge-file.webp",
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-19-008",
      "title": "arXiv preprint 2609.20756",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.20756",
      "canonical_url": "https://arxiv.org/abs/2609.20756",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-17T13:41:51.000-04:00",
      "accessed_at": "2026-09-19T08:30:21.326-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-19-008-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Driving Policies Improved With Three Orders Fewer Simulator Interactions",
    "publisher": "The Machine Press",
    "published_at": "2026-09-19T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-19-008/driving-policies-improved-with-three-orders-fewer-simulator-interactions"
  }
}
