{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-11-005",
    "source_story_id": "tmp-story-belief-shift-rl-branching",
    "edition_id": "mp-2026-09-11-morning-0064",
    "edition_url": "https://themachinepress.com/edition/2026-09-11",
    "position": 5,
    "story_type": "dispatch",
    "section": "research",
    "editorial_classification": "editorial",
    "headline": "The Best Fork Came Just Before the Model Changed Its Mind",
    "slug": "the-best-fork-came-just-before-the-model-changed-its-mind",
    "dek": "Belief-shift branching placed scarce reinforcement-learning rollouts near value pivots and led the reported math and code comparisons.",
    "summary": "Belief-shift branching placed scarce reinforcement-learning rollouts near value pivots and led the reported math and code comparisons.",
    "body_text": "Tree-structured reinforcement learning gets step-level credit by forking a reasoning chain and comparing sibling outcomes, but every fork costs samples. This work probes the model’s answer belief at candidate boundaries and branches just before consecutive beliefs diverge most. The placement probe consumed about 1% of step compute on math and under 5% on code. It ranked first against Monte Carlo value curves in eight model-benchmark panels; in training, it improved OLMo-3-7B’s math aggregate by 2.6 points and LiveCodeBench-medium by 6.5 over the strongest reported baseline. Results remain limited to the tested model families and verifiable-reward domains.",
    "why_it_matters": "Belief-shift branching placed scarce reinforcement-learning rollouts near value pivots and led the reported math and code comparisons.",
    "limitations": [
      "Results remain limited to the tested model families and verifiable-reward domains."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-11-005/the-best-fork-came-just-before-the-model-changed-its-mind",
    "json_url": "https://themachinepress.com/story/mp-2026-09-11-005.json",
    "first_published_at": "2026-09-11T09:00:00.000-04:00",
    "modified_at": "2026-09-11T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-11-005-001",
        "text": "Belief-shift branching placed scarce reinforcement-learning rollouts near value pivots and led the reported math and code comparisons.",
        "source_ids": [
          "source-2026-09-11-005"
        ],
        "qualification": "Results remain limited to the tested model families and verifiable-reward domains."
      }
    ],
    "source_ids": [
      "source-2026-09-11-005"
    ],
    "tags": [
      "reinforcement learning",
      "reasoning",
      "credit assignment"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-11-005",
      "title": "arXiv preprint 2609.11061",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.11061",
      "canonical_url": "https://arxiv.org/abs/2609.11061",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-10T00:06:45.000-04:00",
      "accessed_at": "2026-09-11T08:25:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-11-005-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Best Fork Came Just Before the Model Changed Its Mind",
    "publisher": "The Machine Press",
    "published_at": "2026-09-11T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-11-005/the-best-fork-came-just-before-the-model-changed-its-mind"
  }
}
