{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-26-026",
    "source_story_id": "tmp-story-roblox-search-aware-rl",
    "edition_id": "mp-2026-09-26-morning-0079",
    "edition_url": "https://themachinepress.com/edition/2026-09-26",
    "position": 15,
    "story_type": "dispatch",
    "section": "business-enterprise",
    "editorial_classification": "editorial",
    "headline": "Search Rewards Improved Query Understanding",
    "slug": "search-rewards-improved-query-understanding",
    "dek": "Roblox experiments raised NDCG@20 by 8.9 points over supervised fine-tuning and 3.5 over one end-to-end reward.",
    "summary": "Roblox experiments raised NDCG@20 by 8.9 points over supervised fine-tuning and 3.5 over one end-to-end reward.",
    "body_text": "The framework first distills a teacher into a schema-compliant query-understanding policy, then optimizes intent classification, query expansion and other components with rewards drawn from their actual interaction with the search engine. Giving each component an operational reward improved both component utility and downstream retrieval in the reported Roblox experiments. The result is specific to that game-search pipeline and does not show that reinforcement learning will improve every production search system.",
    "why_it_matters": "Roblox experiments raised NDCG@20 by 8.9 points over supervised fine-tuning and 3.5 over one end-to-end reward.",
    "limitations": [
      "The result is specific to that game-search pipeline and does not show that reinforcement learning will improve every production search system."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-26-026/search-rewards-improved-query-understanding",
    "json_url": "https://themachinepress.com/story/mp-2026-09-26-026.json",
    "first_published_at": "2026-09-26T09:00:00.000-04:00",
    "modified_at": "2026-09-26T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-26-026-001",
        "text": "Roblox experiments raised NDCG@20 by 8.9 points over supervised fine-tuning and 3.5 over one end-to-end reward.",
        "source_ids": [
          "source-2026-09-26-015"
        ],
        "qualification": "The result is specific to that game-search pipeline and does not show that reinforcement learning will improve every production search system."
      }
    ],
    "source_ids": [
      "source-2026-09-26-015"
    ],
    "tags": [
      "search",
      "query understanding",
      "reinforcement learning"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-26-015",
      "title": "arXiv preprint 2609.30177",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.30177",
      "canonical_url": "https://arxiv.org/abs/2609.30177",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-24T13:26:46.000-04:00",
      "accessed_at": "2026-09-26T08:20:31.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-26-026-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Search Rewards Improved Query Understanding",
    "publisher": "The Machine Press",
    "published_at": "2026-09-26T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-26-026/search-rewards-improved-query-understanding"
  }
}
