{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-16-005",
    "source_story_id": "tmp-story-calibrated-disaggregated-router",
    "edition_id": "mp-2026-09-16-morning-0069",
    "edition_url": "https://themachinepress.com/edition/2026-09-16",
    "position": 5,
    "story_type": "dispatch",
    "section": "infrastructure",
    "editorial_classification": "editorial",
    "headline": "Six GPUs Matched Seven After the Router Was Calibrated",
    "slug": "six-gpus-matched-seven-after-the-router-was-calibrated",
    "dek": "A learned request router reached 0.864 mean goodput and matched round robin's seven-GPU result with six GPUs.",
    "summary": "A learned request router reached 0.864 mean goodput and matched round robin's seven-GPU result with six GPUs.",
    "body_text": "Disaggregated serving separates prompt prefill from token decoding, making request placement a latency and capacity problem. This study predicts completion time from prompt length, expected output, post-admission cache pressure and service class, then calibrates the scorer on eight NVIDIA A40 GPUs. Across three bursty traces, mean goodput reached 0.864 versus 0.835 to 0.847 for round robin, least-loaded and length heuristics. Calibration mattered: simulator-only constants erased 4.5 goodput points and much of the tail-latency gain. Under extreme scarcity, greedy routing concentrated work too aggressively, marking a clear operating boundary.",
    "why_it_matters": "A learned request router reached 0.864 mean goodput and matched round robin's seven-GPU result with six GPUs.",
    "limitations": [
      "Calibration mattered: simulator-only constants erased 4.5 goodput points and much of the tail-latency gain."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-16-005/six-gpus-matched-seven-after-the-router-was-calibrated",
    "json_url": "https://themachinepress.com/story/mp-2026-09-16-005.json",
    "first_published_at": "2026-09-16T09:00:00.000-04:00",
    "modified_at": "2026-09-16T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-16-005-001",
        "text": "A learned request router reached 0.864 mean goodput and matched round robin's seven-GPU result with six GPUs.",
        "source_ids": [
          "source-2026-09-16-005"
        ],
        "qualification": "Calibration mattered: simulator-only constants erased 4.5 goodput points and much of the tail-latency gain."
      }
    ],
    "source_ids": [
      "source-2026-09-16-005"
    ],
    "tags": [
      "LLM serving",
      "request routing",
      "GPU efficiency"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-16-005",
      "title": "arXiv preprint 2609.16206",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.16206",
      "canonical_url": "https://arxiv.org/abs/2609.16206",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-14T20:00:00.000-04:00",
      "accessed_at": "2026-09-16T08:18:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-16-005-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Six GPUs Matched Seven After the Router Was Calibrated",
    "publisher": "The Machine Press",
    "published_at": "2026-09-16T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-16-005/six-gpus-matched-seven-after-the-router-was-calibrated"
  }
}
