{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-16-006",
    "source_story_id": "tmp-story-kv-cache-tier-placement",
    "edition_id": "mp-2026-09-16-morning-0069",
    "edition_url": "https://themachinepress.com/edition/2026-09-16",
    "position": 6,
    "story_type": "dispatch",
    "section": "infrastructure",
    "editorial_classification": "editorial",
    "headline": "Tier Capacity Beat Clever Cache Prediction",
    "slug": "tier-capacity-beat-clever-cache-prediction",
    "dek": "GPU, CPU and SSD tiering supported 73.02 times more sessions per GPU; predicted reuse and prefetching did not earn their cost.",
    "summary": "GPU, CPU and SSD tiering supported 73.02 times more sessions per GPU; predicted reuse and prefetching did not earn their cost.",
    "body_text": "Long-lived chats and agent loops fill expensive GPU memory with key-value cache blocks. A calibrated simulator compared recency, reuse frequency, predicted reuse and look-ahead prefetch across GPU HBM, CPU DRAM and SSD. The three-tier capacity model supported 73.02 times more concurrent sessions per GPU and cut cost per session 62.04 times, but those gains came from capacity rather than sophisticated placement. Recency minimized chat migration while reuse frequency led for agents and document work. Even an oracle prefetch policy failed to beat no prefetch on migration traffic, challenging recommendations that prediction alone improves this hierarchy.",
    "why_it_matters": "GPU, CPU and SSD tiering supported 73.02 times more sessions per GPU; predicted reuse and prefetching did not earn their cost.",
    "limitations": [],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-16-006/tier-capacity-beat-clever-cache-prediction",
    "json_url": "https://themachinepress.com/story/mp-2026-09-16-006.json",
    "first_published_at": "2026-09-16T09:00:00.000-04:00",
    "modified_at": "2026-09-16T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-16-006-001",
        "text": "GPU, CPU and SSD tiering supported 73.02 times more sessions per GPU; predicted reuse and prefetching did not earn their cost.",
        "source_ids": [
          "source-2026-09-16-006"
        ],
        "qualification": null
      }
    ],
    "source_ids": [
      "source-2026-09-16-006"
    ],
    "tags": [
      "KV cache",
      "memory tiering",
      "LLM serving"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-16-006",
      "title": "arXiv preprint 2609.16215",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.16215",
      "canonical_url": "https://arxiv.org/abs/2609.16215",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-14T20:00:00.000-04:00",
      "accessed_at": "2026-09-16T08:18:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-16-006-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Tier Capacity Beat Clever Cache Prediction",
    "publisher": "The Machine Press",
    "published_at": "2026-09-16T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-16-006/tier-capacity-beat-clever-cache-prediction"
  }
}
