{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-14-012",
    "source_story_id": "tmp-story-spade-edge-cloud-inference",
    "edition_id": "mp-2026-08-14-morning-0036",
    "edition_url": "https://themachinepress.com/edition/2026-08-14",
    "position": 12,
    "story_type": "dispatch",
    "section": "infrastructure",
    "editorial_classification": "editorial",
    "headline": "The Edge Drafted. The Cloud Corrected Only the Misses",
    "slug": "the-edge-drafted-the-cloud-corrected-only-the-misses",
    "dek": "Distributed speculative decoding cut verifier calls by 76 percent in the reported tests without changing the larger model's accepted output.",
    "summary": "Distributed speculative decoding cut verifier calls by 76 percent in the reported tests without changing the larger model's accepted output.",
    "body_text": "SPADE places a small draft model on the edge and asks a cloud model to verify candidate tokens in parallel. Accepted tokens stay local to the draft path, while rejected ones trigger correction, shifting much of the computation away from repeated cloud generation. Across SpecBench and CNN/DailyMail tasks, the authors report 76 percent fewer cloud-model calls with no accuracy loss relative to using the full model throughout. Network conditions, privacy implications and provider pricing were not established as universal advantages by the benchmark.",
    "why_it_matters": "Distributed speculative decoding cut verifier calls by 76 percent in the reported tests without changing the larger model's accepted output.",
    "limitations": [
      "Network conditions, privacy implications and provider pricing were not established as universal advantages by the benchmark."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-14-012/the-edge-drafted-the-cloud-corrected-only-the-misses",
    "json_url": "https://themachinepress.com/story/mp-2026-08-14-012.json",
    "first_published_at": "2026-08-14T09:00:00.000-04:00",
    "modified_at": "2026-08-14T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-14-012-001",
        "text": "Distributed speculative decoding cut verifier calls by 76 percent in the reported tests without changing the larger model's accepted output.",
        "source_ids": [
          "source-2026-08-14-012"
        ],
        "qualification": "Network conditions, privacy implications and provider pricing were not established as universal advantages by the benchmark."
      }
    ],
    "source_ids": [
      "source-2026-08-14-012"
    ],
    "tags": [
      "edge computing",
      "speculative decoding",
      "cloud inference",
      "efficiency"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-14-012",
      "title": "arXiv preprint 2608.13076",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.13076",
      "canonical_url": "https://arxiv.org/abs/2608.13076",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-13T06:43:57.000-04:00",
      "accessed_at": "2026-08-14T08:27:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-14-012-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Edge Drafted. The Cloud Corrected Only the Misses",
    "publisher": "The Machine Press",
    "published_at": "2026-08-14T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-14-012/the-edge-drafted-the-cloud-corrected-only-the-misses"
  }
}
