{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-17-007",
    "source_story_id": "tmp-story-bayesian-evaluation-stopping",
    "edition_id": "mp-2026-08-17-morning-0039",
    "edition_url": "https://themachinepress.com/edition/2026-08-17",
    "position": 7,
    "story_type": "dispatch",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "The Benchmark Stopped Sampling What It Already Knew",
    "slug": "the-benchmark-stopped-sampling-what-it-already-knew",
    "dek": "A Bayesian stopping rule removed 57 to 97 percent of planned trials in illustrative evaluations while preserving the overall conclusion.",
    "summary": "A Bayesian stopping rule removed 57 to 97 percent of planned trials in illustrative evaluations while preserving the overall conclusion.",
    "body_text": "Optstop treats evaluation as sequential measurement instead of assigning every item the same fixed number of trials. It keeps uncertain items eligible, stops when estimates are precise or stable, and becomes more cautious near zero performance where rare successes matter. Across nine validation settings in an illustrative 200-item, ten-epoch evaluation, it removed 57 to 97 percent of planned trials; realized savings depend on the benchmark and stopping target.",
    "why_it_matters": "A Bayesian stopping rule removed 57 to 97 percent of planned trials in illustrative evaluations while preserving the overall conclusion.",
    "limitations": [],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-17-007/the-benchmark-stopped-sampling-what-it-already-knew",
    "json_url": "https://themachinepress.com/story/mp-2026-08-17-007.json",
    "first_published_at": "2026-08-17T09:00:00.000-04:00",
    "modified_at": "2026-08-17T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-17-007-001",
        "text": "A Bayesian stopping rule removed 57 to 97 percent of planned trials in illustrative evaluations while preserving the overall conclusion.",
        "source_ids": [
          "source-2026-08-17-007"
        ],
        "qualification": null
      }
    ],
    "source_ids": [
      "source-2026-08-17-007"
    ],
    "tags": [
      "evaluation",
      "Bayesian inference",
      "adaptive sampling",
      "compute"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-17-007",
      "title": "arXiv preprint 2608.14425",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.14425",
      "canonical_url": "https://arxiv.org/abs/2608.14425",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-13T20:00:00.000-04:00",
      "accessed_at": "2026-08-17T08:24:13.830-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-17-007-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Benchmark Stopped Sampling What It Already Knew",
    "publisher": "The Machine Press",
    "published_at": "2026-08-17T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-17-007/the-benchmark-stopped-sampling-what-it-already-knew"
  }
}
