{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-08-019",
    "source_story_id": "tmp-sidebar-harnessopt-bench",
    "edition_id": "mp-2026-08-08-morning-0030",
    "edition_url": "https://themachinepress.com/edition/2026-08-08",
    "position": 21,
    "story_type": "ticker",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "The Harness Became an Optimization Benchmark",
    "slug": "the-harness-became-an-optimization-benchmark",
    "dek": "Five frontier models edited prompts, tools, memory and orchestration under a metered evaluation budget.",
    "summary": "Five frontier models edited prompts, tools, memory and orchestration under a metered evaluation budget.",
    "body_text": "HarnessOpt-Bench scores normalized held-out gain while a trusted environment protects the test partition and preserves every candidate. Across four downstream tasks and 111 scored runs, optimizer models separated more than their coding harnesses, native harnesses were not consistently better, and gains varied sharply by task and starting point.",
    "why_it_matters": "Five frontier models edited prompts, tools, memory and orchestration under a metered evaluation budget.",
    "limitations": [
      "Across four downstream tasks and 111 scored runs, optimizer models separated more than their coding harnesses, native harnesses were not consistently better, and gains varied sharply by task and starting point."
    ],
    "importance": 7,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-08-019/the-harness-became-an-optimization-benchmark",
    "json_url": "https://themachinepress.com/story/mp-2026-08-08-019.json",
    "first_published_at": "2026-08-08T09:00:00.000-04:00",
    "modified_at": "2026-08-08T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-08-019-001",
        "text": "Five frontier models edited prompts, tools, memory and orchestration under a metered evaluation budget.",
        "source_ids": [
          "source-2026-08-08-021"
        ],
        "qualification": "Across four downstream tasks and 111 scored runs, optimizer models separated more than their coding harnesses, native harnesses were not consistently better, and gains varied sharply by task and starting point."
      }
    ],
    "source_ids": [
      "source-2026-08-08-021"
    ],
    "tags": [
      "agent harnesses",
      "evaluation",
      "optimization"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-08-021",
      "title": "arXiv preprint 2608.06301",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.06301",
      "canonical_url": "https://arxiv.org/abs/2608.06301",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": null,
      "accessed_at": "2026-08-08T08:24:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-08-019-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Harness Became an Optimization Benchmark",
    "publisher": "The Machine Press",
    "published_at": "2026-08-08T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-08-019/the-harness-became-an-optimization-benchmark"
  }
}
