{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-13-003",
    "source_story_id": "tmp-story-strong-weak-harnesses",
    "edition_id": "mp-2026-08-13-morning-0035",
    "edition_url": "https://themachinepress.com/edition/2026-08-13",
    "position": 3,
    "story_type": "dispatch",
    "section": "frontier-models",
    "editorial_classification": "editorial",
    "headline": "The Small Model Borrowed a Better Skeleton",
    "slug": "the-small-model-borrowed-a-better-skeleton",
    "dek": "Builder models nearly doubled weaker models' average benchmark scores by writing deterministic test-time harnesses instead of changing their weights.",
    "summary": "Builder models nearly doubled weaker models' average benchmark scores by writing deterministic test-time harnesses instead of changing their weights.",
    "body_text": "On four theory-of-mind benchmarks, stronger builder models used five percent of each dataset to iteratively design an inference harness for a weaker target model. Average target performance rose from 0.49 to 0.91 without parameter updates. The largest gains came from moving brittle reasoning into deterministic code, routing cases and enforcing answer formats rather than simply eliciting more chain-of-thought. The result is benchmark-specific scaffolding, not a general transfer of every capability.",
    "why_it_matters": "Builder models nearly doubled weaker models' average benchmark scores by writing deterministic test-time harnesses instead of changing their weights.",
    "limitations": [
      "The result is benchmark-specific scaffolding, not a general transfer of every capability."
    ],
    "importance": 9,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-13-003/the-small-model-borrowed-a-better-skeleton",
    "json_url": "https://themachinepress.com/story/mp-2026-08-13-003.json",
    "first_published_at": "2026-08-13T09:00:00.000-04:00",
    "modified_at": "2026-08-13T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-13-003-001",
        "text": "Builder models nearly doubled weaker models' average benchmark scores by writing deterministic test-time harnesses instead of changing their weights.",
        "source_ids": [
          "source-2026-08-13-003"
        ],
        "qualification": "The result is benchmark-specific scaffolding, not a general transfer of every capability."
      }
    ],
    "source_ids": [
      "source-2026-08-13-003"
    ],
    "tags": [
      "language models",
      "inference harnesses",
      "distillation",
      "evaluation"
    ],
    "image_url": "https://themachinepress.com/issues/2026-08-13/harness-code-file-image.webp",
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-13-003",
      "title": "arXiv preprint 2608.12307",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.12307",
      "canonical_url": "https://arxiv.org/abs/2608.12307",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-12T13:53:18.000-04:00",
      "accessed_at": "2026-08-13T08:18:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-13-003-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Small Model Borrowed a Better Skeleton",
    "publisher": "The Machine Press",
    "published_at": "2026-08-13T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-13-003/the-small-model-borrowed-a-better-skeleton"
  }
}
