{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-06-005",
    "source_story_id": "tmp-story-espo-prompt-optimization",
    "edition_id": "mp-2026-09-06-morning-0059",
    "edition_url": "https://themachinepress.com/edition/2026-09-06",
    "position": 5,
    "story_type": "dispatch",
    "section": "developer-tools",
    "editorial_classification": "editorial",
    "headline": "The Better Prompt Was Forty-Seven Percent Shorter",
    "slug": "the-better-prompt-was-forty-seven-percent-shorter",
    "dek": "ESPO clustered errors before proposing diverse fixes, then used bootstrap stability to keep prompt search from rewarding brittle gains.",
    "summary": "ESPO clustered errors before proposing diverse fixes, then used bootstrap stability to keep prompt search from rewarding brittle gains.",
    "body_text": "Across seven public NLP benchmarks, ESPO averaged 74.67 percent accuracy against 70.91 percent for GEPA while producing prompts of 1,004 rather than 1,878 characters. The method separates diagnosis, four proposal strategies and a stability-based selection stage. An ablation found that adding diversity without bootstrap selection reduced performance by 1.20 points. Cross-model gains were reported on four additional students, but the large per-task variation means the average should not be treated as a universal prompt-optimization guarantee.",
    "why_it_matters": "ESPO clustered errors before proposing diverse fixes, then used bootstrap stability to keep prompt search from rewarding brittle gains.",
    "limitations": [
      "Cross-model gains were reported on four additional students, but the large per-task variation means the average should not be treated as a universal prompt-optimization guarantee."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-06-005/the-better-prompt-was-forty-seven-percent-shorter",
    "json_url": "https://themachinepress.com/story/mp-2026-09-06-005.json",
    "first_published_at": "2026-09-06T09:00:00.000-04:00",
    "modified_at": "2026-09-06T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-06-005-001",
        "text": "ESPO clustered errors before proposing diverse fixes, then used bootstrap stability to keep prompt search from rewarding brittle gains.",
        "source_ids": [
          "source-2026-09-06-005"
        ],
        "qualification": "Cross-model gains were reported on four additional students, but the large per-task variation means the average should not be treated as a universal prompt-optimization guarantee."
      }
    ],
    "source_ids": [
      "source-2026-09-06-005"
    ],
    "tags": [
      "prompt optimization",
      "evaluation",
      "stability"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-06-005",
      "title": "arXiv preprint 2609.04197",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.04197",
      "canonical_url": "https://arxiv.org/abs/2609.04197",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-03T13:59:37.000-04:00",
      "accessed_at": "2026-09-06T08:20:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-06-005-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Better Prompt Was Forty-Seven Percent Shorter",
    "publisher": "The Machine Press",
    "published_at": "2026-09-06T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-06-005/the-better-prompt-was-forty-seven-percent-shorter"
  }
}
