{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-27-010",
    "source_story_id": "tmp-story-sciwalker-scientific-coding",
    "edition_id": "mp-2026-09-27-morning-0080",
    "edition_url": "https://themachinepress.com/edition/2026-09-27",
    "position": 10,
    "story_type": "dispatch",
    "section": "developer-tools",
    "editorial_classification": "editorial",
    "headline": "Execution Feedback Raised Scientific Coding Accuracy 9.9 Points",
    "slug": "execution-feedback-raised-scientific-coding-accuracy-9-9-points",
    "dek": "SciWalker organized 8,178 problems across five domains into operator graphs that could be tested while solving.",
    "summary": "SciWalker organized 8,178 problems across five domains into operator graphs that could be tested while solving.",
    "body_text": "SciWalker contains 8,178 scientific coding problems spanning five domains and 32 subdomains, with solutions represented as operator graphs. The framework uses execution feedback to revise intermediate programs rather than grading only a final answer. In the reported experiment, Qwen3.5-9B accuracy rose from 29.3% to 39.2%; that gain applies to this benchmark and training recipe, not scientific software in general.",
    "why_it_matters": "SciWalker organized 8,178 problems across five domains into operator graphs that could be tested while solving.",
    "limitations": [
      "The framework uses execution feedback to revise intermediate programs rather than grading only a final answer.",
      "In the reported experiment, Qwen3.5-9B accuracy rose from 29.3% to 39.2%; that gain applies to this benchmark and training recipe, not scientific software in general."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-27-010/execution-feedback-raised-scientific-coding-accuracy-9-9-points",
    "json_url": "https://themachinepress.com/story/mp-2026-09-27-010.json",
    "first_published_at": "2026-09-27T09:00:00.000-04:00",
    "modified_at": "2026-09-27T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-27-010-001",
        "text": "SciWalker organized 8,178 problems across five domains into operator graphs that could be tested while solving.",
        "source_ids": [
          "source-2026-09-27-010"
        ],
        "qualification": "The framework uses execution feedback to revise intermediate programs rather than grading only a final answer."
      }
    ],
    "source_ids": [
      "source-2026-09-27-010"
    ],
    "tags": [
      "scientific coding",
      "execution feedback",
      "program synthesis"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-27-010",
      "title": "arXiv preprint 2609.30054",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.30054",
      "canonical_url": "https://arxiv.org/abs/2609.30054",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-24T12:12:14.000-04:00",
      "accessed_at": "2026-09-27T08:26:41.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-27-010-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Execution Feedback Raised Scientific Coding Accuracy 9.9 Points",
    "publisher": "The Machine Press",
    "published_at": "2026-09-27T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-27-010/execution-feedback-raised-scientific-coding-accuracy-9-9-points"
  }
}
