{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-14-007",
    "source_story_id": "tmp-story-graph-theory-agent-representation-selector",
    "edition_id": "mp-2026-09-14-morning-0067",
    "edition_url": "https://themachinepress.com/edition/2026-09-14",
    "position": 7,
    "story_type": "dispatch",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "The Graph Solver Chose Its Own Representation",
    "slug": "the-graph-solver-chose-its-own-representation",
    "dek": "GT Bench spans 100,000 examples and four graph encodings; a selector-and-planner lifted Phi-4 from 33.0% to 41.5% on the hard split.",
    "summary": "GT Bench spans 100,000 examples and four graph encodings; a selector-and-planner lifted Phi-4 from 33.0% to 41.5% on the hard split.",
    "body_text": "Graph Theory Bench tests 24 classical graph problems in 44 task-structure settings and more than 100,000 examples represented as natural language, structured language, adjacency lists or adjacency matrices. Across eight language models, accuracy changed with graph density, size, topology and encoding; no single representation stayed best. The accompanying Graph Theory Agent chooses a representation, plans and decomposes around a frozen executor. On the reported benchmark it raised Phi-4 from 53.5% to 69.1% on the easy split and from 33.0% to 41.5% on the hard split, then transferred without retraining to two other graph benchmarks.",
    "why_it_matters": "GT Bench spans 100,000 examples and four graph encodings; a selector-and-planner lifted Phi-4 from 33.0% to 41.5% on the hard split.",
    "limitations": [],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-14-007/the-graph-solver-chose-its-own-representation",
    "json_url": "https://themachinepress.com/story/mp-2026-09-14-007.json",
    "first_published_at": "2026-09-14T09:00:00.000-04:00",
    "modified_at": "2026-09-14T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-14-007-001",
        "text": "GT Bench spans 100,000 examples and four graph encodings; a selector-and-planner lifted Phi-4 from 33.0% to 41.5% on the hard split.",
        "source_ids": [
          "source-2026-09-14-007"
        ],
        "qualification": null
      }
    ],
    "source_ids": [
      "source-2026-09-14-007"
    ],
    "tags": [
      "graph reasoning",
      "benchmarks",
      "representation selection"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-14-007",
      "title": "arXiv preprint 2609.12265",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.12265",
      "canonical_url": "https://arxiv.org/abs/2609.12265",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": null,
      "accessed_at": "2026-09-14T08:24:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-14-007-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Graph Solver Chose Its Own Representation",
    "publisher": "The Machine Press",
    "published_at": "2026-09-14T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-14-007/the-graph-solver-chose-its-own-representation"
  }
}
