{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-30-010",
    "source_story_id": "tmp-story-agentjudgebench",
    "edition_id": "mp-2026-08-30-morning-0052",
    "edition_url": "https://themachinepress.com/edition/2026-08-30",
    "position": 10,
    "story_type": "dispatch",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "Harder Workflows Flattened Every Judge",
    "slug": "harder-workflows-flattened-every-judge",
    "dek": "AgentJudgeBench tests language-model judges on 3,808 dependency-ordered tool-calling workflows.",
    "summary": "AgentJudgeBench tests language-model judges on 3,808 dependency-ordered tool-calling workflows.",
    "body_text": "Across six workflow graph shapes and three difficulty levels, alignment with the programmatic reference declined as tasks grew harder and fell faster when ground truth was hidden. On hard no-ground-truth cases, six judges converged in a narrow 77-to-82-percent band despite scale differences. Structured rubrics improved alignment by as much as 6.5 points, while reasoning traces and temperature changes had little effect. Ground truth sometimes reduced alignment through apparent over-anchoring.",
    "why_it_matters": "AgentJudgeBench tests language-model judges on 3,808 dependency-ordered tool-calling workflows.",
    "limitations": [],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-30-010/harder-workflows-flattened-every-judge",
    "json_url": "https://themachinepress.com/story/mp-2026-08-30-010.json",
    "first_published_at": "2026-08-30T09:00:00.000-04:00",
    "modified_at": "2026-08-30T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-30-010-001",
        "text": "AgentJudgeBench tests language-model judges on 3,808 dependency-ordered tool-calling workflows.",
        "source_ids": [
          "source-2026-08-30-010"
        ],
        "qualification": null
      }
    ],
    "source_ids": [
      "source-2026-08-30-010"
    ],
    "tags": [
      "LLM judges",
      "tool calling",
      "agent evaluation"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-30-010",
      "title": "arXiv preprint 2608.26623",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.26623",
      "canonical_url": "https://arxiv.org/abs/2608.26623",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-27T01:20:22.000-04:00",
      "accessed_at": "2026-08-30T08:23:14.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-30-010-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Harder Workflows Flattened Every Judge",
    "publisher": "The Machine Press",
    "published_at": "2026-08-30T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-30-010/harder-workflows-flattened-every-judge"
  }
}
