{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-31-012",
    "source_story_id": "tmp-story-linguistic-confidence-divergence",
    "edition_id": "mp-2026-08-31-morning-0053",
    "edition_url": "https://themachinepress.com/edition/2026-08-31",
    "position": 12,
    "story_type": "dispatch",
    "section": "benchmarks-evals",
    "editorial_classification": "editorial",
    "headline": "Saying Confident Did Not Mean Being Calibrated",
    "slug": "saying-confident-did-not-mean-being-calibrated",
    "dek": "Across 30 models, verbal confidence frequently diverged from logits or semantic-entropy uncertainty.",
    "summary": "Across 30 models, verbal confidence frequently diverged from logits or semantic-entropy uncertainty.",
    "body_text": "The study compares linguistic confidence with internal signals across eight classification tasks and two generation tasks. Association was weak on average, instruction tuning often raised reported confidence while worsening calibration, and attitude cues inflated scores without improving alignment. Score exemplars sometimes preserved rank ordering, but the authors conclude that verbal confidence needs multi-axis evaluation before it enters reliability pipelines.",
    "why_it_matters": "Across 30 models, verbal confidence frequently diverged from logits or semantic-entropy uncertainty.",
    "limitations": [],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-31-012/saying-confident-did-not-mean-being-calibrated",
    "json_url": "https://themachinepress.com/story/mp-2026-08-31-012.json",
    "first_published_at": "2026-08-31T09:00:00.000-04:00",
    "modified_at": "2026-08-31T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-31-012-001",
        "text": "Across 30 models, verbal confidence frequently diverged from logits or semantic-entropy uncertainty.",
        "source_ids": [
          "source-2026-08-31-012"
        ],
        "qualification": null
      }
    ],
    "source_ids": [
      "source-2026-08-31-012"
    ],
    "tags": [
      "confidence",
      "calibration",
      "language models"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-31-012",
      "title": "arXiv preprint 2608.28382",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.28382",
      "canonical_url": "https://arxiv.org/abs/2608.28382",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-28T10:37:31.000-04:00",
      "accessed_at": "2026-08-31T08:28:26.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-31-012-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Saying Confident Did Not Mean Being Calibrated",
    "publisher": "The Machine Press",
    "published_at": "2026-08-31T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-31-012/saying-confident-did-not-mean-being-calibrated"
  }
}
