{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-08-003",
    "source_story_id": "tmp-story-selective-context-trust",
    "edition_id": "mp-2026-08-08-morning-0030",
    "edition_url": "https://themachinepress.com/edition/2026-08-08",
    "position": 3,
    "story_type": "dispatch",
    "section": "safety",
    "editorial_classification": "editorial",
    "headline": "Robustness Failed When the Model Ignored Good Advice",
    "slug": "robustness-failed-when-the-model-ignored-good-advice",
    "dek": "MIST tests clean, misleading, correct and irrelevant context together so resistance cannot masquerade as selective judgment.",
    "summary": "MIST tests clean, misleading, correct and irrelevant context together so resistance cannot masquerade as selective judgment.",
    "body_text": "The MIST benchmark renders each reasoning item under four matched context conditions and measures how often misleading context flips an otherwise correct answer. The authors say susceptibility appeared across the open models they tested; their SCOPE training method reduced those flips while preserving accuracy when context was correct, clean or irrelevant. The preprint argues for evaluating selective trust rather than blanket resistance, but does not establish immunity to adversarial context in deployed systems.",
    "why_it_matters": "MIST tests clean, misleading, correct and irrelevant context together so resistance cannot masquerade as selective judgment.",
    "limitations": [
      "The preprint argues for evaluating selective trust rather than blanket resistance, but does not establish immunity to adversarial context in deployed systems."
    ],
    "importance": 8,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-08-003/robustness-failed-when-the-model-ignored-good-advice",
    "json_url": "https://themachinepress.com/story/mp-2026-08-08-003.json",
    "first_published_at": "2026-08-08T09:00:00.000-04:00",
    "modified_at": "2026-08-08T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-08-003-001",
        "text": "MIST tests clean, misleading, correct and irrelevant context together so resistance cannot masquerade as selective judgment.",
        "source_ids": [
          "source-2026-08-08-003"
        ],
        "qualification": "The preprint argues for evaluating selective trust rather than blanket resistance, but does not establish immunity to adversarial context in deployed systems."
      }
    ],
    "source_ids": [
      "source-2026-08-08-003"
    ],
    "tags": [
      "language models",
      "context reliability",
      "benchmarks",
      "alignment"
    ],
    "image_url": "https://themachinepress.com/issues/2026-08-08/selective-context-file-image.webp",
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-08-003",
      "title": "arXiv preprint 2608.06377",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.06377",
      "canonical_url": "https://arxiv.org/abs/2608.06377",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": null,
      "accessed_at": "2026-08-08T08:24:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-08-003-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Robustness Failed When the Model Ignored Good Advice",
    "publisher": "The Machine Press",
    "published_at": "2026-08-08T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-08-003/robustness-failed-when-the-model-ignored-good-advice"
  }
}
