{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-09-09-009",
    "source_story_id": "tmp-story-normviz-global-visual-norms",
    "edition_id": "mp-2026-09-09-morning-0062",
    "edition_url": "https://themachinepress.com/edition/2026-09-09",
    "position": 9,
    "story_type": "dispatch",
    "section": "consumer-products",
    "editorial_classification": "editorial",
    "headline": "The Best Vision Model Read Fewer Than Three in Ten Cultural Pairs",
    "slug": "the-best-vision-model-read-fewer-than-three-in-ten-cultural-pairs",
    "dek": "NormViz-Bench changed one culturally relevant behavior at a time across 6,536 images from 16 countries.",
    "summary": "NormViz-Bench changed one culturally relevant behavior at a time across 6,536 images from 16 countries.",
    "body_text": "NormViz-Bench contains 3,268 human-validated contrastive image pairs spanning 16 countries. Each pair differs only in a behavior that changes whether the scene conforms to, violates or is irrelevant to a local social norm, and pair-level scoring requires both images to be correct. Gemini 3.0 Flash and Qwen2.5-VL-7B reached 26.6 and 21.6 percent pair accuracy, respectively. Fine-tuning smaller Qwen3-VL models on 64,000 explanation-linked images raised relative accuracy, but absolute performance remained below 30 percent. The benchmark measures its curated countries, behaviors and labels; it does not reduce culture to a single universal rulebook.",
    "why_it_matters": "NormViz-Bench changed one culturally relevant behavior at a time across 6,536 images from 16 countries.",
    "limitations": [
      "Each pair differs only in a behavior that changes whether the scene conforms to, violates or is irrelevant to a local social norm, and pair-level scoring requires both images to be correct.",
      "The benchmark measures its curated countries, behaviors and labels; it does not reduce culture to a single universal rulebook."
    ],
    "importance": 9,
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-09-009/the-best-vision-model-read-fewer-than-three-in-ten-cultural-pairs",
    "json_url": "https://themachinepress.com/story/mp-2026-09-09-009.json",
    "first_published_at": "2026-09-09T09:00:00.000-04:00",
    "modified_at": "2026-09-09T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-09-09-009-001",
        "text": "NormViz-Bench changed one culturally relevant behavior at a time across 6,536 images from 16 countries.",
        "source_ids": [
          "source-2026-09-09-009"
        ],
        "qualification": "Each pair differs only in a behavior that changes whether the scene conforms to, violates or is irrelevant to a local social norm, and pair-level scoring requires both images to be correct."
      }
    ],
    "source_ids": [
      "source-2026-09-09-009"
    ],
    "tags": [
      "multimodal AI",
      "culture",
      "benchmarking"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-09-09-009",
      "title": "arXiv preprint 2609.06831",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2609.06831",
      "canonical_url": "https://arxiv.org/abs/2609.06831",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-09-06T17:02:47.000-04:00",
      "accessed_at": "2026-09-09T08:30:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-09-09-009-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "The Best Vision Model Read Fewer Than Three in Ten Cultural Pairs",
    "publisher": "The Machine Press",
    "published_at": "2026-09-09T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-09-09-009/the-best-vision-model-read-fewer-than-three-in-ten-cultural-pairs"
  }
}
