{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-31-009",
    "source_story_id": "tmp-story-geonext-video-geometry",
    "edition_id": "mp-2026-08-31-morning-0053",
    "edition_url": "https://themachinepress.com/edition/2026-08-31",
    "position": 9,
    "story_type": "dispatch",
    "section": "research",
    "editorial_classification": "editorial",
    "headline": "A Video Model Learned Depth as the Next Frame",
    "slug": "a-video-model-learned-depth-as-the-next-frame",
    "dek": "GeoNeXt reframes depth and surface-normal estimation as next-frame prediction inside a pretrained video generator.",
    "summary": "GeoNeXt reframes depth and surface-normal estimation as next-frame prediction inside a pretrained video generator.",
    "body_text": "The method adapts a video generative model to jointly represent images and geometry targets rather than training separate task-specific diffusion systems. Its authors report stronger zero-shot monocular depth and normal estimation than prior generative competitors with substantially less training data, and performance near discriminative systems trained on more than 100 times as much data. Those comparisons remain benchmark results from the proposing team.",
    "why_it_matters": "GeoNeXt reframes depth and surface-normal estimation as next-frame prediction inside a pretrained video generator.",
    "limitations": [
      "Those comparisons remain benchmark results from the proposing team."
    ],
    "importance": 7,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-31-009/a-video-model-learned-depth-as-the-next-frame",
    "json_url": "https://themachinepress.com/story/mp-2026-08-31-009.json",
    "first_published_at": "2026-08-31T09:00:00.000-04:00",
    "modified_at": "2026-08-31T09:00:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-31-009-001",
        "text": "GeoNeXt reframes depth and surface-normal estimation as next-frame prediction inside a pretrained video generator.",
        "source_ids": [
          "source-2026-08-31-009"
        ],
        "qualification": "Those comparisons remain benchmark results from the proposing team."
      }
    ],
    "source_ids": [
      "source-2026-08-31-009"
    ],
    "tags": [
      "computer vision",
      "video generation",
      "depth estimation"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-31-009",
      "title": "arXiv preprint 2608.28549",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.28549",
      "canonical_url": "https://arxiv.org/abs/2608.28549",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-28T13:25:31.000-04:00",
      "accessed_at": "2026-08-31T08:28:26.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-31-009-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "A Video Model Learned Depth as the Next Frame",
    "publisher": "The Machine Press",
    "published_at": "2026-08-31T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-31-009/a-video-model-learned-depth-as-the-next-frame"
  }
}
