{
  "$schema": "https://themachinepress.com/schemas/story-v1.schema.json",
  "schema_version": "1.0.0",
  "document_type": "machine_press_story",
  "story": {
    "story_id": "mp-2026-08-10-006",
    "source_story_id": "tmp-story-sembridge-speech",
    "edition_id": "mp-2026-08-10-morning-0032",
    "edition_url": "https://themachinepress.com/edition/2026-08-10",
    "position": 6,
    "story_type": "dispatch",
    "section": "research",
    "editorial_classification": "editorial",
    "headline": "Speech Kept Its Meaning Without Discrete Tokens",
    "slug": "speech-kept-its-meaning-without-discrete-tokens",
    "dek": "SemBridge adds semantic-token supervision during training while leaving continuous-latent speech generation unchanged at inference.",
    "summary": "SemBridge adds semantic-token supervision during training while leaving continuous-latent speech generation unchanged at inference.",
    "body_text": "Continuous speech models preserve acoustic detail but make linguistic structure less explicit. SemBridge anchors hidden states and the acoustic latent space to discrete semantic tokens during training, then removes that supervision path at inference. Across zero-shot text-to-speech and score-conditioned singing tests, the authors report lower word and character error rates while maintaining competitive speaker similarity and perceptual quality. The claims remain benchmark results from a preprint.",
    "why_it_matters": "SemBridge adds semantic-token supervision during training while leaving continuous-latent speech generation unchanged at inference.",
    "limitations": [
      "The claims remain benchmark results from a preprint."
    ],
    "importance": 7,
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-10-006/speech-kept-its-meaning-without-discrete-tokens",
    "json_url": "https://themachinepress.com/story/mp-2026-08-10-006.json",
    "first_published_at": "2026-08-10T09:00:00.000-04:00",
    "modified_at": "2026-08-10T15:50:00.000-04:00",
    "content_status": "new",
    "is_carryover": false,
    "carryover_reason": null,
    "key_claims": [
      {
        "claim_id": "claim-mp-2026-08-10-006-001",
        "text": "SemBridge adds semantic-token supervision during training while leaving continuous-latent speech generation unchanged at inference.",
        "source_ids": [
          "source-2026-08-10-006"
        ],
        "qualification": "The claims remain benchmark results from a preprint."
      }
    ],
    "source_ids": [
      "source-2026-08-10-006"
    ],
    "tags": [
      "speech generation",
      "text to speech",
      "continuous latents"
    ],
    "image_url": null,
    "corrections": []
  },
  "sources": [
    {
      "source_id": "source-2026-08-10-006",
      "title": "arXiv preprint 2608.07462",
      "publisher": "arXiv",
      "url": "https://arxiv.org/abs/2608.07462",
      "canonical_url": "https://arxiv.org/abs/2608.07462",
      "source_type": "primary_research",
      "is_primary_source": true,
      "published_at": "2026-08-07T13:00:00.000-04:00",
      "accessed_at": "2026-08-10T15:30:00.000-04:00",
      "supports_claim_ids": [
        "claim-mp-2026-08-10-006-001"
      ]
    }
  ],
  "corrections": [],
  "publisher": {
    "name": "The Machine Press",
    "url": "https://themachinepress.com",
    "description": "A daily newspaper for the age of artificial intelligence."
  },
  "cite_this_report": {
    "title": "Speech Kept Its Meaning Without Discrete Tokens",
    "publisher": "The Machine Press",
    "published_at": "2026-08-10T09:00:00.000-04:00",
    "canonical_url": "https://themachinepress.com/story/mp-2026-08-10-006/speech-kept-its-meaning-without-discrete-tokens"
  }
}
