{
  "schema_version": "1.0",
  "id": "s13:https://arxiv.org/abs/2607.21550v1",
  "slug": "2607-21550v1-1j7d28n",
  "url": "https://feed7.dev/p/2607-21550v1-1j7d28n",
  "title": "X$^3$-OPD: Distilling Reasoning into Large Audio-Language Models via On-Policy Alignment",
  "why_included": "X³-OPD transfers a text model’s reasoning into an audio-language model while grounding training in the student’s own acoustic interpretations, including events, prosody, and dialogue.",
  "summary": "**X³-OPD** is an on-policy distillation framework: an audio-language student generates reasoning from acoustic input, while a text teacher supplies token-level guidance from matched transcripts and verified answers.",
  "practical_implication": "The **three-tier corpus** spans speech-rendered text problems, complex audio-event reasoning, and spoken dialogue with paralinguistic cues. Builders working on voice or audio agents should treat non-verbal sound and prosody as reasoning inputs, not merely transcription problems.",
  "agent_context": "**X³-OPD** is an on-policy distillation framework: an audio-language student generates reasoning from acoustic input, while a text teacher supplies token-level guidance from matched transcripts and verified answers.\n\nThe **three-tier corpus** spans speech-rendered text problems, complex audio-event reasoning, and spoken dialogue with paralinguistic cues. Builders working on voice or audio agents should treat non-verbal sound and prosody as reasoning inputs, not merely transcription problems.\n\nResults across **MMSU, MMAU, BIG Bench Audio, and MMAR** indicate better audio-grounded reasoning and chain-of-thought quality with existing abilities largely preserved under domain shift. The abstract provides no effect sizes, implementation details, or released artifacts.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.21550v1",
    "published_at": "2026-07-23T17:35:20.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "model",
  "domains": [
    "audio"
  ],
  "topics": [
    "reasoning"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Results across **MMSU, MMAU, BIG Bench Audio, and MMAR** indicate better audio-grounded reasoning and chain-of-thought quality with existing abilities largely preserved under domain shift. The abstract provides no effect sizes, implementation details, or released artifacts."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-23T17:35:20.000Z",
  "modified_at": "2026-07-23T17:35:20.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-21550v1-1j7d28n",
    "json": "https://feed7.dev/p/2607-21550v1-1j7d28n.json",
    "markdown": "https://feed7.dev/p/2607-21550v1-1j7d28n.md"
  }
}