{
  "schema_version": "1.0",
  "id": "s13:https://arxiv.org/abs/2607.21552v1",
  "slug": "2607-21552v1-1v5rc1p",
  "url": "https://feed7.dev/p/2607-21552v1-1v5rc1p",
  "title": "MIRROR: Learning from the Other View for Multi-Modal Reasoning",
  "why_included": "MIRROR trains a VLM across text, diagram, and combined views by letting its strongest view supervise weaker ones, targeting the modality inconsistency that single-view evals hide.",
  "summary": "ODA-Data pairs geometry problems across **text-dominant**, **image-dominant**, and **combined image+text** views. MIRROR evaluates every view, selects the best-performing one as teacher, then aligns the others toward it with a reverse-KL objective.",
  "practical_implication": "Builders of multimodal agents should test semantically equivalent inputs in each supported modality. Divergent answers expose failures that aggregate accuracy hides, while reciprocal supervision offers a way to reuse the model's strongest representation.",
  "agent_context": "ODA-Data pairs geometry problems across **text-dominant**, **image-dominant**, and **combined image+text** views. MIRROR evaluates every view, selects the best-performing one as teacher, then aligns the others toward it with a reverse-KL objective.\n\nBuilders of multimodal agents should test semantically equivalent inputs in each supported modality. Divergent answers expose failures that aggregate accuracy hides, while reciprocal supervision offers a way to reuse the model's strongest representation.\n\nThe reported gains cover geometry reasoning benchmarks and comparisons with standard RL, but the supplied material includes no scores. It remains unclear whether the method transfers to noisier screenshots, documents, or mixed-modal coding tasks.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.21552v1",
    "published_at": "2026-07-23T17:35:56.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "model",
  "domains": [
    "image",
    "research"
  ],
  "topics": [
    "reasoning"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The reported gains cover geometry reasoning benchmarks and comparisons with standard RL, but the supplied material includes no scores. It remains unclear whether the method transfers to noisier screenshots, documents, or mixed-modal coding tasks."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-23T17:35:56.000Z",
  "modified_at": "2026-07-23T17:35:56.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-21552v1-1v5rc1p",
    "json": "https://feed7.dev/p/2607-21552v1-1v5rc1p.json",
    "markdown": "https://feed7.dev/p/2607-21552v1-1v5rc1p.md"
  }
}