{
  "schema_version": "1.0",
  "id": "s13:https://arxiv.org/abs/2607.21595v1",
  "slug": "2607-21595v1-0m5cxrk",
  "url": "https://feed7.dev/p/2607-21595v1-0m5cxrk",
  "title": "3D-Aware VLMs with Implicit and Explicit Geometries",
  "why_included": "VLM-IE3D adds implicit and reconstructed geometry tokens to an RGB-video VLM, offering an open approach for agents that must reason about spatial scenes without dedicated 3D input.",
  "summary": "VLM-IE3D derives **implicit geometry tokens** from video and **explicit geometry tokens** from reconstructed 3D attributes. A 3D-aware adapter combines both with ordinary 2D visual cues, while requiring only RGB video as input.",
  "practical_implication": "Builders working on scene-aware agents can test this pattern when 2D frames alone lose object position or spatial relationships. The released code and models make the architecture inspectable rather than merely conceptual.",
  "agent_context": "VLM-IE3D derives **implicit geometry tokens** from video and **explicit geometry tokens** from reconstructed 3D attributes. A 3D-aware adapter combines both with ordinary 2D visual cues, while requiring only RGB video as input.\n\nBuilders working on scene-aware agents can test this pattern when 2D frames alone lose object position or spatial relationships. The released code and models make the architecture inspectable rather than merely conceptual.\n\nThe material reports gains across detection, grounding, dense captioning, and spatial reasoning, but provides no scores or deployment costs. Its usefulness outside the evaluated 3D tasks remains open.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.21595v1",
    "published_at": "2026-07-23T17:59:59.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "model",
  "domains": [
    "image",
    "video"
  ],
  "topics": [
    "reasoning",
    "open-models"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The material reports gains across detection, grounding, dense captioning, and spatial reasoning, but provides no scores or deployment costs. Its usefulness outside the evaluated 3D tasks remains open."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-23T17:59:59.000Z",
  "modified_at": "2026-07-23T17:59:59.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-21595v1-0m5cxrk",
    "json": "https://feed7.dev/p/2607-21595v1-0m5cxrk.json",
    "markdown": "https://feed7.dev/p/2607-21595v1-0m5cxrk.md"
  }
}