{
  "schema_version": "1.1",
  "id": "archive:https://arxiv.org/abs/2607.24707v1",
  "slug": "2607-24707v1-1vjhtqg",
  "url": "https://feed7.dev/p/2607-24707v1-1vjhtqg",
  "title": "ERUnderstand: Evaluating Vision-Language Models on Structured ER Diagrams",
  "why_included": "ERUnderstand shows vision-language models can recover common ERD elements but often miss rarer schema constructs, so image-to-schema agent workflows still need structural validation.",
  "summary": "ERUnderstand pairs **2,960 diagrams** with standardized machine-readable schemas. Tested vision-language models recovered common elements above **0.74 F1**, but reached only 0.14 on multivalued attributes and 0.07 on N-ary relationships.",
  "practical_implication": "Builders using agents to turn ERD screenshots into migrations or models should validate uncommon constructs explicitly. Reasoning-augmented models improved overall performance by **15–25%**, but did not remove the structural weak spots.",
  "agent_context": "ERUnderstand pairs **2,960 diagrams** with standardized machine-readable schemas. Tested vision-language models recovered common elements above **0.74 F1**, but reached only 0.14 on multivalued attributes and 0.07 on N-ary relationships.\n\nBuilders using agents to turn ERD screenshots into migrations or models should validate uncommon constructs explicitly. Reasoning-augmented models improved overall performance by **15–25%**, but did not remove the structural weak spots.\n\nThe dataset mixes educational, real-world, and synthetic diagrams, so aggregate scores may not match a specific team's notation. Models also remained sensitive to linguistic priors and increasing complexity; the material does not report end-to-end schema-generation accuracy.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.24707v1",
    "published_at": "2026-07-27T17:46:43.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "image",
    "data"
  ],
  "topics": [
    "agent-evals",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The dataset mixes educational, real-world, and synthetic diagrams, so aggregate scores may not match a specific team's notation. Models also remained sensitive to linguistic priors and increasing complexity; the material does not report end-to-end schema-generation accuracy."
  ],
  "connected_context": null,
  "lifecycle": "Current",
  "published_at": "2026-07-27T17:46:43.000Z",
  "modified_at": "2026-07-27T17:46:43.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-24707v1-1vjhtqg",
    "json": "https://feed7.dev/p/2607-24707v1-1vjhtqg.json",
    "markdown": "https://feed7.dev/p/2607-24707v1-1vjhtqg.md"
  }
}