{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.22008v1",
  "slug": "2609-22008v1-0w8qm00",
  "url": "https://feed7.dev/p/2609-22008v1-0w8qm00",
  "title": "DiaVLo: Diagnosing Behaviours of Vision-Language Models",
  "why_included": "DiaVLo supplements VLM scores with behavior specifications and causal concept estimates, helping builders see how a model reaches aligned or misaligned outputs.",
  "summary": "DiaVLo uses human curation and model generation to specify desired and observed VLM behavior, then surfaces mismatches. It also estimates which concepts most influence behavior and was evaluated on **several open-source VLMs** in **classification and generation** settings.",
  "practical_implication": "For vision-model selection or evaluation, pair aggregate performance with behavioral labels and influential-concept analysis. This can expose how a model perceives, organizes, or prioritizes concepts when a score alone hides the failure mode.",
  "agent_context": "DiaVLo uses human curation and model generation to specify desired and observed VLM behavior, then surfaces mismatches. It also estimates which concepts most influence behavior and was evaluated on **several open-source VLMs** in **classification and generation** settings.\n\nFor vision-model selection or evaluation, pair aggregate performance with behavioral labels and influential-concept analysis. This can expose how a model perceives, organizes, or prioritizes concepts when a score alone hides the failure mode.\n\nThe abstract says behavior labels correlate with performance but provides no effect sizes, model names, or benchmark counts. Human curation also means the usefulness of a diagnosis may depend on the quality of the behavioral specification.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.22008v1",
    "published_at": "2026-09-18T17:06:44.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "image"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "model-selection"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The abstract says behavior labels correlate with performance but provides no effect sizes, model names, or benchmark counts. Human curation also means the usefulness of a diagnosis may depend on the quality of the behavioral specification."
  ],
  "connected_context": {
    "meaning": "DiaVLo adds a diagnostic layer that the prior benchmark-integrity work largely lacks: after controlling task and harness quality, evaluators can characterize how a VLM organizes concepts and why its behavior diverges from the specification. It complements rather than replaces leakage controls, realistic inputs, stress tests, and aggregate scores, and its reliance on curated behavior labels makes specification quality part of the evaluation.",
    "corpus_size": 831,
    "generated_at": "2026-09-21T09:04:18.336Z",
    "connections": [
      {
        "title": "SABRE: Scalable and Automated Benchmarking of VLMs under Stress",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.07435v1",
        "feed7_url": "https://feed7.dev/p/2608-07435v1-0h6gzdk",
        "reason": "SABRE generates specification-led stress cases, while DiaVLo can characterize the behavioral and conceptual mismatch those cases expose; together they connect test construction with failure diagnosis."
      },
      {
        "title": "Can Edge-Deployable Vision-Language Models Identify Species?",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.11916v1",
        "feed7_url": "https://feed7.dev/p/2609-11916v1-1xcd3qe",
        "reason": "The species study shows deployment-image and open-set failures that aggregate clean-photo results miss; DiaVLo offers a way to label such behavioral differences and inspect the concepts influencing them."
      },
      {
        "title": "Stop Evaluating Models Like It's the 50s - Alejandro Vidal, Mindmakers",
        "source_name": "YouTube",
        "source_url": "https://www.youtube.com/watch?v=O3FEoMYvUf8",
        "feed7_url": "https://feed7.dev/p/stop-evaluating-models-like-it-s-the-50s-alejandro-vidal-mindmakers-0spx928",
        "reason": "Item response theory diagnoses weak or informative questions at the suite level, whereas DiaVLo diagnoses model behavior and influential concepts, making the methods complementary rather than interchangeable."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-18T17:06:44.000Z",
  "modified_at": "2026-09-18T17:06:44.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-22008v1-0w8qm00",
    "json": "https://feed7.dev/p/2609-22008v1-0w8qm00.json",
    "markdown": "https://feed7.dev/p/2609-22008v1-0w8qm00.md"
  }
}