{
  "schema_version": "1.1",
  "id": "archive:https://arxiv.org/abs/2608.07435v1",
  "slug": "2608-07435v1-0h6gzdk",
  "url": "https://feed7.dev/p/2608-07435v1-0h6gzdk",
  "title": "SABRE: Scalable and Automated Benchmarking of VLMs under Stress",
  "why_included": "SABRE turns a Markdown test design into generated VLM stress tests, then filters and repairs candidates. It offers a repeatable pattern for refreshing evals as models improve.",
  "summary": "SABRE converts a Markdown task design into specifications, images, and question-answer pairs, then applies model filtering and human review. SABRE-Prior includes **600 images** and **1,000 questions** testing whether models follow visual evidence over learned expectations.",
  "practical_implication": "Teams evaluating vision agents can encode test intent and schema first, generate candidates, discard easy cases, then reserve human effort for validity checks, annotation fixes, and localized image repair.",
  "agent_context": "SABRE converts a Markdown task design into specifications, images, and question-answer pairs, then applies model filtering and human review. SABRE-Prior includes **600 images** and **1,000 questions** testing whether models follow visual evidence over learned expectations.\n\nTeams evaluating vision agents can encode test intent and schema first, generate candidates, discard easy cases, then reserve human effort for validity checks, annotation fixes, and localized image repair.\n\nAcross **six VLMs**, macro-average accuracy ranged from **17.8% to 31.3%**. A real-image control was comparably difficult for the filtering model, so low scores cannot be attributed only to counterfactual generated imagery.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2608.07435v1",
    "published_at": "2026-08-07T17:21:04.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "image"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "generative-media"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Across **six VLMs**, macro-average accuracy ranged from **17.8% to 31.3%**. A real-image control was comparably difficult for the filtering model, so low scores cannot be attributed only to counterfactual generated imagery."
  ],
  "connected_context": {
    "meaning": "SABRE supplies a scalable construction pipeline for adversarial visual-evidence tests, combining specification-first generation, model-based difficulty filtering, and targeted human repair. Relative to the candidates, it makes expectation-versus-evidence conflict directly testable and reports broad failure across six VLMs, while its real-image control narrows the explanation for low scores beyond synthetic-image artifacts.",
    "corpus_size": 409,
    "generated_at": "2026-08-10T10:05:53.626Z",
    "connections": [
      {
        "title": "Evidence-Backed Video Question Answering",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.11862v1",
        "feed7_url": "https://feed7.dev/p/2607-11862v1-18as4nc",
        "reason": "E-VQA’s pixel-level evidence requirement offers a complementary way to diagnose whether SABRE failures reflect weak visual grounding rather than merely incorrect final answers."
      },
      {
        "title": "Evolution of Accuracy and Visual-Cognitive Errors in a Decade of Vision-Language AI Models",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.09654v1",
        "feed7_url": "https://feed7.dev/p/2607-09654v1-0b5dedg",
        "reason": "The decade study’s spatial-attention differences provide a candidate failure signal for the visual-cognitive errors that SABRE provokes when learned expectations conflict with image evidence."
      },
      {
        "title": "Building Closed-Loop Evals for a Multimodal Agent at Scale — Soumya Gupta & Jai Chopra, Uber",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=31GUkCBD-Uc",
        "feed7_url": "https://feed7.dev/p/building-closed-loop-evals-for-a-multimodal-agent-at-scale-soumya-gupta-1cqjbe2",
        "reason": "Uber’s golden-set gates, iterative QA, and production feedback reinforce SABRE’s implementation pattern of automated filtering followed by concentrated human validation, while applying it to a deployed image-editing agent."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-08-07T17:21:04.000Z",
  "modified_at": "2026-08-07T17:21:04.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2608-07435v1-0h6gzdk",
    "json": "https://feed7.dev/p/2608-07435v1-0h6gzdk.json",
    "markdown": "https://feed7.dev/p/2608-07435v1-0h6gzdk.md"
  }
}