{
  "schema_version": "1.1",
  "id": "archive:https://arxiv.org/abs/2608.06361v1",
  "slug": "2608-06361v1-1n3dr85",
  "url": "https://feed7.dev/p/2608-06361v1-1n3dr85",
  "title": "The Low Frequency Trap: Video Language Models Fail at Simple Event Bookkeeping",
  "why_included": "Controlled traces show video models can improve final counting scores without faithfully recovering events, so agents handling video need timestamp-level checks, not answer-only evals.",
  "summary": "The study profiles event counting across **2,190 controlled videos** with executable traces. Gemini 3.6 Flash reached 80% reliability for persistent transitions up to **12 events**, but had no reliable positive-count region for transient blinks.",
  "practical_implication": "For video agents, evaluate the reported event sequence against timestamps, not just the final count. In the high-count, high-frequency regime, only **0.2%** of counts were correct and **18.1%** of true events were recovered.",
  "agent_context": "The study profiles event counting across **2,190 controlled videos** with executable traces. Gemini 3.6 Flash reached 80% reliability for persistent transitions up to **12 events**, but had no reliable positive-count region for transient blinks.\n\nFor video agents, evaluate the reported event sequence against timestamps, not just the final count. In the high-count, high-frequency regime, only **0.2%** of counts were correct and **18.1%** of true events were recovered.\n\nMore frames raised Bounce Ball accuracy from 19.6% to 29.3%, while full sequence agreement remained 3.7%. Prompt changes also brought limited gains, so higher aggregate accuracy may conceal poor event bookkeeping.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2608.06361v1",
    "published_at": "2026-08-06T17:57:06.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "video"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "More frames raised Bounce Ball accuracy from 19.6% to 29.3%, while full sequence agreement remained 3.7%. Prompt changes also brought limited gains, so higher aggregate accuracy may conceal poor event bookkeeping."
  ],
  "connected_context": {
    "meaning": "This makes temporal bookkeeping a distinct video-agent capability that aggregate counting accuracy can conceal. It reinforces the candidates’ case for inspecting trajectories and partial progress, but supplies a stronger executable oracle: compare every reported event with timestamped ground truth. More frames and prompt changes offer limited relief, narrowing the value of simple inference-time adjustments for transient or dense events.",
    "corpus_size": 390,
    "generated_at": "2026-08-08T10:06:15.427Z",
    "connections": [
      {
        "title": "Rethinking Environments for Long-Horizon Work — Rayan Garg, Theta Software",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=2aS7aKoXn64",
        "feed7_url": "https://feed7.dev/p/rethinking-environments-for-long-horizon-work-rayan-garg-theta-software-11r7wbx",
        "reason": "The timestamped event sequence is a concrete form of the queryable trajectory this candidate calls for, and it reveals failures hidden by final-count grading."
      },
      {
        "title": "SocietyBench: Forecasting Counterfactual Social-World Evolution",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.04009v1",
        "feed7_url": "https://feed7.dev/p/2608-04009v1-05m5u8w",
        "reason": "Both separate temporal performance from a broader aggregate result, reinforcing event-level reporting when averages can hide sharply different reliability regimes."
      },
      {
        "title": "Teaching AI to Find Real Vulnerabilities — David Brumley, Bugcrowd",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=ZFxh7sqbUZo",
        "feed7_url": "https://feed7.dev/p/teaching-ai-to-find-real-vulnerabilities-david-brumley-bugcrowd-1ok0f7q",
        "reason": "Executable traces play the role of a deterministic oracle, while event recovery measures partial progress much as a capability ladder does when the final answer is wrong."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-08-06T17:57:06.000Z",
  "modified_at": "2026-08-06T17:57:06.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2608-06361v1-1n3dr85",
    "json": "https://feed7.dev/p/2608-06361v1-1n3dr85.json",
    "markdown": "https://feed7.dev/p/2608-06361v1-1n3dr85.md"
  }
}