{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2607.29602v1",
  "slug": "2607-29602v1-02qoodp",
  "url": "https://feed7.dev/p/2607-29602v1-02qoodp",
  "title": "FriendBench: Benchmarking Dyadic Familiarity Inference in Humans and Multimodal Large Language Models",
  "why_included": "FriendBench shows why aggregate accuracy can hide behavioral bias: top multimodal models matched human panels overall but favored the “stranger” answer and gained less from video.",
  "summary": "FriendBench tests whether a pair is familiar or meeting for the first time using **20-second clips** from **96 balanced dyads**. It compares **26 models from seven companies** with matched human panels across text, audio, and video.",
  "practical_implication": "Builders evaluating socially aware multimodal systems should inspect class balance and channel-specific gains, not just headline accuracy. The strongest models matched the human crowd statistically while leaning toward “stranger,” indicating a different effective prior.",
  "agent_context": "FriendBench tests whether a pair is familiar or meeting for the first time using **20-second clips** from **96 balanced dyads**. It compares **26 models from seven companies** with matched human panels across text, audio, and video.\n\nBuilders evaluating socially aware multimodal systems should inspect class balance and channel-specific gains, not just headline accuracy. The strongest models matched the human crowd statistically while leaning toward “stranger,” indicating a different effective prior.\n\nOnly humans benefited from visible behavior beyond speech. The benchmark covers one constrained ice-breaker setup, so its findings do not establish how models handle broader social contexts.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.29602v1",
    "published_at": "2026-07-31T16:33:39.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "video",
    "audio"
  ],
  "topics": [
    "benchmark-integrity"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Only humans benefited from visible behavior beyond speech. The benchmark covers one constrained ice-breaker setup, so its findings do not establish how models handle broader social contexts."
  ],
  "connected_context": {
    "meaning": "FriendBench adds a controlled test of whether multimodal models infer a social relationship, while showing that human-level aggregate accuracy can conceal a different class prior and no measurable benefit from visible behavior. This reinforces the need to inspect modality contribution and error balance, not just headline scores, but narrows the conclusion to short, balanced ice-breaker interactions rather than general social understanding.",
    "corpus_size": 330,
    "generated_at": "2026-08-03T10:05:14.829Z",
    "connections": [
      {
        "title": "Evolution of Accuracy and Visual-Cognitive Errors in a Decade of Vision-Language AI Models",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.09654v1",
        "feed7_url": "https://feed7.dev/p/2607-09654v1-0b5dedg",
        "reason": "Both show that near-human aggregate vision-language performance can coexist with different perceptual behavior, making error patterns and modality use important evaluation signals."
      },
      {
        "title": "Evidence-Backed Video Question Answering",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.11862v1",
        "feed7_url": "https://feed7.dev/p/2607-11862v1-18as4nc",
        "reason": "FriendBench finds that models did not gain from visible behavior; evidence-backed video QA offers a way to test whether predictions are actually grounded in tracked visual evidence rather than speech alone."
      },
      {
        "title": "Evaling Video Slop — Maor Bril, Character.ai",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=b_PmGocP4rc",
        "feed7_url": "https://feed7.dev/p/evaling-video-slop-maor-bril-character-ai-0cd76sd",
        "reason": "Both caution that video evaluation must measure information across time and modalities rather than treating visually plausible output or headline accuracy as sufficient."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-07-31T16:33:39.000Z",
  "modified_at": "2026-07-31T16:33:39.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-29602v1-02qoodp",
    "json": "https://feed7.dev/p/2607-29602v1-02qoodp.json",
    "markdown": "https://feed7.dev/p/2607-29602v1-02qoodp.md"
  }
}