{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2608.04009v1",
  "slug": "2608-04009v1-05m5u8w",
  "url": "https://feed7.dev/p/2608-04009v1-05m5u8w",
  "title": "SocietyBench: Forecasting Counterfactual Social-World Evolution",
  "why_included": "SocietyBench tests forecasting in anonymized social timelines, exposing gaps that task-completion evals miss and showing agent frameworks did not improve the shared base model.",
  "summary": "SocietyBench converts news and posts from **five platforms** into anonymized, date-shifted timelines. It scores **125 prediction points** across five events on probability calibration and temporal accuracy.",
  "practical_implication": "Builders evaluating research or monitoring agents should score confidence and timing separately, anonymize recognizable events, and test across several event types. The released timelines, questions, ground truth, and scoring code make that setup reproducible.",
  "agent_context": "SocietyBench converts news and posts from **five platforms** into anonymized, date-shifted timelines. It scores **125 prediction points** across five events on probability calibration and temporal accuracy.\n\nBuilders evaluating research or monitoring agents should score confidence and timing separately, anonymize recognizable events, and test across several event types. The released timelines, questions, ground truth, and scoring code make that setup reproducible.\n\nThe strongest of six frontier models reached **75.0/100**, while three agent frameworks failed to beat their shared base model. This is a small event set, and per-event gaps of **21.4 points** warn against broad conclusions from one scenario.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2608.04009v1",
    "published_at": "2026-08-04T17:59:56.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "research"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The strongest of six frontier models reached **75.0/100**, while three agent frameworks failed to beat their shared base model. This is a small event set, and per-event gaps of **21.4 points** warn against broad conclusions from one scenario."
  ],
  "connected_context": {
    "meaning": "This turns social forecasting into a reproducible agent evaluation that separates confidence calibration from temporal accuracy and reduces recognition through anonymized, shifted timelines. It also narrows claims about agent scaffolding: on this small, variable event set, three frameworks did not improve their common base model, so results should be reported per event rather than generalized from the aggregate.",
    "corpus_size": 353,
    "generated_at": "2026-08-05T10:06:09.805Z",
    "connections": [
      {
        "title": "onepot-Bench 0: towards lab-aware in silico chemistry benchmarks",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.02595v1",
        "feed7_url": "https://feed7.dev/p/2608-02595v1-0l7cc8k",
        "reason": "Both reduce contamination through less recognizable evidence, but SocietyBench releases its transformed timelines and scoring code, contrasting with onepot-Bench’s use of private data and its resulting reproducibility tradeoff."
      },
      {
        "title": "Vending-Bench: Long-Horizon Agent Evals — Lukas Petersson, Andon Labs",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=cO8qC6HBuBg",
        "feed7_url": "https://feed7.dev/p/vending-bench-long-horizon-agent-evals-lukas-petersson-andon-labs-0fu78nz",
        "reason": "Vending-Bench’s warning that long-horizon behavior varies between simulations and reality reinforces SocietyBench’s use of multiple evolving events and its caution against generalizing from one scenario."
      },
      {
        "title": "Demystifying evals for AI agents",
        "source_name": "Anthropic",
        "source_url": "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents",
        "feed7_url": "https://feed7.dev/p/demystifying-evals-for-ai-agents-1kh2tdz",
        "reason": "SocietyBench makes the general eval guidance more specific for forecasting: confidence and timing require separate graders, and large per-event variation argues for expanding the task set before using aggregate scores for model or framework decisions."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-08-04T17:59:56.000Z",
  "modified_at": "2026-08-04T17:59:56.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2608-04009v1-05m5u8w",
    "json": "https://feed7.dev/p/2608-04009v1-05m5u8w.json",
    "markdown": "https://feed7.dev/p/2608-04009v1-05m5u8w.md"
  }
}