{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2608.04008v1",
  "slug": "2608-04008v1-0d8bqwv",
  "url": "https://feed7.dev/p/2608-04008v1-0d8bqwv",
  "title": "WorldCup Arena: Prospective, Leakage-Free Evaluation of Frontier LLMs on a Live Tournament",
  "why_included": "A live, pre-kickoff benchmark removes answer leakage by construction and finds six frontier models clustered near a bookmaker-favorite baseline, with no gain from majority voting.",
  "summary": "Across the **39-day 2026 World Cup**, six web-enabled reasoning models made predictions before kickoff for all **104 matches**. The frozen archive contains **4,494 scored predictions**, so answers could not have appeared in training or search results.",
  "practical_implication": "For agent evals, prospective collection is a cleaner leakage control than filtering retrospective questions. Also test disagreement quality before using model ensembles: these systems agreed more often than they were correct, so majority voting added nothing.",
  "agent_context": "Across the **39-day 2026 World Cup**, six web-enabled reasoning models made predictions before kickoff for all **104 matches**. The frozen archive contains **4,494 scored predictions**, so answers could not have appeared in training or search results.\n\nFor agent evals, prospective collection is a cleaner leakage control than filtering retrospective questions. Also test disagreement quality before using model ensembles: these systems agreed more often than they were correct, so majority voting added nothing.\n\nAverage match-outcome accuracy was **63.9%**, level with backing the bookmaker favorite. Results cover one sports tournament, models were narrowly separated, and accuracy fell most on close fixtures despite richer dossiers.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2608.04008v1",
    "published_at": "2026-08-04T17:59:55.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "research"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "model-selection"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Average match-outcome accuracy was **63.9%**, level with backing the bookmaker favorite. Results cover one sports tournament, models were narrowly separated, and accuracy fell most on close fixtures despite richer dossiers."
  ],
  "connected_context": {
    "meaning": "This strengthens leakage-control guidance by showing that predictions recorded before future outcomes provide a cleaner test than repairing already-public benchmarks. It also weakens two common selection shortcuts: the models only matched a simple bookmaker-favorite baseline on average, and correlated errors meant majority voting offered no gain. The evidence remains limited to one tournament with closely grouped systems.",
    "corpus_size": 353,
    "generated_at": "2026-08-05T10:06:09.805Z",
    "connections": [
      {
        "title": "Eval awareness in Claude Opus 4.6’s BrowseComp performance",
        "source_name": "Anthropic",
        "source_url": "https://www.anthropic.com/engineering/eval-awareness-browsecomp",
        "feed7_url": "https://feed7.dev/p/eval-awareness-browsecomp-1q6k277",
        "reason": "BrowseComp shows how a web-enabled model can locate leaked evaluation answers; WorldCup Arena avoids that failure mode by freezing predictions before the answers exist."
      },
      {
        "title": "SocietyBench: Forecasting Counterfactual Social-World Evolution",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.04009v1",
        "feed7_url": "https://feed7.dev/p/2608-04009v1-05m5u8w",
        "reason": "Both evaluate forecasting while controlling event recognition, but SocietyBench transforms historical timelines whereas WorldCup Arena collects predictions prospectively; together they provide complementary reproducible leakage controls."
      },
      {
        "title": "Test-Time Scaling in Reasoning LLMs: Inference Regimes, Evaluation, and Reproducibility",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.04001v1",
        "feed7_url": "https://feed7.dev/p/2608-04001v1-1gj91hk",
        "reason": "WorldCup Arena’s frozen archive supplies replayable prediction evidence, while the test-time-scaling guidance explains what remains necessary for fair comparison: reporting each model’s complete inference and compute protocol."
      },
      {
        "title": "Reward hacking is swamping model intelligence gains",
        "source_name": "Cursor",
        "source_url": "https://cursor.com/blog/reward-hacking-coding-benchmarks",
        "feed7_url": "https://feed7.dev/p/reward-hacking-coding-benchmarks-18ddebo",
        "reason": "Cursor’s sealed harness reduces access to existing fixes, while WorldCup Arena goes further by evaluating answers before ground truth exists; both show that model rankings can change meaningfully when answer access is structurally constrained."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-08-04T17:59:55.000Z",
  "modified_at": "2026-08-04T17:59:55.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2608-04008v1-0d8bqwv",
    "json": "https://feed7.dev/p/2608-04008v1-0d8bqwv.json",
    "markdown": "https://feed7.dev/p/2608-04008v1-0d8bqwv.md"
  }
}