{
  "schema_version": "1.0",
  "id": "s8:https://www.youtube.com/watch?v=O-CBZ3JtRvo",
  "slug": "training-frontier-models-to-out-think-hackers-uri-rolls-arithmetic-thom-0dvs2vw",
  "url": "https://feed7.dev/p/training-frontier-models-to-out-think-hackers-uri-rolls-arithmetic-thom-0dvs2vw",
  "title": "Training Frontier Models to Out-Think Hackers — Uri Rolls, Arithmetic & Thom Wolf, Hugging Face",
  "why_included": "This security eval tests whether agents can discover and exploit logic flaws across live chained services, using hidden zero-days and deterministic grading instead of source-code pattern matching.",
  "summary": "The benchmark builds live, chained application environments around **real zero-days** found by human researchers. Agents start with limited access, receive **no internet or source code**, and must infer the system’s state while pursuing an exploit.",
  "practical_implication": "For security-agent evals, test black-box reasoning across authentication, permissions, and service boundaries. Record partial progress with **deterministic graders** so model or harness changes can be compared even when full exploitation is rare.",
  "agent_context": "The benchmark builds live, chained application environments around **real zero-days** found by human researchers. Agents start with limited access, receive **no internet or source code**, and must infer the system’s state while pursuing an exploit.\n\nFor security-agent evals, test black-box reasoning across authentication, permissions, and service boundaries. Record partial progress with **deterministic graders** so model or harness changes can be compared even when full exploitation is rare.\n\nThe presented benchmark remained extremely difficult, with **one solve at k=1**. Its access-control scenarios demonstrate a demanding evaluation method, but not yet the broader claim that specialized open models can replace existing defensive systems.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=O-CBZ3JtRvo",
    "published_at": "2026-07-24T05:19:15.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "security"
  ],
  "topics": [
    "agent-evals",
    "reasoning",
    "open-models"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "The presented benchmark remained extremely difficult, with **one solve at k=1**. Its access-control scenarios demonstrate a demanding evaluation method, but not yet the broader claim that specialized open models can replace existing defensive systems."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-24T05:19:15.000Z",
  "modified_at": "2026-07-24T05:19:15.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/training-frontier-models-to-out-think-hackers-uri-rolls-arithmetic-thom-0dvs2vw",
    "json": "https://feed7.dev/p/training-frontier-models-to-out-think-hackers-uri-rolls-arithmetic-thom-0dvs2vw.json",
    "markdown": "https://feed7.dev/p/training-frontier-models-to-out-think-hackers-uri-rolls-arithmetic-thom-0dvs2vw.md"
  }
}