{
  "schema_version": "1.1",
  "id": "auto-a24f8f74a4",
  "slug": "computer-use-at-the-edge-of-the-statistical-precipice-pi-a24f8f74a4",
  "url": "https://feed7.dev/p/computer-use-at-the-edge-of-the-statistical-precipice-pi-a24f8f74a4",
  "title": "Computer Use at the Edge of the Statistical Precipice — Pierluca D'Oro, Programma Labs",
  "why_included": "Vary task data, appearance, and initial state so agent evals measure adaptation instead of rewarding replayed action scripts.",
  "summary": "Static computer-use benchmarks can reward memorized action scripts rather than adaptation. Vary task state, verify every generated case, and calculate uncertainty across both actions and environments.",
  "practical_implication": "For agent evals, vary data, appearance, and initial state; automatically reject invalid combinations; and use privileged verifiers inside a sandbox. DGWorld applies this design across 15 apps, 387 scenarios, and 3.2 million verified configurations.",
  "agent_context": "A replay agent stores one winning trajectory per task and blindly repeats it. On deterministic OSWorld and MobileWorld-style evaluations, a script **under 1 MB** can match or beat the frontier model that generated its traces, exposing benchmark replayability.\n\nFor agent evals, vary data, appearance, and initial state; automatically reject invalid combinations; and use privileged verifiers inside a sandbox. DGWorld applies this design across **15 apps, 387 scenarios, and 3.2 million verified configurations**.\n\nUncertainty must cover both model actions and environment variation. The talk reports that rollout-only intervals can provide roughly **17–20% coverage** where a correctly structured method approaches the intended 95%, but applying that method requires a benchmark with explicit variation structure.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=CTLa_p6iOiY",
    "published_at": "2026-08-14T00:00:00.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "agent-reliability"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [],
  "connected_context": null,
  "lifecycle": "New",
  "published_at": "2026-08-14T00:00:00.000Z",
  "modified_at": "2026-08-14T00:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/computer-use-at-the-edge-of-the-statistical-precipice-pi-a24f8f74a4",
    "json": "https://feed7.dev/p/computer-use-at-the-edge-of-the-statistical-precipice-pi-a24f8f74a4.json",
    "markdown": "https://feed7.dev/p/computer-use-at-the-edge-of-the-statistical-precipice-pi-a24f8f74a4.md"
  }
}