{
  "schema_version": "1.0",
  "id": "s8:https://www.youtube.com/watch?v=jRCpXUjz4CI",
  "slug": "everything-is-a-rollout-alex-shaw-ryan-marten-terminal-bench-harbor-laud-0iz4rgx",
  "url": "https://feed7.dev/p/everything-is-a-rollout-alex-shaw-ryan-marten-terminal-bench-harbor-laud-0iz4rgx",
  "title": "Everything Is a Rollout — Alex Shaw + Ryan Marten, Terminal-Bench, Harbor, Laude Institute",
  "why_included": "Harbor frames agent development as an empirical loop: run agents in reproducible sandboxes, verify outcomes, inspect trajectories, and evaluate every harness or model change.",
  "summary": "**Harbor** specifies agent environments and runs any agent, model, sandbox, and task combination in parallel. Its registry reportedly contains **three to four hundred eval sets**, with support for separate verification sandboxes, artifacts, and simulated users.",
  "practical_implication": "Treat prompts, skills, tools, and model choices like tunable system parameters. Run repeated rollouts, grade outcomes, inspect recurring failures, and require measured improvement before merging harness changes.",
  "agent_context": "**Harbor** specifies agent environments and runs any agent, model, sandbox, and task combination in parallel. Its registry reportedly contains **three to four hundred eval sets**, with support for separate verification sandboxes, artifacts, and simulated users.\n\nTreat prompts, skills, tools, and model choices like tunable system parameters. Run repeated rollouts, grade outcomes, inspect recurring failures, and require measured improvement before merging harness changes.\n\nA rollout framework supplies infrastructure, not a useful eval by itself. Builders still need representative tasks, reliable verifiers, and checks for reward hacking or overfitting to the evaluation set.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=jRCpXUjz4CI",
    "published_at": "2026-07-24T16:00:06.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "coding"
  ],
  "topics": [
    "agent-evals",
    "agent-reliability",
    "sandboxing"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "A rollout framework supplies infrastructure, not a useful eval by itself. Builders still need representative tasks, reliable verifiers, and checks for reward hacking or overfitting to the evaluation set."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-24T16:00:06.000Z",
  "modified_at": "2026-07-24T16:00:06.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/everything-is-a-rollout-alex-shaw-ryan-marten-terminal-bench-harbor-laud-0iz4rgx",
    "json": "https://feed7.dev/p/everything-is-a-rollout-alex-shaw-ryan-marten-terminal-bench-harbor-laud-0iz4rgx.json",
    "markdown": "https://feed7.dev/p/everything-is-a-rollout-alex-shaw-ryan-marten-terminal-bench-harbor-laud-0iz4rgx.md"
  }
}