{
  "schema_version": "1.1",
  "id": "auto-82b0afd7a5",
  "slug": "run-terminal-bench-and-other-harbor-evals-on-vercel-sand-82b0afd7a5",
  "url": "https://feed7.dev/p/run-terminal-bench-and-other-harbor-evals-on-vercel-sand-82b0afd7a5",
  "title": "Run Terminal-Bench and other Harbor evals on Vercel Sandbox",
  "why_included": "Harbor can run repeatable agent benchmarks in isolated microVMs, parallelizing model comparisons while keeping secrets outside each sandbox.",
  "summary": "Harbor can run Terminal-Bench and related evals in isolated Vercel microVMs, enabling parallel model comparisons without putting injected credentials inside each sandbox.",
  "practical_implication": "Move repeatable agent evaluations off a constrained local machine, parallelize trials, and swap the gateway `--model` value to compare providers while keeping the benchmark command stable.",
  "agent_context": "**Harbor 0.22.0 or later** can run Terminal-Bench, SWE-bench, tau3-bench, OSWorld, and other registry evals on Vercel Sandbox with `--env vercel`. Each trial gets an isolated **Firecracker microVM**.\n\nMove repeatable agent evaluations off a constrained local machine, parallelize trials, and swap the gateway `--model` value to compare providers while keeping the benchmark command stable.\n\nNetwork policy is enforced outside the VM, and optional secrets are attached only to matching outbound requests. The material does not quantify cost, startup overhead, concurrency limits, or result reproducibility.",
  "source": {
    "name": "Vercel",
    "url": "https://vercel.com/changelog/run-terminal-bench-and-other-harbor-evals-on-vercel-sandbox",
    "published_at": "2026-09-17T00:00:00.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Engineering Post",
  "layer": "benchmark",
  "domains": [
    "coding",
    "security"
  ],
  "topics": [
    "agent-evals",
    "sandboxing",
    "gateways"
  ],
  "verification": {
    "status": "official_source",
    "label": "Official Source",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [],
  "connected_context": null,
  "lifecycle": "New",
  "published_at": "2026-09-17T00:00:00.000Z",
  "modified_at": "2026-09-17T00:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/run-terminal-bench-and-other-harbor-evals-on-vercel-sand-82b0afd7a5",
    "json": "https://feed7.dev/p/run-terminal-bench-and-other-harbor-evals-on-vercel-sand-82b0afd7a5.json",
    "markdown": "https://feed7.dev/p/run-terminal-bench-and-other-harbor-evals-on-vercel-sand-82b0afd7a5.md"
  }
}