{
  "schema_version": "1.1",
  "id": "s4:https://vercel.com/changelog/run-terminal-bench-and-other-harbor-evals-on-vercel-sandbox",
  "slug": "run-terminal-bench-and-other-harbor-evals-on-vercel-sandbox-0o8aa4r",
  "url": "https://feed7.dev/p/run-terminal-bench-and-other-harbor-evals-on-vercel-sandbox-0o8aa4r",
  "title": "Run Terminal-Bench and other Harbor evals on Vercel Sandbox",
  "why_included": "Harbor can run Terminal-Bench and related evals in isolated Vercel microVMs, enabling parallel model comparisons without putting injected credentials inside each sandbox.",
  "summary": "**Harbor 0.22.0 or later** can run Terminal-Bench, SWE-bench, tau3-bench, OSWorld, and other registry evals on Vercel Sandbox with `--env vercel`. Each trial gets an isolated **Firecracker microVM**.",
  "practical_implication": "Move repeatable agent evaluations off a constrained local machine, parallelize trials, and swap the gateway `--model` value to compare providers while keeping the benchmark command stable.",
  "agent_context": "**Harbor 0.22.0 or later** can run Terminal-Bench, SWE-bench, tau3-bench, OSWorld, and other registry evals on Vercel Sandbox with `--env vercel`. Each trial gets an isolated **Firecracker microVM**.\n\nMove repeatable agent evaluations off a constrained local machine, parallelize trials, and swap the gateway `--model` value to compare providers while keeping the benchmark command stable.\n\nNetwork policy is enforced outside the VM, and optional secrets are attached only to matching outbound requests. The material does not quantify cost, startup overhead, concurrency limits, or result reproducibility.",
  "source": {
    "name": "Vercel",
    "url": "https://vercel.com/changelog/run-terminal-bench-and-other-harbor-evals-on-vercel-sandbox",
    "published_at": "2026-09-17T19:00:00.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Engineering Post",
  "layer": "benchmark",
  "domains": [
    "coding",
    "security"
  ],
  "topics": [
    "agent-evals",
    "sandboxing",
    "gateways"
  ],
  "verification": {
    "status": "official_source",
    "label": "Official Source",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "Network policy is enforced outside the VM, and optional secrets are attached only to matching outbound requests. The material does not quantify cost, startup overhead, concurrency limits, or result reproducibility."
  ],
  "connected_context": {
    "meaning": "This makes a stable Harbor command usable across isolated cloud trials and gateway models, lowering local-capacity friction for repeated comparisons. It strengthens the empirical evaluation loop, but does not guarantee reproducibility: VM resources, startup behavior, concurrency, cost, and other infrastructure conditions still need to be recorded and controlled.",
    "corpus_size": 807,
    "generated_at": "2026-09-18T10:05:06.035Z",
    "connections": [
      {
        "title": "Everything Is a Rollout — Alex Shaw + Ryan Marten, Terminal-Bench, Harbor, Laude Institute",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=jRCpXUjz4CI",
        "feed7_url": "https://feed7.dev/p/everything-is-a-rollout-alex-shaw-ryan-marten-terminal-bench-harbor-laud-0iz4rgx",
        "reason": "Harbor’s rollout methodology provides the operating rationale for this integration: repeat trials, verify outcomes, inspect trajectories, and reevaluate harness or model changes."
      },
      {
        "title": "Quantifying infrastructure noise in agentic coding evals",
        "source_name": "Anthropic",
        "source_url": "https://www.anthropic.com/engineering/infrastructure-noise",
        "feed7_url": "https://feed7.dev/p/infrastructure-noise-1jyyyw1",
        "reason": "Anthropic’s measured score swing from resource configurations shows why isolated microVMs alone do not make results comparable unless execution resources are documented and controlled."
      },
      {
        "title": "How Tailscale built a customer-facing model router on AI Gateway",
        "source_name": "Vercel",
        "source_url": "https://vercel.com/blog/how-tailscale-built-a-customer-facing-model-router-on-ai-gateway",
        "feed7_url": "https://feed7.dev/p/how-tailscale-built-a-customer-facing-model-router-on-ai-gateway-186s3tu",
        "reason": "Tailscale’s external identity, credential, and sandbox controls reinforce the integration’s design of enforcing network policy and secret attachment outside agent execution."
      },
      {
        "title": "Preferences Over Benchmarks: Model Routing — Archana Kamath & Tyler Gillam, DigitalOcean",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=FvxY8oPoI8o",
        "feed7_url": "https://feed7.dev/p/preferences-over-benchmarks-model-routing-archana-kamath-tyler-gillam-di-06nxtgu",
        "reason": "The routing demonstration supplies a concrete reason to swap gateway models under one benchmark command, while its uncertain quality comparison reinforces the need for repeatable local evals."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-17T19:00:00.000Z",
  "modified_at": "2026-09-17T19:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/run-terminal-bench-and-other-harbor-evals-on-vercel-sandbox-0o8aa4r",
    "json": "https://feed7.dev/p/run-terminal-bench-and-other-harbor-evals-on-vercel-sandbox-0o8aa4r.json",
    "markdown": "https://feed7.dev/p/run-terminal-bench-and-other-harbor-evals-on-vercel-sandbox-0o8aa4r.md"
  }
}