{
  "schema_version": "1.1",
  "id": "auto-9e82a687ae",
  "slug": "benchmarking-coding-agents-on-new-vs-legacy-codebases-de-9e82a687ae",
  "url": "https://feed7.dev/p/benchmarking-coding-agents-on-new-vs-legacy-codebases-de-9e82a687ae",
  "title": "Benchmarking Coding Agents on New vs Legacy Codebases — Denys Linkov, Wisedocs",
  "why_included": "Judge coding agents with explicit requirements and end-to-end tests, since fast output may be incomplete scaffolding.",
  "summary": "A production refactor shows why coding-agent evaluations need acceptance criteria and end-to-end verification: fast output can still be incomplete scaffolding.",
  "practical_implication": "Benchmark agents against explicit requirements, runnable end-to-end tests, deployment constraints, and hidden assumptions—not elapsed time or lines changed. Wisedocs also found a monorepo simpler for verification and sandbox setup across its former 10+ repositories.",
  "agent_context": "An early O3-assisted task took **3 hours** and produced 10 major mistakes; newer Sonnet 4.6 solved it after one extra iteration and Opus 4.8 nearly one-shot it. A broader GPT-5.5 attempt finished in **10m 22s** but mostly wrote scaffolding.\n\nBenchmark agents against explicit requirements, runnable end-to-end tests, deployment constraints, and hidden assumptions—not elapsed time or lines changed. Wisedocs also found a monorepo simpler for verification and sandbox setup across its former **10+ repositories**.\n\nThese are task-specific observations from one refactor, not controlled cross-model results. Human review remained part of the project, and only **15 of 17 requirements** were met during the migration.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=7vn4WpqNpck",
    "published_at": "2026-08-08T00:00:00.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "coding"
  ],
  "topics": [
    "agent-evals",
    "agent-reliability",
    "coding-agents"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [],
  "connected_context": null,
  "lifecycle": "New",
  "published_at": "2026-08-08T00:00:00.000Z",
  "modified_at": "2026-08-08T00:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/benchmarking-coding-agents-on-new-vs-legacy-codebases-de-9e82a687ae",
    "json": "https://feed7.dev/p/benchmarking-coding-agents-on-new-vs-legacy-codebases-de-9e82a687ae.json",
    "markdown": "https://feed7.dev/p/benchmarking-coding-agents-on-new-vs-legacy-codebases-de-9e82a687ae.md"
  }
}