{
  "schema_version": "1.1",
  "id": "auto-caaf99b872",
  "slug": "an-empirical-study-of-harness-design-for-coding-agents-caaf99b872",
  "url": "https://feed7.dev/p/an-empirical-study-of-harness-design-for-coding-agents-caaf99b872",
  "title": "An Empirical Study of Harness Design for Coding Agents",
  "why_included": "Harness tests favor rule-based elision before summarization, selective planning, and bash-only tools for models already strong at CLI work.",
  "summary": "Harness components pay off differently by model and budget. Elide before summarizing, use planning selectively, and avoid elaborate tools when the model is already strong with bash.",
  "practical_implication": "Stage rule-based elision before LLM summarization. Use planning as an accuracy scaffold for weaker models and a cost control for stronger ones; offer predefined tools when bash skill is weak, but consider bash-only operation for capable models on CLI-heavy work.",
  "agent_context": "Researchers tested **176 matched settings** across **four models**, varying planning, action space, and context management on SWE-Bench Verified and Terminal-Bench 2.1. Context handling mattered more as budgets tightened, chiefly by preventing overflow.\n\nStage rule-based elision before LLM summarization. Use planning as an accuracy scaffold for weaker models and a cost control for stronger ones; offer predefined tools when bash skill is weak, but consider bash-only operation for capable models on CLI-heavy work.\n\nRecoverable elision added machinery without an accuracy gain because models rarely used recovery. The evidence spans **five context strategies** and **four window budgets**, but the abstract provides no effect sizes and covers only two benchmarks.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.20804v1",
    "published_at": "2026-09-17T00:00:00.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "agent",
  "domains": [
    "coding"
  ],
  "topics": [
    "harness-engineering",
    "context-engineering",
    "tool-use"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Automatically selected from source material; feed7 has not independently tested the claim."
  ],
  "connected_context": null,
  "lifecycle": "New",
  "published_at": "2026-09-17T00:00:00.000Z",
  "modified_at": "2026-09-17T00:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/an-empirical-study-of-harness-design-for-coding-agents-caaf99b872",
    "json": "https://feed7.dev/p/an-empirical-study-of-harness-design-for-coding-agents-caaf99b872.json",
    "markdown": "https://feed7.dev/p/an-empirical-study-of-harness-design-for-coding-agents-caaf99b872.md"
  }
}