{
  "schema_version": "1.1",
  "id": "auto-4951efd983",
  "slug": "swe-gate-passing-functional-tests-is-not-enough-for-soft-4951efd983",
  "url": "https://feed7.dev/p/swe-gate-passing-functional-tests-is-not-enough-for-soft-4951efd983",
  "title": "SWE-Gate: Passing Functional Tests Is Not Enough for Software Engineering Agents",
  "why_included": "Green tests missed review constraints in 221 of 644 passing repairs, so encode repository expectations as separate executable checks.",
  "summary": "SWE-Gate shows why green tests are an incomplete agent-eval signal: 221 of 644 functionally passing repairs still violated constraints derived from code review.",
  "practical_implication": "Treat green tests as one gate, not final acceptance, for coding-agent patches. Encode review expectations as executable checks where possible, and evaluate issue resolution separately from compliance with repository-specific requirements.",
  "agent_context": "SWE-Gate contains **303 repair instances** from **75 Python repositories**, with separate tests for functionality and constraints derived from real pull-request reviews. Across four model backends, 644 repairs passed functional tests, but **221** still violated review constraints.\n\nTreat green tests as one gate, not final acceptance, for coding-agent patches. Encode review expectations as executable checks where possible, and evaluate issue resolution separately from compliance with repository-specific requirements.\n\nThe benchmark uses synthesized repair instances and a common agent scaffold, so its failure rates may not transfer directly to your repos. It nevertheless exposes a concrete blind spot in functional-only coding-agent evaluations.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.04167v1",
    "published_at": "2026-09-03T00:00:00.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "coding"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Automatically selected from source material; feed7 has not independently tested the claim."
  ],
  "connected_context": null,
  "lifecycle": "New",
  "published_at": "2026-09-03T00:00:00.000Z",
  "modified_at": "2026-09-03T00:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/swe-gate-passing-functional-tests-is-not-enough-for-soft-4951efd983",
    "json": "https://feed7.dev/p/swe-gate-passing-functional-tests-is-not-enough-for-soft-4951efd983.json",
    "markdown": "https://feed7.dev/p/swe-gate-passing-functional-tests-is-not-enough-for-soft-4951efd983.md"
  }
}