{
  "schema_version": "1.0",
  "id": "s8:https://www.youtube.com/watch?v=Yk87oUPVaxU",
  "slug": "deepswe-a-contamination-resistant-coding-benchmark-james-shi-datacurve-08p61c0",
  "url": "https://feed7.dev/p/deepswe-a-contamination-resistant-coding-benchmark-james-shi-datacurve-08p61c0",
  "title": "DeepSWE: A Contamination-Resistant Coding Benchmark — James Shi, Datacurve",
  "why_included": "DeepSWE uses original long-horizon tasks to reduce contamination and expose coding-agent behaviors hidden by saturated PR-mined suites. Its current task mix still underrepresents some everyday work.",
  "summary": "DeepSWE contains **113 original tasks** spanning **91 repositories** and five languages, with a median of one task per repository. Its short prompts still yield solutions averaging five times the lines of code of SWE-Bench Pro.",
  "practical_implication": "Use it when comparing coding agents on sustained repository work, and inspect behavioral traces alongside scores. **DeepSWE v1.1** separates verifier and agent runtimes and removes extra Git references to limit reward hacking.",
  "agent_context": "DeepSWE contains **113 original tasks** spanning **91 repositories** and five languages, with a median of one task per repository. Its short prompts still yield solutions averaging five times the lines of code of SWE-Bench Pro.\n\nUse it when comparing coding agents on sustained repository work, and inspect behavioral traces alongside scores. **DeepSWE v1.1** separates verifier and agent runtimes and removes extra Git references to limit reward hacking.\n\nThe suite currently underrepresents bug localization and refactoring. Its authors also want broader repository coverage and hybrid verification, so it is not yet a complete proxy for routine engineering work.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=Yk87oUPVaxU",
    "published_at": "2026-07-26T18:10:56.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "coding"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "agent-reliability"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "The suite currently underrepresents bug localization and refactoring. Its authors also want broader repository coverage and hybrid verification, so it is not yet a complete proxy for routine engineering work."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-26T18:10:56.000Z",
  "modified_at": "2026-07-26T18:10:56.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/deepswe-a-contamination-resistant-coding-benchmark-james-shi-datacurve-08p61c0",
    "json": "https://feed7.dev/p/deepswe-a-contamination-resistant-coding-benchmark-james-shi-datacurve-08p61c0.json",
    "markdown": "https://feed7.dev/p/deepswe-a-contamination-resistant-coding-benchmark-james-shi-datacurve-08p61c0.md"
  }
}