{
  "schema_version": "1.1",
  "id": "auto-7ebecc7b84",
  "slug": "swe-prime-fewer-trajectories-better-performance-7ebecc7b84",
  "url": "https://feed7.dev/p/swe-prime-fewer-trajectories-better-performance-7ebecc7b84",
  "title": "SWE-Prime: Fewer Trajectories, Better Performance",
  "why_included": "SWE-Prime's filtered 10% of coding traces beat the full resolved set, showing that passing outcomes are not automatically clean supervision.",
  "summary": "SWE-Prime finds that filtering coding-agent traces by process and segment quality can beat training on every resolved trajectory, reducing noisy imitation from redundant or risky steps.",
  "practical_implication": "The reported 10% trajectory subset beat training on the full resolved dataset, with relative gains up to 12.2% on SWE-Bench Pro and 24.2% on SWE-Bench Verified. Teams training coding models should evaluate how an issue was solved, not treat a passing outcome as clean supervision.",
  "agent_context": "SWE-Prime filters successful coding-agent traces in **two stages**: whole trajectories are screened for process quality, result quality, and representativeness, then semantic segments are judged for contribution, learnability, and risk. Only selected segments contribute to the training loss.\n\nThe reported **10% trajectory subset** beat training on the full resolved dataset, with relative gains up to **12.2% on SWE-Bench Pro** and **24.2% on SWE-Bench Verified**. Teams training coding models should evaluate how an issue was solved, not treat a passing outcome as clean supervision.\n\nAll segments remain in the input sequence to preserve context, so this is selective loss computation rather than simply deleting weak steps. The material reports benchmark results but does not establish whether the selection criteria transfer to other repositories, agents, or training setups.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2608.27449v1",
    "published_at": "2026-08-27T00:00:00.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "coding",
    "data"
  ],
  "topics": [
    "coding-agents",
    "agent-evals",
    "benchmark-integrity"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Automatically selected from source material; feed7 has not independently tested the claim."
  ],
  "connected_context": null,
  "lifecycle": "New",
  "published_at": "2026-08-27T00:00:00.000Z",
  "modified_at": "2026-08-27T00:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/swe-prime-fewer-trajectories-better-performance-7ebecc7b84",
    "json": "https://feed7.dev/p/swe-prime-fewer-trajectories-better-performance-7ebecc7b84.json",
    "markdown": "https://feed7.dev/p/swe-prime-fewer-trajectories-better-performance-7ebecc7b84.md"
  }
}