{
  "schema_version": "1.0",
  "id": "s13:https://arxiv.org/abs/2607.24720v1",
  "slug": "2607-24720v1-0gihy13",
  "url": "https://feed7.dev/p/2607-24720v1-0gihy13",
  "title": "The Physics of Multi-Turn Long-Horizon Planning: From Pre-training to Post-training via Single- and Multi-Teacher On-Policy Agentic Distillation",
  "why_included": "Controlled experiments suggest long-horizon agent planning depends on explicit state transitions, some compositional trajectories, and compatible teacher patterns—not atomic skills alone.",
  "summary": "The study separates planning into **three stages**: acquisition in pre-training, shaping through GRPO or OPD, and integration through MOPD. Explicit state-transition modeling and even limited long-horizon data improved generalization, while suboptimal trajectories compounded errors.",
  "practical_implication": "For agent builders, the practical lesson is to train and evaluate complete trajectories, not just isolated tool skills. **OPD** had a broader useful region than GRPO in low-quality, long-horizon settings, and **MOPD** combined compatible planning patterns across environments.",
  "agent_context": "The study separates planning into **three stages**: acquisition in pre-training, shaping through GRPO or OPD, and integration through MOPD. Explicit state-transition modeling and even limited long-horizon data improved generalization, while suboptimal trajectories compounded errors.\n\nFor agent builders, the practical lesson is to train and evaluate complete trajectories, not just isolated tool skills. **OPD** had a broader useful region than GRPO in low-quality, long-horizon settings, and **MOPD** combined compatible planning patterns across environments.\n\nTeacher knowledge is not automatically additive. Distilling unfamiliar procedures could damage the student's existing world model, while conflicting teacher patterns caused severe interference; the supplied material does not quantify these effects on production coding tasks.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.24720v1",
    "published_at": "2026-07-27T17:55:03.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "agent",
  "domains": [
    "coding"
  ],
  "topics": [
    "harness-engineering",
    "multi-agent",
    "agent-reliability"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Teacher knowledge is not automatically additive. Distilling unfamiliar procedures could damage the student's existing world model, while conflicting teacher patterns caused severe interference; the supplied material does not quantify these effects on production coding tasks."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-27T17:55:03.000Z",
  "modified_at": "2026-07-27T17:55:03.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-24720v1-0gihy13",
    "json": "https://feed7.dev/p/2607-24720v1-0gihy13.json",
    "markdown": "https://feed7.dev/p/2607-24720v1-0gihy13.md"
  }
}