{
  "schema_version": "1.0",
  "id": "s13:https://arxiv.org/abs/2607.24717v1",
  "slug": "2607-24717v1-0468y8t",
  "url": "https://feed7.dev/p/2607-24717v1-0468y8t",
  "title": "DataOrchestra: Learning to Orchestrate Per-Example Curation of Pretraining Data",
  "why_included": "DataOrchestra chooses a processing pipeline per pre-training example, improving average benchmark results while avoiding compute on chunks that need no transformation.",
  "summary": "DataOrchestra routes each data chunk to drop, preserve, or clean, then chooses programmatic edits or LLM rewrites and generates rewrite instructions. Models from **0.5B to 7B** trained on its output showed stable average gains across **11 benchmarks**.",
  "practical_implication": "Builders preparing training corpora should reconsider corpus-wide cleaning rules. Per-example routing can reserve expensive rewriting for material that needs it, and the reported math continued-pretraining results also exceeded stronger processing baselines.",
  "agent_context": "DataOrchestra routes each data chunk to drop, preserve, or clean, then chooses programmatic edits or LLM rewrites and generates rewrite instructions. Models from **0.5B to 7B** trained on its output showed stable average gains across **11 benchmarks**.\n\nBuilders preparing training corpora should reconsider corpus-wide cleaning rules. Per-example routing can reserve expensive rewriting for material that needs it, and the reported math continued-pretraining results also exceeded stronger processing baselines.\n\nThe abstract gives no gain sizes, processing-cost figures, or operational complexity. It therefore supports the routing principle more clearly than any estimate of whether implementing the orchestrator pays off for a particular dataset.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2607.24717v1",
    "published_at": "2026-07-27T17:54:12.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "model",
  "domains": [
    "data"
  ],
  "topics": [
    "open-models"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The abstract gives no gain sizes, processing-cost figures, or operational complexity. It therefore supports the routing principle more clearly than any estimate of whether implementing the orchestrator pays off for a particular dataset."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-27T17:54:12.000Z",
  "modified_at": "2026-07-27T17:54:12.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2607-24717v1-0468y8t",
    "json": "https://feed7.dev/p/2607-24717v1-0468y8t.json",
    "markdown": "https://feed7.dev/p/2607-24717v1-0468y8t.md"
  }
}