{
  "schema_version": "1.1",
  "id": "s8:https://www.youtube.com/watch?v=ewtOo0scUh0",
  "slug": "data-and-environment-curation-for-post-training-llms-mahesh-sathiamoorth-19ab77a",
  "url": "https://feed7.dev/p/data-and-environment-curation-for-post-training-llms-mahesh-sathiamoorth-19ab77a",
  "title": "Data and Environment Curation for Post-Training LLMs — Mahesh Sathiamoorthy, Bespoke Labs",
  "why_included": "Post-training gains depend heavily on task selection, rollout quality, and environment design. For many enterprise agents, curated SFT may deliver most of the value before costly RL.",
  "summary": "Bespoke’s curation recipe selects source prompts, mixes and filters them, generates teacher answers, then filters again. Across its work, **multiple answers per question** helped, while the strongest model was not always the best teacher.",
  "practical_implication": "Treat data recipes and environments as versioned engineering assets. Start with **SFT** for the required behavior, run ablations at each curation stage, and reserve **RL** for gains that justify its added compute and infrastructure.",
  "agent_context": "Bespoke’s curation recipe selects source prompts, mixes and filters them, generates teacher answers, then filters again. Across its work, **multiple answers per question** helped, while the strongest model was not always the best teacher.\n\nTreat data recipes and environments as versioned engineering assets. Start with **SFT** for the required behavior, run ablations at each curation stage, and reserve **RL** for gains that justify its added compute and infrastructure.\n\nSynthetic rewriting and task augmentation did not reliably help in the reported agent work. Production datasets can also be imbalanced, so fine-tuning may amplify rare-looking attributes unless the mix and outputs are checked carefully.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=ewtOo0scUh0",
    "published_at": "2026-07-31T22:00:06.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "agent",
  "domains": [
    "coding",
    "data"
  ],
  "topics": [
    "harness-engineering",
    "agent-reliability",
    "sandboxing"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "Synthetic rewriting and task augmentation did not reliably help in the reported agent work. Production datasets can also be imbalanced, so fine-tuning may amplify rare-looking attributes unless the mix and outputs are checked carefully."
  ],
  "connected_context": {
    "meaning": "This turns post-training data and environments into versioned, ablated engineering systems rather than assuming more synthetic data, a stronger teacher, or RL will improve agents. Against the candidates’ emphasis on runtime harnesses and controls, it adds an upstream reliability constraint: prompt mix, answer diversity, filtering, and production imbalance can determine whether those systems receive useful behavior at all.",
    "corpus_size": 318,
    "generated_at": "2026-08-01T10:08:44.175Z",
    "connections": [
      {
        "title": "OpenForgeRL: Train Harness-native Agents in Any Environment",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.21557v1",
        "feed7_url": "https://feed7.dev/p/2607-21557v1-0blvz16",
        "reason": "OpenForgeRL supplies deployment-harness-native RL infrastructure, while this signal narrows when that machinery is warranted by recommending curated SFT first and RL only for gains that justify its extra cost."
      },
      {
        "title": "Scaling to Long Horizons — Ross Taylor & Chengxi Taylor, General Reasoning",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=2bvtay8wGYI",
        "feed7_url": "https://feed7.dev/p/scaling-to-long-horizons-ross-taylor-chengxi-taylor-general-reasoning-0jwtg4d",
        "reason": "The long-horizon work details RL’s credit, context, and scheduling costs; this signal reinforces treating those costs as a reason to establish SFT gains and stage-by-stage ablations before adopting RL."
      },
      {
        "title": "Why Off-the-Shelf AI Doesn't Understand Money — Udi Menkes, Intuit",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=Owb8g3yDyzo",
        "feed7_url": "https://feed7.dev/p/why-off-the-shelf-ai-doesn-t-understand-money-udi-menkes-intuit-0y6w9rk",
        "reason": "Intuit requires valid state-action-outcome evidence for domain recommendations; this signal broadens that data-quality prerequisite to the full training recipe, including source mix, teacher outputs, filtering, and imbalance checks."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-07-31T22:00:06.000Z",
  "modified_at": "2026-07-31T22:00:06.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/data-and-environment-curation-for-post-training-llms-mahesh-sathiamoorth-19ab77a",
    "json": "https://feed7.dev/p/data-and-environment-curation-for-post-training-llms-mahesh-sathiamoorth-19ab77a.json",
    "markdown": "https://feed7.dev/p/data-and-environment-curation-for-post-training-llms-mahesh-sathiamoorth-19ab77a.md"
  }
}