{
  "schema_version": "1.0",
  "id": "s8:https://www.youtube.com/watch?v=ZyIoTOAbRfs",
  "slug": "state-of-data-sean-cai-independent-state-of-data-0v9fy69",
  "url": "https://feed7.dev/p/state-of-data-sean-cai-independent-state-of-data-0v9fy69",
  "title": "State of Data — Sean Cai, Independent / State of Data",
  "why_included": "Real workflow traces may teach agents more than manufactured tasks, while benchmark scores can shift with the harness. Build pipelines around live work and test across scaffolds.",
  "summary": "Cai separates saved outputs from process data: trajectories, decisions, and reasoning traces. He calls minimally shaped real workflows **Type 1 data** and expert-manufactured examples **Type 2 data**, arguing that realism comes from the work itself.",
  "practical_implication": "For coding agents, capture actual tool use, state transitions, failure recovery, and outcomes. Evaluate across harnesses and infrastructure: a score from **one benchmark under one scaffold** is only one sample, not a stable measure of capability.",
  "agent_context": "Cai separates saved outputs from process data: trajectories, decisions, and reasoning traces. He calls minimally shaped real workflows **Type 1 data** and expert-manufactured examples **Type 2 data**, arguing that realism comes from the work itself.\n\nFor coding agents, capture actual tool use, state transitions, failure recovery, and outcomes. Evaluate across harnesses and infrastructure: a score from **one benchmark under one scaffold** is only one sample, not a stable measure of capability.\n\nThe talk presents a market thesis, not a controlled study. Its claimed **20–30 vendor** diversification and benchmark failure modes are not quantified in the transcript, while domains such as robotics still face unresolved choices about what data modality to collect.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=ZyIoTOAbRfs",
    "published_at": "2026-07-26T17:00:06.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "coding",
    "data"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "harness-engineering"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "The talk presents a market thesis, not a controlled study. Its claimed **20–30 vendor** diversification and benchmark failure modes are not quantified in the transcript, while domains such as robotics still face unresolved choices about what data modality to collect."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-26T17:00:06.000Z",
  "modified_at": "2026-07-26T17:00:06.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/state-of-data-sean-cai-independent-state-of-data-0v9fy69",
    "json": "https://feed7.dev/p/state-of-data-sean-cai-independent-state-of-data-0v9fy69.json",
    "markdown": "https://feed7.dev/p/state-of-data-sean-cai-independent-state-of-data-0v9fy69.md"
  }
}