{
  "schema_version": "1.1",
  "id": "archive:https://www.youtube.com/watch?v=yqF6XhzbWBk",
  "slug": "inside-847-production-clinical-ai-notes-sebastian-fox-composo-0lrc8td",
  "url": "https://feed7.dev/p/inside-847-production-clinical-ai-notes-sebastian-fox-composo-0lrc8td",
  "title": "Inside 847 Production Clinical AI Notes — Sebastian Fox, Composo",
  "why_included": "Plausible outputs can hide consequential omissions that generic LLM judges miss. Production evals need real failure discovery and retrieved expert judgments, not a frozen rubric alone.",
  "summary": "In the cited production study, **about 1 in 20 notes** contained an error capable of significant harm, nearly 1 in 5 had an important omission, and more than 1 in 10 hallucinated. A strong judge still passed notes containing serious errors.",
  "practical_implication": "For any agent where confident mistakes carry real cost, inspect production outputs, organize recurring failure modes, capture expert corrections and reasoning, then retrieve relevant past judgments into each evaluation. Start with experts commenting freely on real cases.",
  "agent_context": "In the cited production study, **about 1 in 20 notes** contained an error capable of significant harm, nearly 1 in 5 had an important omission, and more than 1 in 10 hallucinated. A strong judge still passed notes containing serious errors.\n\nFor any agent where confident mistakes carry real cost, inspect production outputs, organize recurring failure modes, capture expert corrections and reasoning, then retrieve relevant past judgments into each evaluation. Start with experts commenting freely on real cases.\n\nThis approach depends on sustained expert review and representative production data. It improves the judge’s domain context, but the talk does not establish that retrieval catches every novel failure or removes the need for human oversight.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=yqF6XhzbWBk",
    "published_at": "2026-08-22T17:00:32.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "data"
  ],
  "topics": [
    "agent-evals",
    "agent-reliability",
    "retrieval"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "This approach depends on sustained expert review and representative production data. It improves the judge’s domain context, but the talk does not establish that retrieval catches every novel failure or removes the need for human oversight."
  ],
  "connected_context": {
    "meaning": "This supplies production evidence that strong model judges can approve clinically dangerous outputs, tightening the case against self-grading in high-cost domains. It turns expert review into reusable evaluation context: collect real failures, preserve corrections and reasoning, and retrieve relevant judgments per case. That improves domain grounding but does not automate away expert oversight or guarantee coverage of novel failures.",
    "corpus_size": 545,
    "generated_at": "2026-08-23T18:04:14.259Z",
    "connections": [
      {
        "title": "Trading Desks to Clinical Trials: Parallels in Applied Vertical AI — Ayush Bhardwaj, Allos AI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=Yphdry8ttAQ",
        "feed7_url": "https://feed7.dev/p/trading-desks-to-clinical-trials-parallels-in-applied-vertical-ai-ayush-1hwvsg3",
        "reason": "The clinical error findings provide concrete production support for the claim that generic models and self-grading cannot replace expert causal judgment in vertical agents."
      },
      {
        "title": "Demystifying evals for AI agents",
        "source_name": "Anthropic",
        "source_url": "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents",
        "feed7_url": "https://feed7.dev/p/demystifying-evals-for-ai-agents-1kh2tdz",
        "reason": "Anthropic recommends starting with tasks drawn from real failures; this signal extends that recipe by capturing expert reasoning and retrieving related past judgments into each evaluation."
      },
      {
        "title": "Verifiable Environments for AI in Biology — Kenny Workman, LatchBio",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=3ZMUiFaQ3qg",
        "feed7_url": "https://feed7.dev/p/verifiable-environments-for-ai-in-biology-kenny-workman-latchbio-1vs6y66",
        "reason": "Both show that plausible outputs and brittle graders can miss domain-critical errors, reinforcing human validation even when evaluation includes deterministic or model-based checks."
      },
      {
        "title": "Rethinking Environments for Long-Horizon Work — Rayan Garg, Theta Software",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=2aS7aKoXn64",
        "feed7_url": "https://feed7.dev/p/rethinking-environments-for-long-horizon-work-rayan-garg-theta-software-11r7wbx",
        "reason": "The clinical results strengthen the warning that flexible model judges are fallible; inspecting production outputs and expert corrections complements final-state and trajectory-aware evaluation."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-08-22T17:00:32.000Z",
  "modified_at": "2026-08-22T17:00:32.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/inside-847-production-clinical-ai-notes-sebastian-fox-composo-0lrc8td",
    "json": "https://feed7.dev/p/inside-847-production-clinical-ai-notes-sebastian-fox-composo-0lrc8td.json",
    "markdown": "https://feed7.dev/p/inside-847-production-clinical-ai-notes-sebastian-fox-composo-0lrc8td.md"
  }
}