{
  "schema_version": "1.1",
  "id": "s8:https://www.youtube.com/watch?v=cJ0EOzey--o",
  "slug": "what-s-next-after-rlhf-diogo-almeida-typesafe-ai-1scytnx",
  "url": "https://feed7.dev/p/what-s-next-after-rlhf-diogo-almeida-typesafe-ai-1scytnx",
  "title": "What's Next After RLHF? — Diogo Almeida, TypeSafe AI",
  "why_included": "RLHF can make agents persuasive assistants without making them dependable autonomous decision-makers. Builders should separate human-pleasing interaction from calibrated automation and keep stakes bounded.",
  "summary": "Almeida argues that **RLHF** optimizes human preference, which suits interactive assistants but can reward confident, agreeable behavior. He contrasts that with **RLVR**, which optimizes verifiable correctness, and describes a separate TypeSafe direction aimed at calibrated decisions.",
  "practical_implication": "When designing coding-agent workflows, distinguish assistance from unattended automation. Keep humans around consequential decisions, demand external evidence for completion, and avoid treating fluent interaction or benchmark strength as proof that an agent can own business-critical actions.",
  "agent_context": "Almeida argues that **RLHF** optimizes human preference, which suits interactive assistants but can reward confident, agreeable behavior. He contrasts that with **RLVR**, which optimizes verifiable correctness, and describes a separate TypeSafe direction aimed at calibrated decisions.\n\nWhen designing coding-agent workflows, distinguish assistance from unattended automation. Keep humans around consequential decisions, demand external evidence for completion, and avoid treating fluent interaction or benchmark strength as proof that an agent can own business-critical actions.\n\nThe talk presents a thesis rather than comparative evaluation data, and the proposed alternative is not technically specified. It does not establish how calibrated post-training performs, scales, or handles failures in deployed software.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=cJ0EOzey--o",
    "published_at": "2026-07-31T23:30:06.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "model",
  "domains": [
    "coding"
  ],
  "topics": [
    "reasoning",
    "coding-agents",
    "agent-reliability"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "The talk presents a thesis rather than comparative evaluation data, and the proposed alternative is not technically specified. It does not establish how calibrated post-training performs, scales, or handles failures in deployed software."
  ],
  "connected_context": {
    "meaning": "This separates models optimized for agreeable assistance from systems trusted to act autonomously, making external verification and decision gates deployment requirements rather than optional polish. It supports bounded human oversight for consequential coding work, but the proposed calibrated-training direction remains a thesis without comparative evidence, implementation detail, or demonstrated production reliability.",
    "corpus_size": 318,
    "generated_at": "2026-08-01T10:08:32.212Z",
    "connections": [
      {
        "title": "Loop Engineering from First Principles — Kyle Mistele, HumanLayer",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=xIt_mTQp6mY",
        "feed7_url": "https://feed7.dev/p/loop-engineering-from-first-principles-kyle-mistele-humanlayer-1nuq7gf",
        "reason": "Loop Engineering supplies a workflow consequence of the thesis: bound each change, produce a reviewable artifact, and pause for human approval instead of granting unattended ownership."
      },
      {
        "title": "Governing agent autonomy with Auto-review",
        "source_name": "Cursor",
        "source_url": "https://cursor.com/blog/agent-autonomy-auto-review",
        "feed7_url": "https://feed7.dev/p/agent-autonomy-auto-review-10ce67w",
        "reason": "Auto-review operationalizes selective oversight by placing a classifier gate before risky actions, though it is a harness control rather than evidence for the proposed post-training alternative."
      },
      {
        "title": "ScarfBench: Benchmarking AI Agents for Enterprise Java Framework Migration",
        "source_name": "huggingface.co",
        "source_url": "https://huggingface.co/blog/ibm-research/scarfbench",
        "feed7_url": "https://feed7.dev/p/scarfbench-1u8lniy",
        "reason": "ScarfBench’s gap between agents’ completion claims and compiling builds directly reinforces the demand for external evidence rather than fluent self-report."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-07-31T23:30:06.000Z",
  "modified_at": "2026-07-31T23:30:06.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/what-s-next-after-rlhf-diogo-almeida-typesafe-ai-1scytnx",
    "json": "https://feed7.dev/p/what-s-next-after-rlhf-diogo-almeida-typesafe-ai-1scytnx.json",
    "markdown": "https://feed7.dev/p/what-s-next-after-rlhf-diogo-almeida-typesafe-ai-1scytnx.md"
  }
}