{
  "schema_version": "1.0",
  "id": "s8:https://www.youtube.com/watch?v=31GUkCBD-Uc",
  "slug": "building-closed-loop-evals-for-a-multimodal-agent-at-scale-soumya-gupta-1cqjbe2",
  "url": "https://feed7.dev/p/building-closed-loop-evals-for-a-multimodal-agent-at-scale-soumya-gupta-1cqjbe2",
  "title": "Building Closed-Loop Evals for a Multimodal Agent at Scale — Soumya Gupta & Jai Chopra, Uber",
  "why_included": "Uber’s image-editing agent uses routing, iterative QA, golden-set gates, and production feedback to avoid costly edits, hallucinated food, and quality regressions.",
  "summary": "Uber’s pipeline routes each image to enhance or skip, loops edits through a QA agent, and publishes only passing results. Evaluation combines router precision and recall, **pass at K**, pairwise image comparison, and criteria such as faithfulness, completeness, naturalness, and realism.",
  "practical_implication": "Start with **end-to-end logging**, then replay production mismatches against human labels. Tune the responsible agent or configuration, benchmark it on a **golden data set**, and gate deployment before using marketplace outcomes such as conversion.",
  "agent_context": "Uber’s pipeline routes each image to enhance or skip, loops edits through a QA agent, and publishes only passing results. Evaluation combines router precision and recall, **pass at K**, pairwise image comparison, and criteria such as faithfulness, completeness, naturalness, and realism.\n\nStart with **end-to-end logging**, then replay production mismatches against human labels. Tune the responsible agent or configuration, benchmark it on a **golden data set**, and gate deployment before using marketplace outcomes such as conversion.\n\nMore iterations can raise pass rate but also cost compute or reward conservative, trivial edits. A routing miss can degrade an already-good photo or encourage hallucinated items, and the talk does not disclose the proprietary rubric behind its final quality judgment.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=31GUkCBD-Uc",
    "published_at": "2026-07-24T22:00:25.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "image"
  ],
  "topics": [
    "agent-evals",
    "agent-reliability",
    "generative-media"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "More iterations can raise pass rate but also cost compute or reward conservative, trivial edits. A routing miss can degrade an already-good photo or encourage hallucinated items, and the talk does not disclose the proprietary rubric behind its final quality judgment."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-24T22:00:25.000Z",
  "modified_at": "2026-07-24T22:00:25.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/building-closed-loop-evals-for-a-multimodal-agent-at-scale-soumya-gupta-1cqjbe2",
    "json": "https://feed7.dev/p/building-closed-loop-evals-for-a-multimodal-agent-at-scale-soumya-gupta-1cqjbe2.json",
    "markdown": "https://feed7.dev/p/building-closed-loop-evals-for-a-multimodal-agent-at-scale-soumya-gupta-1cqjbe2.md"
  }
}