{
  "schema_version": "1.1",
  "id": "s8:https://www.youtube.com/watch?v=KMR_RBoCa4M",
  "slug": "simulationmaxxing-how-we-ship-agents-20-faster-aman-gupta-nubank-shreya-0r3nm6u",
  "url": "https://feed7.dev/p/simulationmaxxing-how-we-ship-agents-20-faster-aman-gupta-nubank-shreya-0r3nm6u",
  "title": "SimulationMaxxing: How we ship agents 20× faster — Aman Gupta (Nubank) + Shreya Rajpal (Snowglobe)",
  "why_included": "Nubank uses simulated multi-turn traces to evaluate agent changes before production, shortening release cycles while checking simulation results against real data and human review.",
  "summary": "Nubank reports using simulation across **five production agents** serving a bank with **135 million customers**. Simulated personas exercise real agents with mocked tools and consistent synthetic state, producing multi-turn traces that feed the existing evaluation pipeline.",
  "practical_implication": "Generate eval cases before production when real traces are slow or risky to collect. Nubank says this shortened its release cycle by about **20×**, enabled more experiments, caught regressions, and helped one agent double its customer-satisfaction measure.",
  "agent_context": "Nubank reports using simulation across **five production agents** serving a bank with **135 million customers**. Simulated personas exercise real agents with mocked tools and consistent synthetic state, producing multi-turn traces that feed the existing evaluation pipeline.\n\nGenerate eval cases before production when real traces are slow or risky to collect. Nubank says this shortened its release cycle by about **20×**, enabled more experiments, caught regressions, and helped one agent double its customer-satisfaction measure.\n\nSynthetic traces are useful only when the sim-to-real gap is measured. Domain experts judged **80%** of reviewed simulations usable, but teams still need online comparisons, human review, aligned metrics, and real production data.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=KMR_RBoCa4M",
    "published_at": "2026-07-29T19:00:06.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [],
  "topics": [
    "agent-evals",
    "agent-reliability",
    "harness-engineering"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "Synthetic traces are useful only when the sim-to-real gap is measured. Domain experts judged **80%** of reviewed simulations usable, but teams still need online comparisons, human review, aligned metrics, and real production data."
  ],
  "connected_context": {
    "meaning": "This provides production-scale evidence that synthetic, state-consistent multi-turn traces can move evaluation earlier and accelerate agent iteration, even when real traces are scarce or risky. It also narrows the claim: simulation is an eval-data multiplier, not a substitute for production feedback, because usefulness and sim-to-real alignment still require expert review, online comparison, and real outcomes.",
    "corpus_size": 297,
    "generated_at": "2026-07-31T10:07:27.765Z",
    "connections": [
      {
        "title": "From Agent Traces to Agent Simulations — Rustem Feyzkhanov, Snorkel AI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=Ib5t2RLtxvM",
        "feed7_url": "https://feed7.dev/p/from-agent-traces-to-agent-simulations-rustem-feyzkhanov-snorkel-ai-0zwlzjq",
        "reason": "The approaches are complementary: Nubank generates cases before sufficient production data exists, while trace reconstruction turns observed production behavior into fixed replay environments."
      },
      {
        "title": "Vending-Bench: Long-Horizon Agent Evals — Lukas Petersson, Andon Labs",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=cO8qC6HBuBg",
        "feed7_url": "https://feed7.dev/p/vending-bench-long-horizon-agent-evals-lukas-petersson-andon-labs-0fu78nz",
        "reason": "Vending-Bench reinforces the stated limitation that repeatable simulations must be paired with real-world tests because agent behavior can diverge outside the eval."
      },
      {
        "title": "Building Closed-Loop Evals for a Multimodal Agent at Scale — Soumya Gupta & Jai Chopra, Uber",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=31GUkCBD-Uc",
        "feed7_url": "https://feed7.dev/p/building-closed-loop-evals-for-a-multimodal-agent-at-scale-soumya-gupta-1cqjbe2",
        "reason": "Uber’s closed production-feedback loop supplies the downstream mechanism needed to measure and correct the sim-to-real gap."
      },
      {
        "title": "State of Data — Sean Cai, Independent / State of Data",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=ZyIoTOAbRfs",
        "feed7_url": "https://feed7.dev/p/state-of-data-sean-cai-independent-state-of-data-0v9fy69",
        "reason": "The candidate’s preference for live workflow trajectories qualifies Nubank’s synthetic-first method: generated traces speed early testing, but real trajectories remain necessary evidence."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-07-29T19:00:06.000Z",
  "modified_at": "2026-07-29T19:00:06.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/simulationmaxxing-how-we-ship-agents-20-faster-aman-gupta-nubank-shreya-0r3nm6u",
    "json": "https://feed7.dev/p/simulationmaxxing-how-we-ship-agents-20-faster-aman-gupta-nubank-shreya-0r3nm6u.json",
    "markdown": "https://feed7.dev/p/simulationmaxxing-how-we-ship-agents-20-faster-aman-gupta-nubank-shreya-0r3nm6u.md"
  }
}