{
  "schema_version": "1.1",
  "id": "s8:https://www.youtube.com/watch?v=dQ-_i1tZiws",
  "slug": "tribal-dungeons-of-global-shipping-ai-agents-at-global-scale-dmitry-buyk-0kcigeh",
  "url": "https://feed7.dev/p/tribal-dungeons-of-global-shipping-ai-agents-at-global-scale-dmitry-buyk-0kcigeh",
  "title": "Tribal Dungeons of Global Shipping: AI Agents at Global Scale — Dmitry Buykin, Maersk",
  "why_included": "Maersk’s production agents depend less on a clever loop than on executable SOPs, bounded tools, replayable traces, and a correction system shared by experts and engineers.",
  "summary": "Maersk runs **over 200 agent instances** for shipping operations, where legacy systems can stretch latency from minutes to **10 minutes**. Its executable SOP corpus captures preconditions, decisions, calls, validation, recovery, and evidence.",
  "practical_implication": "Treat the harness and correction loop as the product. Convert screenshots and expert habits into testable procedures, constrain production rights, cluster failures, and replay real cases before promoting changes.",
  "agent_context": "Maersk runs **over 200 agent instances** for shipping operations, where legacy systems can stretch latency from minutes to **10 minutes**. Its executable SOP corpus captures preconditions, decisions, calls, validation, recovery, and evidence.\n\nTreat the harness and correction loop as the product. Convert screenshots and expert habits into testable procedures, constrain production rights, cluster failures, and replay real cases before promoting changes.\n\nThis approach required **over 100,000 corrections in 9 months**, with expert time still the bottleneck. The talk reports Maersk’s operating method, not a portable benchmark or proof that the same architecture fits smaller workflows.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=dQ-_i1tZiws",
    "published_at": "2026-08-29T17:30:21.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "agent",
  "domains": [
    "coding"
  ],
  "topics": [
    "harness-engineering",
    "agent-reliability",
    "tool-use"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "This approach required **over 100,000 corrections in 9 months**, with expert time still the bottleneck. The talk reports Maersk’s operating method, not a portable benchmark or proof that the same architecture fits smaller workflows."
  ],
  "connected_context": {
    "meaning": "This turns familiar harness controls into evidence from a large, latency-heavy production operation: reliability comes from executable procedures, constrained rights, replay, and an institutional correction loop. The volume of corrections and continuing expert bottleneck narrow the automation claim, showing that stronger infrastructure reorganizes domain labor rather than eliminating it, and that the pattern is not automatically portable to smaller workflows.",
    "corpus_size": 669,
    "generated_at": "2026-09-03T10:01:24.920Z",
    "connections": [
      {
        "title": "AI Agents Are Just Distributed Systems Now — Salman Munaf, TikTok",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=hD9-V56FNRI",
        "feed7_url": "https://feed7.dev/p/ai-agents-are-just-distributed-systems-now-salman-munaf-tiktok-1v4yc47",
        "reason": "Maersk’s validation, recovery, evidence, and constrained production rights provide operating evidence for the distributed-systems controls this candidate says are required when agents mutate external state."
      },
      {
        "title": "Twin: Playing an Unknown Game with a Test-Time Digital Twin",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.14490v1",
        "feed7_url": "https://feed7.dev/p/2608-14490v1-0d3xjvt",
        "reason": "Both make executable replay a promotion gate, but Twin demonstrates it in a cheap simulator while Maersk shows the much heavier correction and expert-maintenance burden of applying replay to production operations."
      },
      {
        "title": "Learning on the Job: The Future of Post-Training — Raymond Feng, Applied Compute",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=k35LeKZEhiE",
        "feed7_url": "https://feed7.dev/p/learning-on-the-job-the-future-of-post-training-raymond-feng-applied-com-17u0m7t",
        "reason": "Maersk’s large correction corpus is the kind of production feedback infrastructure this candidate considers valuable, while its slow legacy workflows and expert bottleneck illustrate why such experience is harder to replay and convert into training updates."
      },
      {
        "title": "Your Agent Just Authorized What?! — Jay Mok & Ben Coumes, Paypal",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=vGn6N4-bxBY",
        "feed7_url": "https://feed7.dev/p/your-agent-just-authorized-what-jay-mok-ben-coumes-paypal-024znqi",
        "reason": "Maersk’s constrained production rights reinforce the candidate’s consequence-sensitive authorization principle, adding a concrete operational setting where agent permissions must be bounded and actions leave evidence."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-08-29T17:30:21.000Z",
  "modified_at": "2026-08-29T17:30:21.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/tribal-dungeons-of-global-shipping-ai-agents-at-global-scale-dmitry-buyk-0kcigeh",
    "json": "https://feed7.dev/p/tribal-dungeons-of-global-shipping-ai-agents-at-global-scale-dmitry-buyk-0kcigeh.json",
    "markdown": "https://feed7.dev/p/tribal-dungeons-of-global-shipping-ai-agents-at-global-scale-dmitry-buyk-0kcigeh.md"
  }
}