{
  "schema_version": "1.1",
  "id": "s8:https://www.youtube.com/watch?v=IDNfAZVKvPE",
  "slug": "my-name-is-my-name-is-a-linguistic-map-for-voice-agents-midam-kim-servic-176sjnb",
  "url": "https://feed7.dev/p/my-name-is-my-name-is-a-linguistic-map-for-voice-agents-midam-kim-servic-176sjnb",
  "title": "\"My name is... my name is...\": A Linguistic Map for Voice Agents — Midam Kim, ServiceNow",
  "why_included": "Voice-agent failures span recognition, wording, turn-taking, and shared context. Debugging only ASR misses the interaction failures that make users repeat themselves or escalate to a person.",
  "summary": "The proposed framework maps voice interaction across listening and speaking channels, each covering sounds, words, interaction timing, and the evolving mental model. These layers are **interdependent** and unfold over a timeline whose spoken evidence immediately disappears.",
  "practical_implication": "Instrument failures by layer: recognition and pronunciation, understood and chosen vocabulary, turn detection and latency, then intent and context retention. Treat corrections as updates to shared state rather than asking the user to repeat the same input.",
  "agent_context": "The proposed framework maps voice interaction across listening and speaking channels, each covering sounds, words, interaction timing, and the evolving mental model. These layers are **interdependent** and unfold over a timeline whose spoken evidence immediately disappears.\n\nInstrument failures by layer: recognition and pronunciation, understood and chosen vocabulary, turn detection and latency, then intent and context retention. Treat corrections as updates to shared state rather than asking the user to repeat the same input.\n\nThis is a diagnostic map, not a prescribed implementation. Dynamic handling must adapt to different speakers, emotions, and language change; the talk names **Eva benchmark** as an end-to-end check but supplies no results.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=IDNfAZVKvPE",
    "published_at": "2026-09-15T16:30:08.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "craft",
  "domains": [
    "audio"
  ],
  "topics": [
    "interface-quality",
    "sound-design",
    "agent-reliability"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "This is a diagnostic map, not a prescribed implementation. Dynamic handling must adapt to different speakers, emotions, and language change; the talk names **Eva benchmark** as an end-to-end check but supplies no results."
  ],
  "connected_context": {
    "meaning": "This supplies a shared diagnostic map for voice-agent failures that were previously treated as separate transcription, latency, pronunciation, or context problems. It emphasizes their interaction over an ephemeral spoken timeline and reframes correction as state repair, while remaining neutral about the architecture needed to implement or evaluate that behavior.",
    "corpus_size": 812,
    "generated_at": "2026-09-19T09:06:33.588Z",
    "connections": [
      {
        "title": "5 Voice Agent Failure Modes You'll Hit in Week One — Venky B, Plivo",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=vblnYHzBgS4",
        "feed7_url": "https://feed7.dev/p/5-voice-agent-failure-modes-you-ll-hit-in-week-one-venky-b-plivo-0xcwuby",
        "reason": "Its transcription, data-capture, latency, and pronunciation failures become concrete instances of the map’s word, sound, and interaction-timing layers."
      },
      {
        "title": "While my guitar gently speaks — Todd Fisher, Philo Ventures",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=E_Txocq-Lrw",
        "feed7_url": "https://feed7.dev/p/while-my-guitar-gently-speaks-todd-fisher-philo-ventures-0mvwcmy",
        "reason": "The talking-guitar system supplies an embodied example of latency and segmentation failures that the linguistic map would instrument across listening and speaking."
      },
      {
        "title": "200 Million Patient Interactions Later — Vivek Muppalla, Hippocratic AI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=AN65uc645mE",
        "feed7_url": "https://feed7.dev/p/200-million-patient-interactions-later-vivek-muppalla-hippocratic-ai-1axcolg",
        "reason": "The clinical stack’s specialist checks and contextual recognition illustrate one possible implementation of layer-specific detection, although the map itself prescribes no architecture."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-15T16:30:08.000Z",
  "modified_at": "2026-09-15T16:30:08.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/my-name-is-my-name-is-a-linguistic-map-for-voice-agents-midam-kim-servic-176sjnb",
    "json": "https://feed7.dev/p/my-name-is-my-name-is-a-linguistic-map-for-voice-agents-midam-kim-servic-176sjnb.json",
    "markdown": "https://feed7.dev/p/my-name-is-my-name-is-a-linguistic-map-for-voice-agents-midam-kim-servic-176sjnb.md"
  }
}