{
  "schema_version": "1.1",
  "id": "atlas-generative-media",
  "slug": "generative-media",
  "title": "Generative Media",
  "url": "https://feed7.dev/atlas/generative-media",
  "current_answer": null,
  "implementation_consequence": null,
  "agent_context": null,
  "confidence": "auto_collected",
  "last_verified": null,
  "last_updated": null,
  "evidence": [],
  "conflicting_sources": [],
  "superseded_claims": [],
  "corpus_evidence": [
    {
      "title": "SOTA Generative Media Panel — Dumitru Erhan, Shane Gu & Nicole Brichtova, Google DeepMind",
      "url": "https://www.youtube.com/watch?v=KLDdXOw6jIc",
      "source_name": "AI Engineer",
      "published_at": "2026-08-30T14:00:06+00:00",
      "summary": "DeepMind’s panel shows why generative-media evals need task-specific human review: broad preferences can miss repeated artifacts, exact sizing, text errors, and brand consistency."
    },
    {
      "title": "MiniMax H3 and H3 Max are 50% off on AI Gateway",
      "url": "https://vercel.com/changelog/minimax-h3-and-h3-max-are-50-off-on-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-08-30T00:00:00+00:00",
      "summary": "Vercel is halving AI Gateway charges for MiniMax H3 and H3 Max through September 13; existing model IDs receive the discount without code changes."
    },
    {
      "title": "Agentic Sites: Building Hyper Personalized Websites — Carlos Sanchez, Adobe",
      "url": "https://www.youtube.com/watch?v=jebp4V0vh30",
      "source_name": "AI Engineer",
      "published_at": "2026-08-29T17:00:17+00:00",
      "summary": "Adobe’s prototype assembles intent-specific page blocks from existing site content in roughly a second, making model latency and per-site evaluation part of frontend architecture."
    },
    {
      "title": "Muse Image now available on AI Gateway",
      "url": "https://vercel.com/changelog/muse-image-now-available-on-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-08-26T00:00:00+00:00",
      "summary": "Muse Image is available through Vercel AI Gateway for both generation and instruction-based editing, including reference-image guidance through the AI SDK."
    },
    {
      "title": "Gemini 3.5 Transcribe now available on AI Gateway",
      "url": "https://vercel.com/changelog/gemini-3-5-transcribe-now-available-on-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-08-26T00:00:00+00:00",
      "summary": "Gemini 3.5 Transcribe adds batch and live WebSocket transcription to AI Gateway, with automatic language detection, 85+ languages, and custom vocabulary."
    },
    {
      "title": "Wan 3.0 now available on AI Gateway",
      "url": "https://vercel.com/changelog/wan-3-0-now-available-on-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-08-25T00:00:00+00:00",
      "summary": "Wan 3.0 gives AI Gateway one video model ID for text, image, frame, and reference workflows, with async renders up to 30 seconds at 1080p and synchronized audio."
    },
    {
      "title": "Re$^3$Cap: Retrieval-Guided Refinement for Image Captioning Enhancement via Reinforcement Learning",
      "url": "https://arxiv.org/abs/2608.21305v1",
      "source_name": "arXiv",
      "published_at": "2026-08-21T17:07:41+00:00",
      "summary": "Re³Cap uses multimodal retrieval to find caption omissions and hallucinations before refinement, offering a concrete retrieval-and-review pattern for vision agents."
    },
    {
      "title": "DeepSeek V4 Flash Vision Experimental now available on AI Gateway",
      "url": "https://vercel.com/changelog/deepseek-v4-flash-with-vision-now-available-on-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-08-21T00:00:00+00:00",
      "summary": "DeepSeek V4 Flash Vision adds screenshot, image, and chart input to Vercel AI Gateway while retaining tool use, reasoning, and caching, but its experimental ID signals production risk."
    },
    {
      "title": "Fish Audio models now available on Vercel AI Gateway for free",
      "url": "https://vercel.com/changelog/fish-audio-models-now-available-on-ai-gateway-for-free",
      "source_name": "Vercel",
      "published_at": "2026-08-19T00:00:00+00:00",
      "summary": "Vercel AI Gateway added four Fish Audio models for speech generation and transcription, with AI SDK 7 support and a free window whose model naming determines later billing."
    },
    {
      "title": "The Next Medium: Why Real-Time Interactive Video Changes Everything — Ahmed Ahres, Reactor",
      "url": "https://www.youtube.com/watch?v=5dCAmSDOAjI",
      "source_name": "AI Engineer",
      "published_at": "2026-08-18T17:30:18+00:00",
      "summary": "Reactor frames real-time video as a programmable session rather than a generated file, enabling interactive worlds and live editing but exposing hard state, latency, and evaluation problems."
    },
    {
      "title": "Generative Video at the Speed of Light — Keegan McCallum, uRun",
      "url": "https://www.youtube.com/watch?v=Xln-On3syJk",
      "source_name": "AI Engineer",
      "published_at": "2026-08-18T16:30:29+00:00",
      "summary": "Real-time video models are becoming cheap and responsive enough for agent interfaces, but builders still need global GPU routing, streaming infrastructure, and multi-model orchestration."
    },
    {
      "title": "Voice agents with Realtime Video — Sidney Primas, LemonSlice",
      "url": "https://www.youtube.com/watch?v=z1dqv74SpUs",
      "source_name": "AI Engineer",
      "published_at": "2026-08-18T16:00:06+00:00",
      "summary": "LemonSlice’s avatar stack treats long-running visual stability, audio-conditioned emotion, and deterministic action timing as the core engineering problems beyond lip sync."
    },
    {
      "title": "Training Krea 2: What matters in generative model training — Sangwu Lee, Krea.ai",
      "url": "https://www.youtube.com/watch?v=-tviRdpmHvs",
      "source_name": "AI Engineer",
      "published_at": "2026-08-18T14:00:06+00:00",
      "summary": "Krea 2’s training notes put data curation and iteration speed ahead of architecture novelty, with explicit safeguards against filtering away unusual visual styles."
    },
    {
      "title": "Grok Imagine Image 2.0 now available on Vercel AI Gateway",
      "url": "https://vercel.com/changelog/grok-imagine-image-2-0-preview-now-available-on-vercel-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-08-08T00:00:00+00:00",
      "summary": "Vercel AI Gateway now exposes xAI’s image model through the AI SDK, including 1K/2K generation, batches, and targeted edits that aim to preserve untouched details."
    },
    {
      "title": "SABRE: Scalable and Automated Benchmarking of VLMs under Stress",
      "url": "https://arxiv.org/abs/2608.07435v1",
      "source_name": "arXiv",
      "published_at": "2026-08-07T17:21:04+00:00",
      "summary": "SABRE turns a Markdown test design into generated VLM stress tests, then filters and repairs candidates. It offers a repeatable pattern for refreshing evals as models improve."
    },
    {
      "title": "Seedance 2.5 now available on Vercel AI Gateway",
      "url": "https://vercel.com/changelog/seedance-2-5-now-available-on-vercel-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-08-06T00:00:00+00:00",
      "summary": "Seedance 2.5 adds multimodal video generation and local edits to AI Gateway, with clips up to 30 seconds and separate image and video references."
    },
    {
      "title": "Gadgets: Personal app vibe coding that is actually safe — Kenton Varda, Cloudflare",
      "url": "https://www.youtube.com/watch?v=RmS5s6Wbin4",
      "source_name": "AI Engineer",
      "published_at": "2026-08-05T22:42:10+00:00",
      "summary": "Kenton Varda argues that personal AI-generated apps need per-user code and strong isolation, not one server-owned version. The demo shows agents modifying app code inside a constrained local runtime."
    },
    {
      "title": "Agogic: Performance-Timed Music Tokens for LLM-Native Text-to-Symbolic-Music Generation",
      "url": "https://arxiv.org/abs/2608.03999v1",
      "source_name": "arXiv",
      "published_at": "2026-08-04T17:56:49+00:00",
      "summary": "Controlled text-to-MIDI tests find token representation matters more than a 34× model-size increase for distributional fidelity; performance timing also beats beat-grid tokenization."
    },
    {
      "title": "MiniMax H3 now available on AI Gateway",
      "url": "https://vercel.com/changelog/minimax-h3-now-available-on-vercel-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-07-30T00:00:00+00:00",
      "summary": "MiniMax H3 brings short 2K video generation to Vercel AI Gateway, with text, keyframe, and multimodal reference inputs. Reference and keyframe modes cannot be combined."
    },
    {
      "title": "Grok Voice Think Fast 2.0 now available on AI Gateway",
      "url": "https://vercel.com/changelog/grok-voice-think-fast-2-0-now-available-on-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-07-29T00:00:00+00:00",
      "summary": "Grok Voice Think Fast 2.0 brings speech-to-speech reasoning and earlier tool calls to Vercel’s realtime API, with server-minted tokens keeping gateway keys off clients."
    },
    {
      "title": "MMOE: Modernizing Diffusion Transformers with Efficient Expert Design",
      "url": "https://arxiv.org/abs/2607.24665v1",
      "source_name": "arXiv",
      "published_at": "2026-07-27T17:05:04+00:00",
      "summary": "ModernMOE applies efficient expert-routing patterns from LLMs to diffusion transformers, improving convergence and quality-cost balance without relying only on larger parameter counts."
    },
    {
      "title": "Evaling Video Slop — Maor Bril, Character.ai",
      "url": "https://www.youtube.com/watch?v=b_PmGocP4rc",
      "source_name": "AI Engineer",
      "published_at": "2026-07-25T00:00:02+00:00",
      "summary": "Video evaluators can reward polish while missing frozen action, broken physics, or failed storytelling. Builders need time-aware criteria and human-calibrated data, not frame quality alone."
    },
    {
      "title": "Building Closed-Loop Evals for a Multimodal Agent at Scale — Soumya Gupta & Jai Chopra, Uber",
      "url": "https://www.youtube.com/watch?v=31GUkCBD-Uc",
      "source_name": "AI Engineer",
      "published_at": "2026-07-24T22:00:25+00:00",
      "summary": "Uber’s image-editing agent uses routing, iterative QA, golden-set gates, and production feedback to avoid costly edits, hallucinated food, and quality regressions."
    },
    {
      "title": "MedGame: Storytelling Gamification Empowered by Large Language Models for Medical Education",
      "url": "https://arxiv.org/abs/2607.21570v1",
      "source_name": "arXiv",
      "published_at": "2026-07-23T17:50:28+00:00",
      "summary": "MedGame turns static clinical cases into executable decision stories with separate narrative and orchestration stages, a useful architecture pattern for case-grounded learning agents."
    },
    {
      "title": "Audio-Native Speech Recognition with a Frozen Discrete-Diffusion Language Model",
      "url": "https://arxiv.org/abs/2607.13013v1",
      "source_name": "arXiv",
      "published_at": "2026-07-14T17:53:22+00:00",
      "summary": "A frozen diffusion language model can transcribe speech by refining the full transcript in parallel. The prototype trains a small audio interface and reaches 6.6% WER in roughly eight steps."
    },
    {
      "title": "Seedream 5.0 Pro is now available on AI Gateway",
      "url": "https://vercel.com/changelog/seedream-5-0-pro-is-now-available-on-ai-gateway",
      "source_name": null,
      "published_at": null,
      "summary": "Seedream 5.0 Pro adds image generation and editing to Vercel AI Gateway, targeting reliable text rendering and dense infographic layouts through the AI SDK."
    },
    {
      "title": "Evidence-Backed Video Question Answering",
      "url": "https://arxiv.org/abs/2607.11862v1",
      "source_name": null,
      "published_at": null,
      "summary": "E-VQA requires video answers to include tracked pixel-level evidence, revealing when good QA scores hide weak perception and supplying grounded training data."
    },
    {
      "title": "calesthio/OpenMontage",
      "url": "https://github.com/calesthio/OpenMontage",
      "source_name": "GitHub",
      "published_at": null,
      "summary": "OpenMontage turns coding agents into pipeline-driven video producers, covering research through rendering with approval gates, provider selection, and auditable costs."
    },
    {
      "title": "Encoder-Side Neuron Identification and Amplification for Acoustic Perception in Large Audio-Language Models",
      "url": "https://arxiv.org/abs/2607.11801v1",
      "source_name": null,
      "published_at": null,
      "summary": "IAAN boosts selected audio-encoder neurons at inference, improving fine-grained speech perception across three models without retraining or labels."
    },
    {
      "title": "Search Beyond What Can Be Taught: Evolving the Knowledge Boundary in Agentic Visual Generation",
      "url": "https://arxiv.org/abs/2607.05382v1",
      "source_name": null,
      "published_at": null,
      "summary": "SearchGen-Bench shows open image generators score 21–28/100 on long-tail entities, and naive search retrieval only adds noise; a teach-then-search co-training recipe learns when to retrieve versus rely on weights."
    },
    {
      "title": "Meta 3D AssetGen: Generating 3D Worlds With AI",
      "url": "https://engineering.fb.com/2025/09/29/virtual-reality/assetgen-generating-3d-worlds-with-ai/",
      "source_name": null,
      "published_at": null,
      "summary": "Meta's Tech Podcast covers AssetGen, its foundation model for generating 3D assets from text, and the path toward AI-generated worlds in Horizon Studio. A podcast episode, so light on specifics."
    },
    {
      "title": "Start building with Nano Banana 2 Lite and Gemini Omni Flash",
      "url": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-omni-flash-nano-banana-2-lite/",
      "source_name": null,
      "published_at": null,
      "summary": "Two new Gemini API models: Nano Banana 2 Lite generates 1K images in ~4s at $0.034 each, and Omni Flash does video at $0.10/sec in public preview — cheap enough to wire asset generation into agent pipelines."
    },
    {
      "title": "The latest AI news we announced in June 2026",
      "url": "https://blog.google/innovation-and-ai/technology/ai/google-ai-updates-june-2026/",
      "source_name": null,
      "published_at": null,
      "summary": "Google's June roundup: Gemma 4 12B runs locally in 16GB of memory, Gemini 3.5 Flash adds computer use for desktop, mobile, and browser agents, and Nano Banana 2 Lite ships as a cheaper image model."
    },
    {
      "title": "Comfy-Org/ComfyUI",
      "url": "https://github.com/Comfy-Org/ComfyUI",
      "source_name": "GitHub",
      "published_at": null,
      "summary": "ComfyUI turns multimodal generation into reusable node graphs with API access, incremental execution, and offline operation. Pin stable releases if custom nodes matter to your workflow."
    }
  ]
}