{
  "schema_version": "1.1",
  "id": "atlas-generative-media",
  "slug": "generative-media",
  "title": "Generative Media",
  "url": "https://feed7.dev/atlas/generative-media",
  "current_answer": null,
  "implementation_consequence": null,
  "agent_context": null,
  "confidence": "auto_collected",
  "last_verified": null,
  "last_updated": null,
  "evidence": [],
  "conflicting_sources": [],
  "superseded_claims": [],
  "corpus_evidence": [
    {
      "title": "MiniMax H3 now available on AI Gateway",
      "url": "https://vercel.com/changelog/minimax-h3-now-available-on-vercel-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-07-30T00:00:00+00:00",
      "summary": "MiniMax H3 brings short 2K video generation to Vercel AI Gateway, with text, keyframe, and multimodal reference inputs. Reference and keyframe modes cannot be combined."
    },
    {
      "title": "Grok Voice Think Fast 2.0 now available on AI Gateway",
      "url": "https://vercel.com/changelog/grok-voice-think-fast-2-0-now-available-on-ai-gateway",
      "source_name": "Vercel",
      "published_at": "2026-07-29T00:00:00+00:00",
      "summary": "Grok Voice Think Fast 2.0 brings speech-to-speech reasoning and earlier tool calls to Vercel’s realtime API, with server-minted tokens keeping gateway keys off clients."
    },
    {
      "title": "MMOE: Modernizing Diffusion Transformers with Efficient Expert Design",
      "url": "https://arxiv.org/abs/2607.24665v1",
      "source_name": "arXiv",
      "published_at": "2026-07-27T17:05:04+00:00",
      "summary": "ModernMOE applies efficient expert-routing patterns from LLMs to diffusion transformers, improving convergence and quality-cost balance without relying only on larger parameter counts."
    },
    {
      "title": "Evaling Video Slop — Maor Bril, Character.ai",
      "url": "https://www.youtube.com/watch?v=b_PmGocP4rc",
      "source_name": "AI Engineer",
      "published_at": "2026-07-25T00:00:02+00:00",
      "summary": "Video evaluators can reward polish while missing frozen action, broken physics, or failed storytelling. Builders need time-aware criteria and human-calibrated data, not frame quality alone."
    },
    {
      "title": "Building Closed-Loop Evals for a Multimodal Agent at Scale — Soumya Gupta & Jai Chopra, Uber",
      "url": "https://www.youtube.com/watch?v=31GUkCBD-Uc",
      "source_name": "AI Engineer",
      "published_at": "2026-07-24T22:00:25+00:00",
      "summary": "Uber’s image-editing agent uses routing, iterative QA, golden-set gates, and production feedback to avoid costly edits, hallucinated food, and quality regressions."
    },
    {
      "title": "MedGame: Storytelling Gamification Empowered by Large Language Models for Medical Education",
      "url": "https://arxiv.org/abs/2607.21570v1",
      "source_name": "arXiv",
      "published_at": "2026-07-23T17:50:28+00:00",
      "summary": "MedGame turns static clinical cases into executable decision stories with separate narrative and orchestration stages, a useful architecture pattern for case-grounded learning agents."
    },
    {
      "title": "Audio-Native Speech Recognition with a Frozen Discrete-Diffusion Language Model",
      "url": "https://arxiv.org/abs/2607.13013v1",
      "source_name": "arXiv",
      "published_at": "2026-07-14T17:53:22+00:00",
      "summary": "A frozen diffusion language model can transcribe speech by refining the full transcript in parallel. The prototype trains a small audio interface and reaches 6.6% WER in roughly eight steps."
    },
    {
      "title": "Seedream 5.0 Pro is now available on AI Gateway",
      "url": "https://vercel.com/changelog/seedream-5-0-pro-is-now-available-on-ai-gateway",
      "source_name": null,
      "published_at": null,
      "summary": "Seedream 5.0 Pro adds image generation and editing to Vercel AI Gateway, targeting reliable text rendering and dense infographic layouts through the AI SDK."
    },
    {
      "title": "Evidence-Backed Video Question Answering",
      "url": "https://arxiv.org/abs/2607.11862v1",
      "source_name": null,
      "published_at": null,
      "summary": "E-VQA requires video answers to include tracked pixel-level evidence, revealing when good QA scores hide weak perception and supplying grounded training data."
    },
    {
      "title": "Search Beyond What Can Be Taught: Evolving the Knowledge Boundary in Agentic Visual Generation",
      "url": "https://arxiv.org/abs/2607.05382v1",
      "source_name": null,
      "published_at": null,
      "summary": "SearchGen-Bench shows open image generators score 21–28/100 on long-tail entities, and naive search retrieval only adds noise; a teach-then-search co-training recipe learns when to retrieve versus rely on weights."
    },
    {
      "title": "Meta 3D AssetGen: Generating 3D Worlds With AI",
      "url": "https://engineering.fb.com/2025/09/29/virtual-reality/assetgen-generating-3d-worlds-with-ai/",
      "source_name": null,
      "published_at": null,
      "summary": "Meta's Tech Podcast covers AssetGen, its foundation model for generating 3D assets from text, and the path toward AI-generated worlds in Horizon Studio. A podcast episode, so light on specifics."
    },
    {
      "title": "Start building with Nano Banana 2 Lite and Gemini Omni Flash",
      "url": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-omni-flash-nano-banana-2-lite/",
      "source_name": null,
      "published_at": null,
      "summary": "Two new Gemini API models: Nano Banana 2 Lite generates 1K images in ~4s at $0.034 each, and Omni Flash does video at $0.10/sec in public preview — cheap enough to wire asset generation into agent pipelines."
    },
    {
      "title": "Encoder-Side Neuron Identification and Amplification for Acoustic Perception in Large Audio-Language Models",
      "url": "https://arxiv.org/abs/2607.11801v1",
      "source_name": null,
      "published_at": null,
      "summary": "IAAN boosts selected audio-encoder neurons at inference, improving fine-grained speech perception across three models without retraining or labels."
    },
    {
      "title": "The latest AI news we announced in June 2026",
      "url": "https://blog.google/innovation-and-ai/technology/ai/google-ai-updates-june-2026/",
      "source_name": null,
      "published_at": null,
      "summary": "Google's June roundup: Gemma 4 12B runs locally in 16GB of memory, Gemini 3.5 Flash adds computer use for desktop, mobile, and browser agents, and Nano Banana 2 Lite ships as a cheaper image model."
    }
  ]
}