{
  "schema_version": "1.0",
  "id": "s8:https://www.youtube.com/watch?v=b_PmGocP4rc",
  "slug": "evaling-video-slop-maor-bril-character-ai-0cd76sd",
  "url": "https://feed7.dev/p/evaling-video-slop-maor-bril-character-ai-0cd76sd",
  "title": "Evaling Video Slop — Maor Bril, Character.ai",
  "why_included": "Video evaluators can reward polish while missing frozen action, broken physics, or failed storytelling. Builders need time-aware criteria and human-calibrated data, not frame quality alone.",
  "summary": "A first evaluator gave **9.2 for camera work** to a clip whose camera stayed still for **4 seconds**. Frame-level metrics caught appearance and drift, but missed whether motion, physics, pacing, sound, and story matched the intended video.",
  "practical_implication": "Train and calibrate evaluators around the exact quality axis you need. **Pairwise comparisons** proved easier to align than numeric ratings, and checking short generations before assembly can prevent wasted rendering and editing.",
  "agent_context": "A first evaluator gave **9.2 for camera work** to a clip whose camera stayed still for **4 seconds**. Frame-level metrics caught appearance and drift, but missed whether motion, physics, pacing, sound, and story matched the intended video.\n\nTrain and calibrate evaluators around the exact quality axis you need. **Pairwise comparisons** proved easier to align than numeric ratings, and checking short generations before assembly can prevent wasted rendering and editing.\n\nManufactured defects taught the first model to recognize gloss and artificial artifacts instead of quality. Mixing real and generated footage can instead produce an AI detector, while lip-sync evaluation remains unresolved and human-calibrated judging is slow and costly.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=b_PmGocP4rc",
    "published_at": "2026-07-25T00:00:02.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "benchmark",
  "domains": [
    "video"
  ],
  "topics": [
    "agent-evals",
    "benchmark-integrity",
    "generative-media"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "Manufactured defects taught the first model to recognize gloss and artificial artifacts instead of quality. Mixing real and generated footage can instead produce an AI detector, while lip-sync evaluation remains unresolved and human-calibrated judging is slow and costly."
  ],
  "lifecycle": "Current",
  "published_at": "2026-07-25T00:00:02.000Z",
  "modified_at": "2026-07-25T00:00:02.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/evaling-video-slop-maor-bril-character-ai-0cd76sd",
    "json": "https://feed7.dev/p/evaling-video-slop-maor-bril-character-ai-0cd76sd.json",
    "markdown": "https://feed7.dev/p/evaling-video-slop-maor-bril-character-ai-0cd76sd.md"
  }
}