{
  "schema_version": "1.1",
  "id": "archive:https://www.youtube.com/watch?v=-tviRdpmHvs",
  "slug": "training-krea-2-what-matters-in-generative-model-training-sangwu-lee-kre-0m4n8q5",
  "url": "https://feed7.dev/p/training-krea-2-what-matters-in-generative-model-training-sangwu-lee-kre-0m4n8q5",
  "title": "Training Krea 2: What matters in generative model training — Sangwu Lee, Krea.ai",
  "why_included": "Krea 2’s training notes put data curation and iteration speed ahead of architecture novelty, with explicit safeguards against filtering away unusual visual styles.",
  "summary": "Krea trained Krea 2 on **2–10 billion images**, using basic hashes before semantic deduplication and roughly **30–40 in-house filters**. Its pipeline progressed from **256px to 1K** before mid-training, supervised tuning, preference optimization, and reinforcement learning.",
  "practical_implication": "For image systems, treat dataset composition as a product decision. Audit aesthetic filters for style collapse, distill expensive vision-model judgments into cheap classifiers, and keep the training stack simple enough to test changes quickly.",
  "agent_context": "Krea trained Krea 2 on **2–10 billion images**, using basic hashes before semantic deduplication and roughly **30–40 in-house filters**. Its pipeline progressed from **256px to 1K** before mid-training, supervised tuning, preference optimization, and reinforcement learning.\n\nFor image systems, treat dataset composition as a product decision. Audit aesthetic filters for style collapse, distill expensive vision-model judgments into cheap classifiers, and keep the training stack simple enough to test changes quickly.\n\nThese are research notes from one model rather than controlled comparisons of every lever. More reliable outputs may still trade away diversity, and the talk does not quantify Krea 2’s gains against alternatives.",
  "source": {
    "name": "AI Engineer",
    "url": "https://www.youtube.com/watch?v=-tviRdpmHvs",
    "published_at": "2026-08-18T14:00:06.000Z"
  },
  "source_class": "video",
  "content_type": "Video",
  "layer": "model",
  "domains": [
    "image"
  ],
  "topics": [
    "generative-media",
    "open-models"
  ],
  "verification": {
    "status": "source_linked",
    "label": "Source Linked",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "These are research notes from one model rather than controlled comparisons of every lever. More reliable outputs may still trade away diversity, and the talk does not quantify Krea 2’s gains against alternatives."
  ],
  "connected_context": null,
  "lifecycle": "Current",
  "published_at": "2026-08-18T14:00:06.000Z",
  "modified_at": "2026-08-18T14:00:06.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/training-krea-2-what-matters-in-generative-model-training-sangwu-lee-kre-0m4n8q5",
    "json": "https://feed7.dev/p/training-krea-2-what-matters-in-generative-model-training-sangwu-lee-kre-0m4n8q5.json",
    "markdown": "https://feed7.dev/p/training-krea-2-what-matters-in-generative-model-training-sangwu-lee-kre-0m4n8q5.md"
  }
}