{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.04180v1",
  "slug": "2609-04180v1-1krjayr",
  "url": "https://feed7.dev/p/2609-04180v1-1krjayr",
  "title": "Knowledge Acquisition During Pre-training? Large Language Models Learn Better With Auxiliary Views",
  "why_included": "Controlled pre-training experiments suggest varied reformulations can teach facts more efficiently than repeating documents under the same token budget, though paraphrasing gains depend on batch size.",
  "summary": "Controlled experiments found that repetition remains necessary for knowledge acquisition, but shifting a fixed token budget from repeated documents to **auxiliary views** improved learning, including factual recall.",
  "practical_implication": "For builders training or adapting models, data diversity may matter at the representation level: supply contextual or foundational reformulations instead of spending every extra token on duplicates. The generating teacher’s strength was not decisive.",
  "agent_context": "Controlled experiments found that repetition remains necessary for knowledge acquisition, but shifting a fixed token budget from repeated documents to **auxiliary views** improved learning, including factual recall.\n\nFor builders training or adapting models, data diversity may matter at the representation level: supply contextual or foundational reformulations instead of spending every extra token on duplicates. The generating teacher’s strength was not decisive.\n\nThe effect is conditional: paraphrasing helped only at **smaller batch sizes**, and the abstract does not quantify gains or establish how well the recipe transfers to production-scale training.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.04180v1",
    "published_at": "2026-09-03T17:57:02.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "model",
  "domains": [
    "data"
  ],
  "topics": [
    "reasoning"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The effect is conditional: paraphrasing helped only at **smaller batch sizes**, and the abstract does not quantify gains or establish how well the recipe transfers to production-scale training."
  ],
  "connected_context": {
    "meaning": "This turns the broad data-quality case into a narrower training recipe: repetition still matters, but auxiliary reformulations can use a fixed token budget better than duplicates. It supports representation-level diversity while adding an important batch-size dependency and leaving production-scale transfer unresolved.",
    "corpus_size": 691,
    "generated_at": "2026-09-05T10:08:08.498Z",
    "connections": [
      {
        "title": "Data Quality Is the Compute Multiplier — Ari Morcos, DatologyAI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=_PdK6x7PQNM",
        "feed7_url": "https://feed7.dev/p/data-quality-is-the-compute-multiplier-ari-morcos-datologyai-0x7k2ve",
        "reason": "Provides controlled evidence for the broader claim that information value and selective synthesis can matter more than simply adding duplicate training tokens."
      },
      {
        "title": "LittleLearner: Language Models Under Pedagogically Controlled Knowledge Exposure",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.13545v1",
        "feed7_url": "https://feed7.dev/p/2608-13545v1-1ray1wb",
        "reason": "Complements LittleLearner’s controlled exposure framework by testing a specific condition under which repeated exposure produces stronger knowledge acquisition."
      },
      {
        "title": "Rethinking On-Policy Distillation of Large Language Models II: One Training Example",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.04172v1",
        "feed7_url": "https://feed7.dev/p/2609-04172v1-0kofdqa",
        "reason": "Both reduce emphasis on raw example count: this work favors multiple representational views, while OPD favors diverse visited states from few queries."
      },
      {
        "title": "OctoLong: Mid-Training On Cross-Repository Code Contexts Enhances Long-Context Modeling",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.05141v1",
        "feed7_url": "https://feed7.dev/p/2608-05141v1-0kai3a6",
        "reason": "Reinforces the implementation consequence that training diversity should preserve useful structure, whether through auxiliary knowledge views or dependency-linked code contexts."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-03T17:57:02.000Z",
  "modified_at": "2026-09-03T17:57:02.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-04180v1-1krjayr",
    "json": "https://feed7.dev/p/2609-04180v1-1krjayr.json",
    "markdown": "https://feed7.dev/p/2609-04180v1-1krjayr.md"
  }
}