{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.11870v1",
  "slug": "2609-11870v1-1xbduu5",
  "url": "https://feed7.dev/p/2609-11870v1-1xbduu5",
  "title": "Augustinian BabyLM: What Ostensive Definition Can and Cannot Teach a Small Language Model",
  "why_included": "Visual embedding initialization gave a small language model durable object-property knowledge that most standard benchmarks missed, showing how broad eval suites can hide narrow causal gains.",
  "summary": "A DeBERTa model trained on **10M words** received image-derived embeddings for visually grounded tokens. The effect persisted through training and improved **COMPS in every configuration**, plus a tailored color, material, size, and shape test.",
  "practical_implication": "When testing a new initialization or grounding method, add probes aligned with the injected knowledge and track results per affected token. Aggregate language benchmarks may miss a real but localized capability change.",
  "agent_context": "A DeBERTa model trained on **10M words** received image-derived embeddings for visually grounded tokens. The effect persisted through training and improved **COMPS in every configuration**, plus a tailored color, material, size, and shape test.\n\nWhen testing a new initialization or grounding method, add probes aligned with the injected knowledge and track results per affected token. Aggregate language benchmarks may miss a real but localized capability change.\n\nMost BabyLM benchmarks showed no effect, and gains on the tailored test stayed confined to seeded words. Even lower held-out loss for visually seeded function and abstract words was not captured by any benchmark used, leaving the right evaluation unresolved.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.11870v1",
    "published_at": "2026-09-10T17:43:09.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "benchmark",
  "domains": [
    "research",
    "image"
  ],
  "topics": [
    "benchmark-integrity"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "Most BabyLM benchmarks showed no effect, and gains on the tailored test stayed confined to seeded words. Even lower held-out loss for visually seeded function and abstract words was not captured by any benchmark used, leaving the right evaluation unresolved."
  ],
  "connected_context": {
    "meaning": "This confirms that a real, localized learning effect can remain invisible to broad aggregate benchmarks. It narrows the grounding claim to seeded tokens and aligned probes, so the result supports controlled, capability-specific evaluation rather than general language improvement. The unexplained held-out-loss change also shows that even targeted suites may omit affected behaviors and should not be treated as exhaustive.",
    "corpus_size": 757,
    "generated_at": "2026-09-12T10:06:49.330Z",
    "connections": [
      {
        "title": "LittleLearner: Language Models Under Pedagogically Controlled Knowledge Exposure",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.13545v1",
        "feed7_url": "https://feed7.dev/p/2608-13545v1-1ray1wb",
        "reason": "Both use controlled knowledge exposure to distinguish newly introduced information from general capability, while this work adds token-level probes for effects tied to the intervention."
      },
      {
        "title": "Phantom Gains: Auditing Self-Improvement Against a Measured Null",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.20290v1",
        "feed7_url": "https://feed7.dev/p/2608-20290v1-13o5gau",
        "reason": "Adds a prerequisite for interpreting the localized gains: repeated frozen baselines and a measured null can separate small intervention effects from training and evaluation noise."
      },
      {
        "title": "Molecular Déjà Vu: Digit-Level Retrieval of Published Values in Frontier Language Models",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2609.05381v1",
        "feed7_url": "https://feed7.dev/p/2609-05381v1-1y2qhyz",
        "reason": "Offers the complementary integrity concern that apparent knowledge can reflect prior retrieval; controlled seeding and per-token tracking help establish which exposure produced the measured behavior."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-10T17:43:09.000Z",
  "modified_at": "2026-09-10T17:43:09.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-11870v1-1xbduu5",
    "json": "https://feed7.dev/p/2609-11870v1-1xbduu5.json",
    "markdown": "https://feed7.dev/p/2609-11870v1-1xbduu5.md"
  }
}