{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.10445v1",
  "slug": "2609-10445v1-0tad2cs",
  "url": "https://feed7.dev/p/2609-10445v1-0tad2cs",
  "title": "Building Multilingual Bridges: Data Mixing as the Pillar of Generalization for In-Language Reasoning",
  "why_included": "Tiny Aya L2-Thinker shows multilingual reasoning can transfer through data mixing, suggesting builders should evaluate whether agents reason in the user's language, not only answer in it.",
  "summary": "Tiny Aya L2-Thinker is a **3.35B-parameter** model trained through a data-centric SFT recipe. It reports an in-language reasoning rate above **93%** across **60 languages** and **6 benchmarks**, covering math, commonsense, instructions, open generation, and cultural reasoning.",
  "practical_implication": "For multilingual agents, test the language of intermediate reasoning as well as final answers. The reported recipe combines broad language coverage, multilingual non-reasoning data, and a sufficient English reasoning base rather than requiring reasoning supervision for every target language.",
  "agent_context": "Tiny Aya L2-Thinker is a **3.35B-parameter** model trained through a data-centric SFT recipe. It reports an in-language reasoning rate above **93%** across **60 languages** and **6 benchmarks**, covering math, commonsense, instructions, open generation, and cultural reasoning.\n\nFor multilingual agents, test the language of intermediate reasoning as well as final answers. The reported recipe combines broad language coverage, multilingual non-reasoning data, and a sufficient English reasoning base rather than requiring reasoning supervision for every target language.\n\nThe material gives aggregate coverage but no language-level scores, failure cases, or comparisons. Claims about transfer to held-out languages therefore need validation on the exact languages and agent tasks a product serves.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.10445v1",
    "published_at": "2026-09-09T16:56:25.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "model",
  "domains": [
    "research"
  ],
  "topics": [
    "reasoning",
    "open-models"
  ],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The material gives aggregate coverage but no language-level scores, failure cases, or comparisons. Claims about transfer to held-out languages therefore need validation on the exact languages and agent tasks a product serves."
  ],
  "connected_context": {
    "meaning": "This makes training-data composition, rather than per-language reasoning supervision, the central lever for compact multilingual reasoning. It also adds a stricter product evaluation requirement: verify the language used in intermediate reasoning, not merely the final response. The aggregate result supports broad transfer, but does not yet identify which languages or agent tasks benefit reliably.",
    "corpus_size": 732,
    "generated_at": "2026-09-10T10:09:54.523Z",
    "connections": [
      {
        "title": "Data Quality Is the Compute Multiplier — Ari Morcos, DatologyAI",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=_PdK6x7PQNM",
        "feed7_url": "https://feed7.dev/p/data-quality-is-the-compute-multiplier-ari-morcos-datologyai-0x7k2ve",
        "reason": "It gives concrete multilingual support to the broader claim that balanced, task-shaped data mixtures can improve capability without simply increasing model size or compute."
      },
      {
        "title": "OctoLong: Mid-Training On Cross-Repository Code Contexts Enhances Long-Context Modeling",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.05141v1",
        "feed7_url": "https://feed7.dev/p/2608-05141v1-0kai3a6",
        "reason": "Both make data structure and mixture a model-capability intervention: Tiny Aya targets cross-language reasoning transfer, while OctoLong targets dependency-aware long-context coding."
      },
      {
        "title": "DFM Mimir v1: An Open HRM Delivering Frontier Performance at 1B Parameters Using Only Permissible Post-Training Data",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2608.13517v1",
        "feed7_url": "https://feed7.dev/p/2608-13517v1-10qer54",
        "reason": "It broadens the compact multilingual model evidence beyond Mimir’s Danish-oriented case, while both still require language- and task-level evaluation before deployment comparisons are justified."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-09T16:56:25.000Z",
  "modified_at": "2026-09-09T16:56:25.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-10445v1-0tad2cs",
    "json": "https://feed7.dev/p/2609-10445v1-0tad2cs.json",
    "markdown": "https://feed7.dev/p/2609-10445v1-0tad2cs.md"
  }
}