{
  "schema_version": "1.1",
  "id": "s13:https://arxiv.org/abs/2609.11892v1",
  "slug": "2609-11892v1-0opi07x",
  "url": "https://feed7.dev/p/2609-11892v1-0opi07x",
  "title": "Nuha-Speech: Building General-Purpose Arabic Speech-LLMs",
  "why_included": "Nuha-Speech adds an Arabic speech-QA corpus, Qwen-Omni fine-tuning, and a tailored evaluation framework—a useful blueprint for adapting speech models where language resources are scarce.",
  "summary": "Nuha-Speech combines an Arabic speech question-answering corpus with supervised fine-tuning and evaluation. The corpus contains **over 1.5 million training samples**, and training uses **Qwen-Omni variants** at multiple scales.",
  "practical_implication": "Builders adapting speech agents to underrepresented languages should treat data construction, model tuning, and task-specific evaluation as one pipeline. Broad instruction coverage matters when existing speech resources are limited.",
  "agent_context": "Nuha-Speech combines an Arabic speech question-answering corpus with supervised fine-tuning and evaluation. The corpus contains **over 1.5 million training samples**, and training uses **Qwen-Omni variants** at multiple scales.\n\nBuilders adapting speech agents to underrepresented languages should treat data construction, model tuning, and task-specific evaluation as one pipeline. Broad instruction coverage matters when existing speech resources are limited.\n\nThe abstract provides no benchmark results, licensing details, dialect coverage, or release terms. It therefore establishes the initiative’s scope, not how well the resulting models generalize or compare with alternatives.",
  "source": {
    "name": "arXiv",
    "url": "https://arxiv.org/abs/2609.11892v1",
    "published_at": "2026-09-10T17:50:35.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Paper",
  "layer": "model",
  "domains": [
    "audio"
  ],
  "topics": [],
  "verification": {
    "status": "needs_review",
    "label": "Needs Review",
    "method": "unverified",
    "verified_at": null
  },
  "uncertainty": [
    "The abstract provides no benchmark results, licensing details, dialect coverage, or release terms. It therefore establishes the initiative’s scope, not how well the resulting models generalize or compare with alternatives."
  ],
  "connected_context": {
    "meaning": "This expands the speech-model landscape with an Arabic-focused data-and-tuning pipeline rather than a demonstrated model-selection result. Against candidates centered on realtime products, transcription, or training techniques, it emphasizes that underrepresented-language coverage begins with corpus construction and task-specific evaluation. Missing scores, dialect coverage, licensing, and release terms prevent judging generalization or deployability.",
    "corpus_size": 757,
    "generated_at": "2026-09-12T10:06:52.849Z",
    "connections": [
      {
        "title": "X$^3$-OPD: Distilling Reasoning into Large Audio-Language Models via On-Policy Alignment",
        "source_name": "arXiv",
        "source_url": "https://arxiv.org/abs/2607.21550v1",
        "feed7_url": "https://feed7.dev/p/2607-21550v1-1j7d28n",
        "reason": "X³-OPD offers a method for transferring reasoning into audio models, whereas this Signal addresses the complementary prerequisite of building language-specific speech data and evaluations for adaptation."
      },
      {
        "title": "Gemini 3.5 Transcribe now available on AI Gateway",
        "source_name": "Vercel",
        "source_url": "https://vercel.com/changelog/gemini-3-5-transcribe-now-available-on-ai-gateway",
        "feed7_url": "https://feed7.dev/p/gemini-3-5-transcribe-now-available-on-ai-gateway-08us8jk",
        "reason": "Gemini provides a deployable multilingual transcription route, but its missing accuracy evidence and this Signal’s missing benchmark results leave Arabic workload comparison unresolved and requiring task-specific testing."
      },
      {
        "title": "Hugging Face and Cerebras bring Gemma 4 to real-time voice AI",
        "source_name": "huggingface.co",
        "source_url": "https://huggingface.co/blog/cerebras-gemma4-voice-ai",
        "feed7_url": "https://feed7.dev/p/cerebras-gemma4-voice-ai-1v86fff",
        "reason": "The open speech-to-speech pipeline shows how existing components can be assembled for deployment; Nuha-Speech instead concentrates on Arabic corpus creation and supervised tuning, with release terms still unknown."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-10T17:50:35.000Z",
  "modified_at": "2026-09-10T17:50:35.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/2609-11892v1-0opi07x",
    "json": "https://feed7.dev/p/2609-11892v1-0opi07x.json",
    "markdown": "https://feed7.dev/p/2609-11892v1-0opi07x.md"
  }
}