@InProceedings{ahmed-EtAl:2026:latell,
  author    = {Ahmed, Zehra  and  Inayat, Farah  and  Aqib, Zuha  and  Haider, Sajjad},
  title     = {Enhancing Urdu ASR with Whisper v3: Fine-Tuning on Latest Datasets and Realistic Multi-Speaker Evaluation with SLM Post-Processing},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {408--417},
  abstract  = {Automatic Speech Recognition (ASR) for low-resource languages like Urdu remains challenging due to limited training data, dialectal variation, code-switching, and orthographic inconsistencies. This work evaluates OpenAI Whisper large-v3-Turbo for Urdu ASR and performs Low Rank Adaption (LoRA) fine-tuning on Common Voice v23, FLEURS and CSaLT datasets, drawn from a pooled corpus of 31 hours of Urdu speech. To gauge robustness in the real world, we also created an evaluation set from YouTube that includes news segments with two people, namely an anchor and an on-site reporter, with natural accents, interruptions, and background noise. Apart from this, the paper explored Urdu-capable Small Language Models (SLMs) between 3 to 14 billion parameters for generative error correction (GEC). Whisper-Turbo achieved 25.33\% WER on the evaluation set from YouTube, whereas Whisper-Turbo with Gemma3-12B 4-bit Quantized as SLM GEC attained 21.75\% WER, which translates to a 14.14\% relative reduction. The results indicate that there is a potential application of SLM post-processing for long-form Urdu transcription, at the cost of extra runtime.},
  url       = {https://aclanthology.org/2026.latell-1.42}
}

