@InProceedings{murugaraj-EtAl:2026:latell,
  author    = {Murugaraj, Keerthana  and  Tebourbi, Hedi  and  Friezas Gonçalves, Christophe  and  Lamsiyah, Salima},
  title     = {LUXDIAG-RAG: Diagnostic Evaluation of Retrieval-Augmented Generation for Luxembourgish Reading Comprehension},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {161--171},
  abstract  = {Retrieval-augmented generation (RAG) is widely used to ground large language model (LLM) outputs in external evidence, but its evaluation remains concentrated on high-resource languages. We present LUXDIAG-RAG, a diagnostic evaluation framework for Luxembourgish retrieval-augmented reading comprehension using LuxDiagRC, a corpus of 640 multiple-choice questions over 16 annotated texts. We compare closed-book, full-text oracle, text-restricted RAG, and open-corpus RAG settings to separate answerability without context, comprehension with gold context, evidence selection, and document-plus-evidence retrieval. We evaluate lexical, dense, and hybrid retrievers with four LLMs and report answer accuracy, source-text recall, span-level evidence recall, distractor-span retrieval, diagnostic annotation-based analyses, and targeted human evaluation of answer correctness and evidence sufficiency. Results show that LLMs can use Luxembourgish context effectively when the full passage is provided, with oracle accuracy between 0.81 and 0.85. RAG accuracy improves with larger retrieved contexts and approaches oracle performance in the text-restricted setting, while open-corpus retrieval remains a bottleneck. Character-level lexical and weighted hybrid retrieval outperforms off-the-shelf multilingual dense retrieval for opencorpus evidence selection. Overall, LUXDIAG-RAG provides a reproducible protocol for diagnosing retrieval and comprehension failures in low-resource RAG evaluation.},
  url       = {https://aclanthology.org/2026.latell-1.18}
}

