@Book{latell:2026,
  editor    = {Lamsiyah, Salima  and  Ranasinghe, Tharindu  and  Ezzini, Saad  and  Estevanell-Valladares, Ernesto Luis  and  Mitkov, Ruslan},
  title     = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  url       = {https://aclanthology.org/2026.latell-1}
}

@InProceedings{mukusheva-fusco-chesi:2026:latell,
  author    = {Mukusheva, Albina  and  Fusco, Achille  and  Chesi, Cristiano},
  title     = {A Kazakh–Russian Corpus of Child-Directed Language for Low-Resource Languages},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {1--7},
  abstract  = {Morphologically rich and low-resource languages present challenges for tokenization and other natural language processing (NLP) tasks. Standard tokenization algorithms often fail to segment words into meaningful morphological units. At the same time, datasets for many low-resource languages remain limited. We introduce a new dataset of child-directed language in Kazakh and Russian compiled from multiple sources including child–adult dialogue transcripts, fairy tales, cartoons and children's literature texts. The dataset is manually collected and contains 1M tokens in Kazakh and 2.3M tokens in Russian. Dialogue data includes metadata such as child age and speaker identity, enabling research on language acquisition and linguistic development. We describe the process of data collection, corpus organization, and basic statistical properties of the dataset. The corpus provides a new resource for research on low-resource NLP, language acquisition, and morphologically-aware tokenization methods.},
  url       = {https://aclanthology.org/2026.latell-1.1}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0003-1935-1348
@InProceedings{daignaultpichette-lareau:2026:latell,
  author    = {Daignault-Pichette, Loïc  and  Lareau, François},
  title     = {A finite-state model of Innu verbal inflection},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {80--88},
  abstract  = {This article describes a finite-state transducer (FST) of Innu verbal inflection. Unlike approaches that model word formation by concatenating unanalysed inflectional chunks to a root, the proposed model is linguistically motivated and reflects the largely agglutinative nature of Innu verbal morphology. Our implementation relies on a sequence of intermediate representations, making it modular. It covers 19,628 unique forms for 68 lemmas and it has been evaluated against data from the most comprehensive resources currently available. We argue that FSTs can serve not only as a practical natural language processing tool, but also as a means of testing the descriptive adequacy of a morphological analysis. We discuss the challenges encountered in developing this model.},
  url       = {https://aclanthology.org/2026.latell-1.10}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0001-7027-0514
@InProceedings{rabiei-silberztein:2026:latell,
  author    = {Rabiei, Marzieh  and  Silberztein, Max},
  title     = {Morphological Operations in the Persian Verbal System},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {89--95},
  abstract  = {This paper presents a formal computational model of Persian verbal morphology implemented in the NooJ framework. It focuses on the interaction between lexical preverbs, grammatical prefixes, and morphophonological constraints that affect verbal formation. We introduce two operators: <F>, which detects internal lexical boundaries and correctly positions prefixes in compound verbs, and <G>, which models glide insertion in alef-initial stems in prefixal contexts. These mechanisms allow for a structure-sensitive analysis of complex verbal forms that cannot be handled by purely concatenative approaches. The model is evaluated on a contemporary Persian literary corpus and demonstrates high coverage and precision, while also highlighting the impact of orthographic variation, particularly the use of the Zero Width Non-Joiner (ZWNJ).},
  url       = {https://aclanthology.org/2026.latell-1.11}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{gutirrez-EtAl:2026:latell,
  author    = {Gutiérrez, Yoan  and  Consuegra-Ayala, Juan Pablo  and  Sepúlveda-Torres, Robiert  and  Muñoz Guillena, Rafael},
  title     = {A Massive Open-Source Corpus for Valencian: Over 4.7 Billion Tokens for Low-Resource Language Modelling},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {96--101},
  abstract  = {Valencian is a severely underrepresented language in Natural Language Processing (NLP). Building robust foundation models requires massive, high-quality datasets that are currently unavailable for this Romance language variant. In this paper, we present the AITANA Valencian Corpus, to the best of our knowledge the largest open-source dataset curated for Valencian to date. The resource comprises 3.01~billion non-parallel tokens and 1.72~billion parallel tokens (Valencian--Spanish/English), for a total of 4.73~billion tokens across institutional, legislative, and media registers. We detail the acquisition, Markdown-based structuring, and parallel-alignment pipelines, and report corpus analytics covering lexical-diversity metrics and a gender-bias audit. The corpus is publicly released to advance equitable NLP research.},
  url       = {https://aclanthology.org/2026.latell-1.12}
}

Author{1}{Orcid}:https://orcid.org/0000-0002-4052-7427
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:http://orcid.org/0000-0001-8127-9012
@InProceedings{valline-EtAl:2026:latell,
  author    = {Valline, Julian  and  Lothritz, Cedric  and  Guo, Siwen  and  Cabot Sagrera, Jordi},
  title     = {LuxIT: A Luxembourgish Instruction Tuning Dataset from Monolingual Seed Data},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {102--120},
  abstract  = {The effectiveness of instruction-tuned Large Language Models (LLMs) is often limited in low-resource linguistic settings due to a lack of high-quality training data. We introduce LuxIT, a monolingual instruction tuning dataset for Luxembourgish developed to mitigate this challenge. We synthesize the dataset from a corpus of native Luxembourgish texts, utilizing DeepSeek-R1-0528, chosen for its shown proficiency in Luxembourgish. Following generation, we apply a quality assurance process, employing an LLM-as-a-judge approach, retaining 227,507 high-quality instruction-answer pairs. To investigate the practical utility of the dataset, we fine-tune 14 smaller-scale LLMs (≤15B parameters) on LuxIT and evaluate them on standardized Luxembourgish proficiency exams and five downstream NLP tasks. Training on LuxIT yields a mean accuracy change of +5.37 percentage points on language exams across all 14 models, with 12 of 14 showing improvement. On NLP downstream tasks, 9 of 14 models improve in macro-averaged F1, though gains on the two benchmarks do not systematically correlate. These results underscore the feasibility of leveraging monolingual synthetic data to improve LLM capabilities in low-resource languages, while highlighting the multi-faceted nature of language proficiency.},
  url       = {https://aclanthology.org/2026.latell-1.13}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{mekaoui-EtAl:2026:latell,
  author    = {Mekaoui, Salma  and  Chaker, Ilham  and  Zarghili, Arsalane  and  Nikolov, Nikola S.},
  title     = {Topic Modeling for Moroccan Darija: A Comparative Study of Classical Machine Learning Approaches},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {121--130},
  abstract  = {Topic modeling for low-resource dialects such as Moroccan Arabic (Darija) remains largely underexplored, mainly due to the scarcity of linguistic resources and standardized datasets. In this work, we present a comparative study of lightweight topic modeling methods, including Latent Dirichlet Allocation (LDA), Non-negative Matrix Factorization (NMF), and Latent Semantic Analysis (LSA), applied to two datasets of different sizes derived from Moroccan Darija Wikipedia corpora. To identify the optimal number of topics and ensure robust evaluation, we employ multiple complementary metrics, namely coherence C\_v, normalized pointwise mutual information (NPMI), and Topic Diversity. Our results highlight the importance of jointly considering these metrics, as relying on a single measure may lead to suboptimal conclusions. Across all experiments, NMF consistently achieves the best overall performance, reaching a coherence C\_v score of 0.76, an NPMI score of 0.20, and high Topic Diversity across both small and large corpora. To the best of our knowledge, this is among the first studies to systematically compare multiple classical topic modeling methods on native Moroccan Darija text. These findings provide a baseline for future work on dialectal Arabic and pave the way for exploring more advanced neural topic modeling approaches.},
  url       = {https://aclanthology.org/2026.latell-1.14}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{soufiane-EtAl:2026:latell,
  author    = {Soufiane, Mehdi  and  Hammale, Mourad  and  Zerhouni, Kawtar  and  Ahidar-Coutrix, Adil  and  Benouini, Rachid},
  title     = {Distil and Evolve: Compact Adaptive Document Classification with Continual Learning for Low-Resource Settings},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {131--139},
  abstract  = {Document image classification is the entry point of document-processing pipelines, yet the institutions that need them most operate under tight compute and annotation budgets. Knowledge distillation (KD) compresses large document Transformers for such deployment, but whether a heavily distilled model can still undergo continual learning (CL) without catastrophic forgetting, and which CL strategy suits it, is unclear. We distil a DiT-Large teacher (307 M parameters) into a ViT-Tiny student (5.5 M, 21.2 MiB) and evaluate four CL methods in a two-phase class-split protocol on RVL-CDIP across five random seeds. Replay-based methods significantly outperform regularisation-based ones: DER++ retains 90.4 ± 1.2\% of prior accuracy against 75.9 ± 2.1\% for the best regulariser, a 14.5 pp paired difference (p < 10\^{}-4). Every pairwise ordering is significant and holds at every seed. Because hyperparameters were selected on one seed, we also measure the resulting optimism: the selection seed overstates retention by up to 7.4 pp, and most for the weakest method. Per-class analysis shows macro retention conceals concentrated damage, with two of five target classes absorbing most of the loss. We profile each strategy against deployment metrics, including instrumented GPU energy and single-thread CPU latency, and find that counting method preparation shrinks the time advantage of regularisation over replay from 9.1 to 3.3 minutes. A same-size non-distilled control starts near chance and therefore cannot isolate a distillation effect; we report it as a diagnostic only. We release the distilled model, the harness, and all seed-level artifacts.},
  url       = {https://aclanthology.org/2026.latell-1.15}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{ingason-mechler-stefnsdttir:2026:latell,
  author    = {Ingason, Anton Karl  and  Mechler, Johanna  and  Stefánsdóttir, Lilja Björk},
  title     = {Do Large Language Models Style-Shift? Register-Conditioned Stylistic Fronting in AI-Generated Icelandic},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {140--149},
  abstract  = {We examine whether large language models (LLM) encode sociolinguistic knowledge of formal and informal register in a low-resource language, Icelandic, finding that register sensitivity is present in the two systems tested (GPT-5.5 and Gemini 3.1 Pro). Our findings are based on a 2×2 persona design with elicited production (n=1,200), manually coded for Stylistic Fronting, a property of formal language. The direction and size of their style shift are not detectably different, but they differ sharply in overall rate; against a matched human comparator, only GPT-5.5 produces a persona-appropriate rate. Neither system reproduces the gender pattern documented for human speakers. Thus, register sensitivity does not guarantee that output rates are appropriate to the persona the model was instructed to adopt, an issue that may reflect limited training data for low-resource languages.},
  url       = {https://aclanthology.org/2026.latell-1.16}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{zhao-EtAl:2026:latell,
  author    = {Zhao, Xiaojing  and  Lamsiyah, Salima  and  Chersoni, Emmanuele  and  Xu, Han},
  title     = {Benchmarking Large Language Models on Mandarin Proverb Explanation and Contextual Matching},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {150--160},
  abstract  = {Mandarin proverbs condense historical allusions, figurative imagery, and conventionalized pragmatic functions into short expressions, making them a challenging test of culturally grounded language understanding. We evaluate four large language models (LLMs) on the Mandarin subset of the WISDOM dataset through two complementary tasks under zero-shot and few-shot prompting: situation-to-proverb selection, assessed by accuracy, and bilingual proverb explanation, assessed with automatic metrics and human judgments of semantic correctness, cultural faithfulness, clarity, and learner usefulness. The results reveal a clear gap between recognition and explanation. Models achieve high selection accuracy, with GPT-5.4 and DeepSeek-V4-Flash above 98\%. Yet, human evaluation shows that fluent explanations often fail to preserve culturally conventionalized meanings, particularly for proverbs that rest on historical allusions, and are correspondingly less useful for learners. Few-shot demonstrations benefit some weaker models in open-ended explanation but do not consistently help stronger ones. These findings suggest that selection accuracy alone does not capture proverb understanding, and that Mandarin proverbs remain a demanding task for evaluating how LLMs handle figurative and culturally embedded language.},
  url       = {https://aclanthology.org/2026.latell-1.17}
}

Author{1}{Orcid}:https://orcid.org/0009-0009-7325-5366
Author{2}{Orcid}:
Author{3}{Orcid}:https://orcid.org/0000-0001-8742-0451
Author{4}{Orcid}:
@InProceedings{murugaraj-EtAl:2026:latell,
  author    = {Murugaraj, Keerthana  and  Tebourbi, Hedi  and  Friezas Gonçalves, Christophe  and  Lamsiyah, Salima},
  title     = {LUXDIAG-RAG: Diagnostic Evaluation of Retrieval-Augmented Generation for Luxembourgish Reading Comprehension},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {161--171},
  abstract  = {Retrieval-augmented generation (RAG) is widely used to ground large language model (LLM) outputs in external evidence, but its evaluation remains concentrated on high-resource languages. We present LUXDIAG-RAG, a diagnostic evaluation framework for Luxembourgish retrieval-augmented reading comprehension using LuxDiagRC, a corpus of 640 multiple-choice questions over 16 annotated texts. We compare closed-book, full-text oracle, text-restricted RAG, and open-corpus RAG settings to separate answerability without context, comprehension with gold context, evidence selection, and document-plus-evidence retrieval. We evaluate lexical, dense, and hybrid retrievers with four LLMs and report answer accuracy, source-text recall, span-level evidence recall, distractor-span retrieval, diagnostic annotation-based analyses, and targeted human evaluation of answer correctness and evidence sufficiency. Results show that LLMs can use Luxembourgish context effectively when the full passage is provided, with oracle accuracy between 0.81 and 0.85. RAG accuracy improves with larger retrieved contexts and approaches oracle performance in the text-restricted setting, while open-corpus retrieval remains a bottleneck. Character-level lexical and weighted hybrid retrieval outperforms off-the-shelf multilingual dense retrieval for opencorpus evidence selection. Overall, LUXDIAG-RAG provides a reproducible protocol for diagnosing retrieval and comprehension failures in low-resource RAG evaluation.},
  url       = {https://aclanthology.org/2026.latell-1.18}
}

Author{1}{Orcid}:https://orcid.org/0009-0008-5100-055X
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{friezasgonalves-EtAl:2026:latell,
  author    = {Friezas Gonçalves, Christophe  and  Tebourbi, Hedi  and  Schommer, Christoph  and  Lamsiyah, Salima},
  title     = {Assessment of Human-in-the-Loop Multi-LLMs for Low-Resource Educational Data Expansion},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {172--180},
  abstract  = {Low-resource languages often lack the pedagogically grounded datasets needed for language learning and assessment. To address this gap, we assess the potential of large language models (LLMs) to synthetically expand educational datasets in a low-resource language setting. We evaluate LLM performance within a complex, multi-layer annotation framework. Our approach builds upon LuxDiagRC, a Luxembourgish reading comprehension dataset fully annotated by human experts, which serves both as a methodological foundation and as a gold-standard benchmark. The evaluation follows the two-layer design of LuxDiagRC. In the first layer, three LLMs annotate linguistic features in Luxembourgish source texts. In the second layer, the models generate multiple-choice reading comprehension questions grounded in these annotations. Our results indicate that a multi-LLM approach combined with simple human curation outperforms individual and combined model approaches. The framework produces diagnostically grounded reading comprehension questions that closely align with the structure and pedagogical objectives of the original dataset. Overall, the findings suggest that combining multiple LLMs with expert human oversight form a promising strategy for the scalable expansion of educational corpora in a low-resource language setting, while highlighting the continued necessity of human expertise in dataset quality and pedagogical validity.},
  url       = {https://aclanthology.org/2026.latell-1.19}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:0000-0002-0308-7637
Author{4}{Orcid}:
@InProceedings{martnarista:2026:latell,
  author    = {Martín Arista, Javier},
  title     = {Data Curation, Annotation Quality, and Error Patterns in Old English Automatic Lemmatisation},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {8--17},
  abstract  = {Old English (c. 450–1150 CE), preserved in approximately three million words, is a historical language that poses distinctive problems for natural language processing: pervasive spelling variation, rich inflectional morphology, and heterogeneous annotated sources. This paper provides an analysis of the data curation pipeline, annotation quality, and linguistically motivated error patterns for Old English lemmatisation. We describe the extraction, harmonisation, and validation of training data from three historical sources, including a parsed corpus, a scholarly dictionary, and a historical dictionary, each with different formats, POS conventions, and coverage. Quantitative analysis shows that 11.3\% of word types receive inconsistent lemma assignments, identity transformations account for 28.3\% of training data, and the ge- prefix affects 11.7\% of word types. Error analysis organised by morphological category shows that strong-verb ablaut accounts for approximately 25\% of lemmatisation errors, spelling variation for 20\%, and prefix handling for 15\%. We contextualise these findings within the broader landscape of low-resource NLP and formulate practical guidelines for building lemmatisation systems for historical low-resource languages, including recommendations on annotation consistency, POS completeness, and source harmonisation.},
  url       = {https://aclanthology.org/2026.latell-1.2}
}

Author{1}{Orcid}:
@InProceedings{elmoussa:2026:latell,
  author    = {El Moussa, Noura},
  title     = {Human-Centred Approaches in Low-Resource Languages for Educational Applications and Language Learning: A Systematic Literature Review},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {181--190},
  abstract  = {Natural language processing increasingly targets education and human-centred design, yet these advances rarely reach low-resource languages, whose speakers remain underserved on both fronts. Little is known about how often low-resource NLP, human-centred AI, and education genuinely converge rather than being pursued separately. This systematic review asks whether educational NLP tasks extend to low-resource languages, whether progress is matched by fairness, inclusion, accessibility, and human-centred design, and how both converge in practice. We screened 24,795 main-conference papers from five core NLP venues, using keyword matching and a weighted scoring scheme to identify 196 papers with relevant signals. Grammatical error correction has progressed furthest for low-resource languages, while fairness- and accessibility-focused work is mature but rarely paired with broad low-resource coverage. A combined case study confirms that reused multilingual infrastructure can extend support to under-resourced languages.},
  url       = {https://aclanthology.org/2026.latell-1.20}
}

Author{1}{Orcid}:
@InProceedings{friezasgonalves-schommer-lamsiyah:2026:latell,
  author    = {Friezas Gonçalves, Christophe  and  Schommer, Christoph  and  Lamsiyah, Salima},
  title     = {SimuLe-Lux: A Neuro-Symbolic Teaching System for Diagnosing L2 Reading Comprehension},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {191--198},
  abstract  = {Understanding why second-language learners make reading comprehension errors is essential for effective instruction and teaching. However, providing fine-grained diagnostic feedback requires teacher expertise and time. While recent work has explored large language models as student simulators, existing approaches lack grounding in linguistic and educational evidence. We introduce SimuLe-Lux, a neuro-symbolic framework for simulating learner profiles and supporting teacher-facing diagnostic training in Luxembourgish reading comprehension. The system is built on LuxDiagRC and utilizes a large language model to create and simulate learner responses across six CEFR proficiency levels, including weaker-state variants. These responses are aggregated into interpretable diagnostic profiles. A retrieval-augmented teacher-facing agent then uses this structured knowledge base to generate grounded explanations of learner behaviour and instructional guidance. Evaluation results show that the simulated learners exhibit CEFR-consistent error patterns and that the teacher-facing agent produces relevant and informative diagnostic feedback for structured pedagogical queries.},
  url       = {https://aclanthology.org/2026.latell-1.21}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0000-0002-0308-7637
Author{3}{Orcid}:
@InProceedings{atou-EtAl:2026:latell,
  author    = {Atou, Houdaifa  and  Khallaf, Nouran  and  Lamsiyah, Salima  and  Mitkov, Ruslan},
  title     = {Towards Readability Assessment for Under-Resourced Arabic Dialects: A Study of Moroccan Darija},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {199--225},
  abstract  = {Despite recent progress in readability assessment for Modern Standard Arabic (MSA), Moroccan Darija remains largely unexplored. This paper presents the first systematic study of sentence-level readability assessment for Moroccan Darija. To support this study, we introduce Bayan, the first corpus and benchmark for Moroccan Darija readability assessment. It contains 2,030 Arabic-script sentences from seven sources covering journalistic, conversational, informal, spoken, lexical, figurative, and poetic language. Two native speakers independently annotated each sentence using four ordered levels: Immediate (DR1), Accessible (DR2), Demanding (DR3), and Advanced (DR4). We also introduce BayanEx, a subset of 912 sentences annotated for orthographic, lexical, syntactic, and semantic difficulty. Using these resources, we evaluate standalone readability formulae, classical machine learning methods, pretrained encoders, transfer learning from MSA readability models, and large language models (LLMs). MARBERTv2 fine-tuned on Bayan achieves the best overall performance, with 76.3\% accuracy, 75.0\% macro F1, and 0.838 quadratic weighted kappa (QWK). AARI is the strongest standalone formula at 0.740 QWK, Support Vector Regression using readability formula features reaches 0.756, and NileChat-3B achieves the best zero-shot performance, with 0.693. Our transfer learning experiments show that MSA readability pretraining benefits linear probing but does not improve full fine-tuning over a strong dialectal encoder.},
  url       = {https://aclanthology.org/2026.latell-1.22}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{lamsiyah-mitkov:2026:latell,
  author    = {Lamsiyah, Salima  and  Mitkov, Ruslan},
  title     = {Why Is Current XAI Not Enough for Arabic NLP? A Critical Survey of the Explainability Gap},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {226--236},
  abstract  = {Explainable AI (XAI) is now a major theme in NLP; however, Arabic NLP remains under-explained in three connected senses. First, there is a method gap: Arabic XAI relies heavily on a small set of post-hoc techniques such as LIME, SHAP, attention visualization, and saliency, while broader NLP XAI offers richer diagnostic, counterfactual, probing, rationale-based, and human-centered methods. Second, there is a task gap: existing Arabic XAI work is concentrated in classification tasks, especially sentiment analysis, hate/offensive language detection, fake news, and spam, with weaker coverage of generation, retrieval, translation, summarization, structured prediction, and dialogue. Third, there is a linguistic gap: many explanations identify influential tokens, but rarely explain Arabic-specific phenomena such as morphology, clitics, dialectal variation, diglossia, orthographic ambiguity, diacritics, code-switching, named entities, cultural references, or Classical and religious registers. This critical structured survey synthesizes the reviewed literature on Arabic XAI across text, speech, and multimodal settings. We argue that Arabic NLP does not only need explanations of model decisions; it needs explanations that are faithful to Arabic as a linguistic, cultural, and sociotechnical object. We introduce a taxonomy of tasks, methods, linguistic units, varieties, goals, and evaluation practices, and propose a research agenda for linguistically grounded Arabic XAI.},
  url       = {https://aclanthology.org/2026.latell-1.23}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{ghezaielhammouda-alharbi-mitkov:2026:latell,
  author    = {Ghezaiel Hammouda, Nadia  and  Alharbi, Maram I.  and  Mitkov, Ruslan},
  title     = {Benchmarking Speech Foundation Models and LLMs for Sentiment Analysis in Low-Resource Najdi Arabic of Saudi Arabia},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {237--245},
  abstract  = {This paper addresses a gap on two fronts simultaneously in the Najdi dialect: unsupervised speech-based pseudo-labeling and instruction-tuned LLM classification. On the speech side, fourteen audio foundation models—spanning self-supervised learning, audio-language alignment, spectrogram encoding, and ASR—are evaluated through a K-Means pseudo-labeling pipeline applied to 181 recordings (15,377 segments). Cluster quality is assessed via Silhouette Score, Davies-Bouldin Index, Calinski-Harabasz Index, and Shannon entropy; inter-encoder agreement is quantified through pairwise Cohen's κ and Fleiss' κ, treating each encoder as an independent annotator. XLSR-53 leads all geometric metrics by a wide margin (SS = 0.320, DBI = 1.309, CHI = 8660.7), attributable to its Arabic-inclusive pre-training. On the text side, three LLMs; SILMA-9B, LLaMA3-8B, and Qwen2.5-7B—are evaluated across zero-shot, few-shot, LoRA zero-shot, and LoRA few-shot conditions on 181 conversational transcripts (289,888 words). Analysis of prediction distributions across all twelve model–condition pairs reveals that label skew, output-parsing failures, and LoRA adaptation interact differently across architectures: Qwen2.5-7B zero-shot most closely mirrors the ground-truth label distribution, while SILMA-9B suffers from severe Positive bias and low parse rates under all conditions. Together, the results yield a composite encoder ranking—an annotation-free, practical tool for selecting speech models in downstream multimodal fusion pipelines targeting Najdi Arabic and similarly under-resourced Saudi dialects.},
  url       = {https://aclanthology.org/2026.latell-1.24}
}

Author{1}{Orcid}:https://orcid.org/0000-0003-3438-7883
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{elhoubri-EtAl:2026:latell,
  author    = {El Houbri, Fatima Ezzahra  and  Idrissi, Najlae  and  Roche, Mathieu  and  Valentin, Sarah},
  title     = {Spatial Entity Extraction Methods from Arabic Texts in the Context of Epidemiological Surveillance: A Comparative Study},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {246--256},
  abstract  = {The location of events plays a key role in epidemiological surveillance, as it allows epidemic outbreaks to be identified and their spread to be tracked. Several event-based surveillance systems rely on English translation before extracting spatial entities using pre-trained models adapted to English. However, the step of translating Arabic texts prior to extraction can influence the quality of spatial entity recognition. This study compares two strategies for extracting spatial entities from Arabic texts related to epidemiological surveillance evaluated in terms of precision, recall, and F-measure. The first approach relies on machine translation using DeepL, Microsoft Azure, and Reverso, followed by the extraction of entities using the spaCy (small, medium, and large) and GLiNER models. The second strategy consists of directly extracting entities from Arabic texts by fine-tuning the AraBERT, AraELECTRA, ARBERT, and MARBERT models. The results indicate that the second approach offers superior performance, particularly with the AraBERT and AraELECTRA models, demonstrating the effectiveness of direct extraction on Arabic data compared to the translation-based strategy.},
  url       = {https://aclanthology.org/2026.latell-1.25}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{protonina-ignatev:2026:latell,
  author    = {Protonina, Kseniia  and  Ignatev, Daniil},
  title     = {Building an Aligned Speech Corpus for Northern Russian},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {257--263},
  abstract  = {Russian dialects remain underrepresented in speech recognition resources despite phonological and lexical divergence from the standard language. The paper reports the results of a pilot study on constructing a corpus for dialect-aware speech processing. We first make an argument in favor of dialect-aware ASR -- particularly, for Russian. We then describe a reproducible procedure for converting long archival recordings, coarse timestamp indices and manual transcripts into utterance-level aligned data, and introduce a pilot corpus of the Vologda dialect of Russian that is highly scored by expert evaluators. Testing zero-shot ASR models against the corpus reveals that current models remain sensitive to dialectal speech, motivating further expansion of our data collection.},
  url       = {https://aclanthology.org/2026.latell-1.26}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0009-0006-0455-5224
@InProceedings{kanel-EtAl:2026:latell,
  author    = {Kanel, Avery Cole  and  Schuler, Christian  and  Laracha, Bouazza  and  Lbouhli, Imrane  and  Chaouri, Yassine  and  Al Ghussin, Yusser  and  Baumann, Timo},
  title     = {Probing Geographic Performance Gaps in Moroccan ASR with ALAMA: Annotated Local Audio of Moroccan Arabic},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {264--271},
  abstract  = {Moroccan Arabic (Darija) is routinely treated as a single dialect in NLP research, yet it exhibits substantial regional variation across its geography. This pilot study questions the monolithic framing by collecting a regionally annotated speech corpus, 60 minutes of broadcast audio from Casablanca, Tangier, and Oujda, transcribed and verified by native speakers, and evaluating 17 state-of-the-art ASR systems against it. Our results show architecture-dependent performance gaps across regions: systems exhibit significant gaps in WER points between the best- and worst-served regions, and fine-tuned "Moroccan Arabic" checkpoints exhibit inconsistent regional profiles that are hard to reconcile with balanced training data. These findings suggest that national-level dialect labels obscure regionally uneven model coverage with real performance consequences. We release the dataset and support the adoption of more fine-grained provenance labeling as standard metadata for Arabic dialect resources.},
  url       = {https://aclanthology.org/2026.latell-1.27}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0002-2686-6350
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:0000-0003-2203-1783
@InProceedings{ouattara-EtAl:2026:latell,
  author    = {Ouattara, Maimouna  and  Diallo, El-Hacen  and  Philippy, Fred  and  Kaboré, Abdoul Kader  and  Klein, Jacques  and  Bissyandé, Tegawendé F.},
  title     = {Generator-Guided Amount Recovery for Voice-Based Financial Record-Keeping in Mooré-French Code-Switched Speech},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {272--283},
  abstract  = {Voice interfaces promise to bring digital financial services to the roughly 800 million adults worldwide who cannot read, yet informal markets demand more than broad speech-recognition coverage. In Burkina Faso, traders record credit in Mooré and in French–Mooré code-switched speech, where the value of a spoken amount depends on the active counting convention, a historical base-five wakir system in Mooré against base ten in French, so that a correctly transcribed numeral can still encode the wrong sum, and standard cascades and frontier audio Large Language Models (LLMs) fail even when the convention is supplied in context. We present a voice-to-ledger pipeline that recovers the customer, the transaction direction, and the monetary amount from naturalistic market speech. Its core is Generator-Guided Analysis-by-Synthesis (GG-Abs), a deterministic amount-recovery module that treats each spoken amount as a latent integer. GG-Abs proposes candidates with a bidirectional Mooré numeral generator, renders each phonetically, keeps only those that re-synthesise to the noisy transcript, and abstains otherwise, so every accepted amount carries an explicit reconstruction; a conservative LLM judge supplies the customer and direction. We also release a voice-ledger corpus of Mooré–French market recordings spanning four credit intents. On a sealed real-speech holdout, the pipeline attains 76.0\% amount accuracy on Mooré and 85.0\% on code-switched speech, with direction accuracy above 90\% in both regimes, improving Mooré amount recovery over Automatic Speech Recognition (ASR)→LLM and audio-native LLM baselines by 16 to 47 points.},
  url       = {https://aclanthology.org/2026.latell-1.28}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{elbachyr-EtAl:2026:latell,
  author    = {El Bachyr, Omar  and  Philippy, Fred  and  Bernardy, Laura Maria  and  Ezzini, Saad  and  Klein, Jacques  and  Bissyandé, Tegawendé},
  title     = {LëtzCross: A Cross-Lingual Page-Level Benchmark for Multimodal Retrieval over Luxembourgish Documents},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {284--294},
  abstract  = {Recent page-image retrievers such as ColPali have improved retrieval over visually rich documents, yet little is known about how they behave in cross-lingual, low-resource settings. We introduce LëtzCross, a benchmark for cross-lingual page-level retrieval over Luxembourgish PDF documents, with document pages indexed as images and queries provided in English, French, German, and Luxembourgish. The benchmark combines text-focused QA pairs with visually grounded QA pairs, covering both textual and visual retrieval needs in PDF-based RAG. We use LëtzCross to compare OCR-based text-only retrievers with ColPali-style page-image retrievers and find that the latter perform better across query languages in this system-level comparison. We also examine single-language and multilingual fine-tuning. Fine-tuning transfers across query languages, with French yielding the highest mean performance on Luxembourgish queries among the single-language settings. In the multilingual setting, including Luxembourgish gives the strongest results and substantially improves retrieval for Luxembourgish queries.},
  url       = {https://aclanthology.org/2026.latell-1.29}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0001-5842-4285
Author{3}{Orcid}:0009-0002-8307-4576
Author{4}{Orcid}:0000-0001-7657-4738
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{derych-EtAl:2026:latell,
  author    = {Derych, Alicja Helena  and  Alberski, Bartłomiej  and  Jankowski, Hubert  and  Dembowski, Paweł},
  title     = {Creating the RozMuz Corpus: Applying Ethics and Technology in Language Research},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {18--23},
  abstract  = {The paper aims to present work in progress: the corpus Rozmowy przy muzyce – korpus języka mówionego RozMuz ‘Conversations over music – RozMuz speech corpus.' It contains authentic spoken material from instrument lessons (group and individual), recording sessions, and rehearsals of musical groups. RozMuz serves as a subcorpus of the specialized musical corpus being constructed for the purpose of creating the Musical Subwordnet (an experimental resource dependent on plWordNet). The preparation methodology of the spoken subcorpus consists of three stages: 1) recordings followed by file preparation (including anonymization of audio files with pre-recorded labels), 2) their semi-automatic transcription with manual verification, and 3) generation of output documents: a situation description card and a metadata card compliant with Dublin Core.},
  url       = {https://aclanthology.org/2026.latell-1.3}
}

Author{1}{Orcid}:https://orcid.org/0000-0002-4301-1581
Author{2}{Orcid}:https://orcid.org/0000-0002-9099-6627
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{lasmanis:2026:latell,
  author    = {Lasmanis, Viesturs Jūlijs},
  title     = {Improving OCR for a Latvian Pronunciation Dictionary},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {295--303},
  abstract  = {This paper presents evaluation and data augmentation methods for improving OCR of the Latvian Language Spelling and Pronunciation Dictionary (LVPPV), a specialised dictionary containing numerous non-standard diacritic symbols used for transcribing Latvian pronunciation. To reduce the amount of manual correction required for OCR output, two new Tesseract v5 models were trained: one using a pre-existing annotated dataset and another using the same dataset supplemented with artificially generated training data. To enable more task-oriented model evaluation, two metrics were defined based on agreement between extracted lexeme–pronunciation pairs and two Latvian pronunciation resources: the Modern Latvian Language Dictionary and a rules-based phonetic transcriber. Both newly trained models outperform the previously published OCR model on these evaluation metrics, increasing transcription match rate from 71.81\% to 87.56\%. While synthetic data augmentation yields only limited improvements in overall transcription match rate, it substantially improves recognition of placenames containing uppercase letters, resulting in an 81\% increase in correctly extracted placename pronunciations compared to the non-augmented model.},
  url       = {https://aclanthology.org/2026.latell-1.30}
}

Author{1}{Orcid}:
@InProceedings{alahapperuma-vlachidis-bikakis:2026:latell,
  author    = {Alahapperuma, Malithi P.  and  Vlachidis, Andreas  and  Bikakis, Antonis},
  title     = {LLM-based conversational systems for Sinhala: Investigating model limitations and exploring a framework for agentic retrieval},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {304--308},
  abstract  = {Large Language Models (LLMs) have introduced significant shifts to the way we communicate and seek information. However, LLMs perform suboptimally in low-resource languages. Given their unprecedented capabilities, LLMs provide a unique opportunity for under-served linguistic communities to access knowledge and skills that were hitherto inaccessible. Therefore, finding practical and realistic ways for low-resource language users to access LLM-based conversational systems can have far-reaching societal and economic benefits. In this study, we characterise three limitations that reduce the usability of LLM-based conversational systems for Sinhala speakers and conceptually explore the design of an LLM agent-based retrieval framework.},
  url       = {https://aclanthology.org/2026.latell-1.31}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{lotman-EtAl:2026:latell,
  author    = {Lotman, Eno-Martin  and  Hübner, David  and  Collins, Kris  and  Boytcheva, Svetla  and  Nikolova-Koleva, Ivelina},
  title     = {Clinical Entity Recognition from Electronic Health Records and Linking to Biomedical Knowledge Bases in Low-resource Language Settings},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {309--317},
  abstract  = {Clinical Natural Language Processing (NLP) in low-resource languages is severely constrained by strict privacy regulations and a scarcity of localized biomedical ontologies. In Estonian, for instance, the localized SNOMED CT translation covers less than 10\% of the international target concept space. In this paper, we present a unified framework for clinical Named Entity Recognition (NER) and Entity Linking (EL) in low-resource settings, evaluated on real-world clinical narratives in Estonian, German, and Dutch. For NER, we deploy a hybrid architecture combining fine-tuned multitask Transformers, a few-shot generative LLM (Qwen2.5-14B), and dictionary-based lookups. For EL, we implement a two-stage pipeline using deterministic gazetteers alongside SapBERT encoders cross-lingually augmented via SNOMED CT terms and multi-ontology mappings. On clinical datasets, the NER pipeline achieved F1-scores of 0.87 (Dutch), 0.84 (Estonian), and 0.67 (German). However, direct cross-lingual performance comparisons are confounded by substantial variations in corpus size. Compact SapBERT models deliver superior EL precision (0.76) compared to larger multilingual baselines. This work aims to inform future NLP projects by identifying both limitations and promising pathways for advancing low-resource language technologies in biomedical informatics and beyond.},
  url       = {https://aclanthology.org/2026.latell-1.32}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0002-5542-9168
Author{5}{Orcid}:https://orcid.org/0009-0006-6322-0256
@InProceedings{bouamir-EtAl:2026:latell,
  author    = {Bouamir, Assia  and  Bonnin, Marie  and  Al Mouatamid, Youssef  and  Zahir, Jihad},
  title     = {Beyond RAG: A Multi-Graph, Multi-Agent, Recursive Retrieval Architecture for Traceable and Source-Grounded Legal Answers},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {318--325},
  abstract  = {Classical Retrieval-Augmented Generation (RAG) systems frequently generate legally relevant answers that lack accurate source citations, undermining their reliability for professional legal practice. This limitation is further compounded by document mismatch retrieval (DMR) — a systematic failure mode in which retrieved passages belong to structurally similar but legally distinct documents — which significantly reduces precision in complex multi-jurisdictional corpora. We present a novel architecture organized around three complementary knowledge layers: structural lexical graphs that capture document hierarchy, citation graphs enabling deterministic cross-reference resolution, and definition graphs for grounding formal legal terminology. Orchestrated by a multi-agent pipeline employing hybrid BM25 and vector fusion retrieval, the system enforces citation discipline at inference time and substantially improves answer accuracy. Evaluated on a novel benchmark in African francophone statutory law, consisting of 263 expert-validated question-answer pairs across two thematic domains—whale hunting regulation and hydrocarbon discharge control—our system achieves an average BERTScore F1 of 0.845, compared to 0.378 for a standard RAG baseline. Citation correctness reaches 94.0\% on average (versus 25.9\% for the baseline), as confirmed by expert ground truth and an independent human evaluation panel.},
  url       = {https://aclanthology.org/2026.latell-1.33}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{almutairi-EtAl:2026:latell,
  author    = {Almutairi, Ali  and  Alsuhaibani, Abdullah  and  Jameel, Shoaib  and  Joshi, Aditya  and  Mohammadi, Gelareh  and  Razzak, Imran},
  title     = {FLICK: Few-Label Incremental Learning for Low-Resource Dialects},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {326--338},
  abstract  = {Low-resource (LR) languages continue to face persistent challenges in natural language processing due to limited labelled datasets, scarce online textual resources, and the lack of NLP tools. Pre-trained language models and fine-tuning offer a promising starting point, yet their performance remains constrained by the scarcity of reliable annotations. Recent works have explored intermediate training with clusters and pseudo-labels to mitigate the limitation of few-label in fine-tuning. However, relying on cluster assignments or repeated can degrade performance, particularly for linguistically complex languages such as Arabic. To address these challenges, we propose FLICK, a framework for intermediate fine-tuning in few-label learning. FLICK first fine-tunes a language-specific encoder on a held-out split of pseudo-labelled data and retains only the Top-R clusters whose labels are reliably predicted; then, we train the model on these refined pseudo-labels. The resulting model is then transferred to the final few-label fine-tuning stage. Experiments across seven Arabic datasets spanning sentiment, sarcasm, dialect, and offensive language detection show that FLICK achieves the highest Macro-F1 scores among the evaluated Arabic BERT baselines and state-of-the-art few-label methods.},
  url       = {https://aclanthology.org/2026.latell-1.34}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:https://orcid.org/0000-0003-2200-9703
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{singh-EtAl:2026:latell,
  author    = {Singh, Pooja  and  Shahid, M Kaab Bin  and  Khan, Atai Waris  and  Jha, Aryan Kumar  and  Kumar, Sandeep},
  title     = {Linguistic Proximity Enables Pivot-Based Machine Translation for an Under-Resourced Tribal Language},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {339--349},
  abstract  = {For more than 10.4 million speakers of Bhili, an under-resourced tribal Indo-Aryan language of central and western India, limited machine translation support restricts digital accessibility and language inclusion. Beyond data scarcity, this challenge is intensified by the linguistic distance between English and Bhili. We examined whether this gap can be reduced through linguistically motivated pivot translation via Hindi and Marathi, two better-resourced languages that are more closely related to Bhili. Computational analysis shows substantially higher character-level overlap with Hindi and Marathi (67.6\% and 57.5\%) than with English (1.9\%), providing an empirical basis for pivot selection. In supervised experiments, cascaded fine-tuning via pivot languages improves the strongest direct translation of NLLB-200 baseline from 7.87 to 9.46 BLEU for English$\rightarrow$Bhili and from 25.23 to 29.84 BLEU for Bhili$\rightarrow$English, with corresponding ChrF++ gains of +1.91 and +5.18. Pivot-based prompting further shows that LLMs can benefit from related-language cues: Gemini 2.5 Flash improves by +20.09 ChrF++ for English$\rightarrow$Bhili using Hindi as the pivot, while GPT-4.5 reaches 46.87 ChrF++ for Bhili$\rightarrow$English. We release \textbf{BhilSetu}, a five-language parallel corpus of 1,33,000 sentences and the first Bhili FLORES-200-aligned benchmark. Our results suggest that leveraging linguistic proximity can improve translation quality while supporting broader efforts toward digital inclusion and preservation of under-resourced indigenous languages.},
  url       = {https://aclanthology.org/2026.latell-1.35}
}

Author{1}{Orcid}:
Author{2}{Orcid}:0009-0007-5683-9912
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{yousefi-estevanellvalladares-mitkov:2026:latell,
  author    = {Yousefi, Shahin  and  Estevanell-Valladares, Ernesto Luis  and  Mitkov, Ruslan},
  title     = {Cross-Lingual Hate Speech Detection in Low-Resource Languages: The Case of Persian and Kurdish},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {350--360},
  abstract  = {Hate speech detection in low-resource languages remains severely constrained by data scarcity and linguistic complexity. This study presents a comprehensive investigation of Persian-to-Kurdish cross-lingual transfer, evaluating four methodological paradigms: rule-based, traditional machine learning, transformer-based, and large language models, across monolingual and cross-lingual settings. Through a two-phase experimental design, we demonstrate that Persian data can substantially enhance Kurdish detection performance. Notably, models trained solely on Persian data frequently outperform the Kurdish (only), with ParsBERT achieving the strongest overall results. While translation-based data augmentation yields limited gains and introduces noise, model-level cross-lingual transfer proves highly effective. These findings highlight the practical value of leveraging related higher-resource languages for low-resource hate speech detection and provide actionable insights for building more equitable content moderation systems in underrepresented linguistic communities.},
  url       = {https://aclanthology.org/2026.latell-1.36}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0002-1168-1767
Author{3}{Orcid}:
@InProceedings{gupta-mitkov:2026:latell,
  author    = {Gupta, Divya  and  Mitkov, Ruslan},
  title     = {Is Overconfidence Language-Specific? Cross-Lingual Calibration and Recalibration Transfer in Base and Instruction-Tuned LLMs},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {361--369},
  abstract  = {Large language models are often treated as if the probability assigned to an answer were a reliable measure of confidence. Most evidence for this assumption, however, comes from English. This paper asks two related questions: whether instruction tuning changes calibration across languages, and whether a calibration correction learned in one language transfers to others. We compare base and instruction-tuned versions of Qwen2.5-7B and Llama-3.1-8B on Global-MMLU and Belebele, covering 42 languages across high-, mid-, and low-resource groups. Confidence is measured from the normalized next-token probabilities of the four answer options, and calibration is evaluated using expected calibration error (ECE), overconfidence error, and the multiclass Brier score. Instruction tuning raises ECE in 154 of 168 model–benchmark–language comparisons at the reported unadjusted p < 0.05 threshold, while accuracy changes much less consistently. The increase is largest in lower-resource languages. Temperature scaling fitted separately for each language substantially improves calibration across all three resource groups. A temperature fitted on English transfers reasonably well to high- and mid-resource targets, but leaves a larger residual error for low-resource targets. These results suggest that multilingual overconfidence is partly language-specific and that an English-only post-hoc correction may be least effective where labelled calibration data are scarcest. The findings apply to multiple-choice confidence and should not yet be generalized to open-ended generation.},
  url       = {https://aclanthology.org/2026.latell-1.37}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
@InProceedings{kirouchenassamy-EtAl:2026:latell,
  author    = {Kirouchenassamy, Badmavasan  and  Singh, Rahul  and  Dev, Vishnu  and  Garg, Meenakshi  and  Sawhney, Sukhna  and  Gowrishankar, Sudeep},
  title     = {Knowledge Tracing for Early Childhood Learners with Minimal Telemetry from Low-Connectivity Environments: Evidence from India's Anganwadi Ecosystem},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {370--377},
  abstract  = {Many of the students who could benefit most from distance learning lack the infrastructure it assumes. Most knowledge-tracing research takes personal devices, stable internet, and rich interaction logs for granted, and it has mostly been developed and evaluated with older learners. Early childhood in low-connectivity environments is therefore still underexplored. We ask whether two established Knowledge Tracing (KT) models, Bayesian Knowledge Tracing (BKT) and the Dynamic Key-Value Memory Network (DKVMN), can accurately model early-childhood Hindi literacy in India's Anganwadi ecosystem. Both were designed for high-resource settings, whereas here browser-based Learning Games are played on shared phones over slow, intermittent connections. Using only a timestamped sequence of correct and incorrect responses from 210,500 question attempts by 7,547 learners aged 3 to 6, we validate the models' mastery estimates against offline field assessments of 552 children. DKVMN predicts the next answer well (AUC 0.858), on par with high-resource benchmarks, and its mastery estimates correlate significantly with field assessments from as few as three attempts. BKT is less accurate overall, but gives more reliable per-skill estimates, interpretable diagnostics, and greater robustness to submissions lost to poor connectivity. These results show that meaningful learner modeling is feasible for early childhood education in low-connectivity environments, and that evaluating such models requires external validation against offline assessments and robustness to data loss.},
  url       = {https://aclanthology.org/2026.latell-1.38}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
@InProceedings{hennara-EtAl:2026:latell1,
  author    = {Hennara, Khalil  and  Chrouf, Sara  and  Hamed, Mohamed Motasim  and  Aldallal, Zeina  and  AlModhayan, Safwan},
  title     = {Kuwain 1.5B: An Arabic SLM via Language Injection},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {378--387},
  abstract  = {Enhancing existing models with new knowledge is a crucial aspect of AI development. This paper introduces a novel method for integrating a new language into a large language model (LLM). Our approach successfully incorporates a previously unseen target language into an existing LLM without compromising its prior knowledge. We trained a tiny model with 1.5 billion parameters named Kuwain by injecting the Arabic language into a small open-source model mainly trained in English. Our method demonstrates significant improvements in Arabic language performance, with an average 8\% improvement across various benchmarks, while retaining the model's ex- isting knowledge with a minimum amount of the original model's data. This offers a cost-effective alternative to training a comprehensive model in both English and Arabic. The results highlight the potential for efficient, targeted language model expansion without extensive retraining or resource-intensive processes. To support the research community, we also open-source Kuwain's model weights, enabling further development for Arabic and other underrepresented languages.},
  url       = {https://aclanthology.org/2026.latell-1.39}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
@InProceedings{afrouni-ataaallah-abarnous:2026:latell,
  author    = {Afrouni, Azzeddine  and  Ataa Allah, Fadoua  and  Abarnous, Jamal},
  title     = {Towards a Universal Dependencies Treebank for Amazigh: A Tarifit Pilot Study},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {24--29},
  abstract  = {This paper presents a pilot study on the construction of a Universal Dependencies (UD) treebank for Tarifit, a low-resource variety of Moroccan Amazigh. To bridge the syntax-resource gap, we develop a manually annotated corpus comprising 123 sentences collected from textbooks and digital sources, including IRCAM and MAP portals. The selected texts were chosen for their orthographic consistency to ensure annotation quality and reliability. Morphosyntactic annotation has been completed for the full corpus, while dependency annotation has been finalized for 50 sentences. Annotated under UD guidelines v2, the study focuses on language-specific phenomena including pronominal cliticization and word order flexibility. We detail the annotation challenges in adapting the UD framework to Tarifit and propose cross-linguistically consistent solutions within the UD ecosystem.},
  url       = {https://aclanthology.org/2026.latell-1.4}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{hadiza-EtAl:2026:latell,
  author    = {Hadiza, Assoumana Souley  and  Al Mouatamid, Youssef  and  Bonnin, Marie  and  Zahir, Jihad},
  title     = {A Neuro-Symbolic RAG System for Marine Environmental Law: from Domain Ontology to Knowledge Graphs and Context-Enriched Legal Question Answering},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {388--397},
  abstract  = {Legal question answering in multi-jurisdictional normative domains demands factual precision that flat vector retrieval cannot guarantee. We present a neuro-symbolic RAG system for expert-level marine environmental law QA targeting national legislation across 15 West and North African coastal states. The system is grounded in a formally verified OWL 2.0 DL ontology of 120 classes, 8,844 RDF triples, and 1,217 individuals covering six prohibition categories from 12 international conventions, aligned with LKIF-Core. Its central contribution is a controlled ontology-to-graph transition: 68 T-Box axioms filter LLM-extracted A-Box assertions, and the validated graph is exported to Neo4j, decoupling formal reasoning from retrieval. A SKOS lexical bridge drives synonym integration; BGE-M3 and BM25 retrieval are fused via Reciprocal Rank Fusion, then re-ranked before a seven-step Ontological Reasoning Agent injects multi-hop legal facts and normative gap signals. An ablation on 66 expert-annotated queries, using Llama 3.2 3B as a deliberate stress test, shows the full pipeline achieves Context Precision 0.8929 (+23.1\% over the agent-only baseline) and a +32\% Faithfulness gain, establishing that context construction quality matters more than generator capacity in formally structured legal domains. On the hardest exception-discrimination queries, the full pipeline reaches 62.5\% accuracy against expert ground truth, against 54.2\% for dense-only retrieval. Keywords: Neuro-symbolic RAG, Marine Environmental Law, Legal Question Answering, Ontology Engineering, Knowledge Graphs, Multi-Jurisdictional Retrieval, Explainable AI},
  url       = {https://aclanthology.org/2026.latell-1.40}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{younes-dahou-mathiak:2026:latell,
  author    = {Younes, Yousef  and  Dahou, Abdelhalim Hafedh  and  Mathiak, Brigitte},
  title     = {Evaluating Model-Task Fit in Arabic Word Sense Disambiguation},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {398--407},
  abstract  = {Word Sense Disambiguation (WSD) is the computational task of identifying a word's intended meaning within a specific context when the word exhibits lexical ambiguity. It remains a core and challenging problem in Natural Language Processing (NLP). We formulate the problem in two ways: binary classification (context-gloss alignment) and multi-choice selection (identifying the correct sense from candidate glosses) using Large Language Models (LLMs) and Small Language Models (SLMs). Our results on the KSAA-CAD shared task dataset show a critical task-model dependency, where fine-tuned SLMs excel in binary classification (mBERT achieving 83.74\% F1), while LLMs show strong performance on multi-choice (fine-tuned Gemma2 7B attaining 84.26\% F1). Notably, Camel-BERT-MSA shows robust cross-task consistency, securing second place in both evaluations. This insight underscores that the effectiveness of a model in WSD depends heavily on the specific task. It also emphasizes the value of aligning model architecture with task requirements, offering useful insights for applying WSD models in real-world settings.},
  url       = {https://aclanthology.org/2026.latell-1.41}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{ahmed-EtAl:2026:latell,
  author    = {Ahmed, Zehra  and  Inayat, Farah  and  Aqib, Zuha  and  Haider, Sajjad},
  title     = {Enhancing Urdu ASR with Whisper v3: Fine-Tuning on Latest Datasets and Realistic Multi-Speaker Evaluation with SLM Post-Processing},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {408--417},
  abstract  = {Automatic Speech Recognition (ASR) for low-resource languages like Urdu remains challenging due to limited training data, dialectal variation, code-switching, and orthographic inconsistencies. This work evaluates OpenAI Whisper large-v3-Turbo for Urdu ASR and performs Low Rank Adaption (LoRA) fine-tuning on Common Voice v23, FLEURS and CSaLT datasets, drawn from a pooled corpus of 31 hours of Urdu speech. To gauge robustness in the real world, we also created an evaluation set from YouTube that includes news segments with two people, namely an anchor and an on-site reporter, with natural accents, interruptions, and background noise. Apart from this, the paper explored Urdu-capable Small Language Models (SLMs) between 3 to 14 billion parameters for generative error correction (GEC). Whisper-Turbo achieved 25.33\% WER on the evaluation set from YouTube, whereas Whisper-Turbo with Gemma3-12B 4-bit Quantized as SLM GEC attained 21.75\% WER, which translates to a 14.14\% relative reduction. The results indicate that there is a potential application of SLM post-processing for long-form Urdu transcription, at the cost of extra runtime.},
  url       = {https://aclanthology.org/2026.latell-1.42}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{hennara-EtAl:2026:latell2,
  author    = {Hennara, Khalil  and  Hreden, Muhammad  and  Hamed, Mohamed Motasim  and  Bastati, Ahmad  and  Aldallal, Zeina  and  Chrouf, Sara  and  AlModhayan, Safwan},
  title     = {Baseer: A Vision-Language Model for Arabic Document-to-Markdown OCR},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {418--428},
  abstract  = {Arabic Optical Character Recognition (OCR) remains a challenging task due to the language's cursive script, diverse fonts, diacritics, and right-to-left orientation. While modern Multimodal Large Language Models (MLLMs) have advanced document understanding for high-resource languages, their performance on Arabic remains limited. In this work, we introduce Baseer, a vision-language model fine-tuned specifically for Arabic document OCR. Leveraging a large-scale dataset combining synthetic and real-world documents, Baseer is trained using a decoder-only fine-tuning strategy to adapt a pre-trained MLLM while preserving general visual features. We also present Misraj-DocOCR, a high-quality, expert verified benchmark designed for rigorous evaluation of Arabic OCR systems. Our experiments show that Baseer significantly outperforms existing open-source and commercial solutions, establishing a new state-of-the-art in the domain of Arabic document OCR. Our results highlight the benefits of domain-specific adaptation of general-purpose MLLMs and establish a strong baseline for high-accuracy OCR on morphologically rich languages like Arabic.},
  url       = {https://aclanthology.org/2026.latell-1.43}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
@InProceedings{ilman:2026:latell,
  author    = {Ilman, Tetiana},
  title     = {Worldview Annotation for Low-Resource Languages},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {429--437},
  abstract  = {Low-resource language technology has advanced significantly in preserving linguistic form, yet a critical gap remains: the conceptual systems that languages encode - how communities structure personhood, agency, time, evidence, relationality, and moral responsibility - are systematically lost in current documentation pipelines. We argue that the problem is not only the annotation scale but annotation level. Annotation pipelines are not neutral: they systematically remove meaning-encoding distinctions that are not directly representable in dominant-language categories, even when data originates from native speakers. We introduce the Worldview Annotation Schema (WAS), a three-layer framework that starts at community member level in the field and ends with machine-readable structured data, catching culturally grounded meaning before downstream AI analysis can homogenize it. We track where worldview meaning dissolves in existing pipelines, propose the WAS architecture with a worked example from Maa (Eastern Africa), and outline a governance model that maintains interpretive authority with speaker communities while producing graph-compatible structured output. In the age of AI, language preservation without worldview preservation risks producing languages that survive archivally while disappearing cognitively.},
  url       = {https://aclanthology.org/2026.latell-1.44}
}

Author{1}{Orcid}:
@InProceedings{petalinkar-EtAl:2026:latell,
  author    = {Petalinkar, Saša  and  Ikonić Nešić, Milica  and  Stanković, Ranka  and  Graovac, Jelena},
  title     = {A Multilingual Sentence-Transformer Baseline for Serbian News Topic Classification},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {438--448},
  abstract  = {We present a baseline for topic classification of Serbian newspaper articles. Starting from a corpus of roughly 123,000 newspaper articles annotated with section labels, we apply a deterministic label-normalization procedure that corrects encoding-corrupted spellings, merges truncated labels, and removes a non-coherent catch-all category, resulting in 35 normalized classes. The multilingual MiniLM sentence-transformer is fine-tuned with a multiple-negatives ranking objective and, at test time, all 35 section labels are ranked by cosine similarity, with the top-ranked label treated as a single direct classification hit. On a held-out test split of 18,487 articles, the model reaches a top-1 accuracy of 0.784 (Wilson 95\% interval 0.778--0.790), with a weighted F1 of 0.790 and a macro F1 of 0.653. Per-section analysis shows strong performance on high-frequency sections and substantial over-prediction of small, semantically adjacent classes. The fine-tuned model, per-section metrics, a confusion analysis, and documented label-cleaning corrections are released to support further work on Serbian news classification.},
  url       = {https://aclanthology.org/2026.latell-1.45}
}

Author{1}{Orcid}:
Author{2}{Orcid}:https://orcid.org/0000-0002-0835-8889
Author{3}{Orcid}:0000-0001-5123-6273
Author{4}{Orcid}:
@InProceedings{hennara-EtAl:2026:latell3,
  author    = {Hennara, Khalil  and  Bastati, Ahmad  and  Hreden, Muhammad  and  Hamed, Mohamed Motasim  and  Aldallal, Zeina  and  Chrouf, Sara  and  AlModhayan, Safwan},
  title     = {Wasm: A Pipeline for Constructing Structured Arabic Interleaved Multimodal Corpora},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {449--458},
  abstract  = {The performance of large language models (LLMs) and large multimodal models (LMMs) depends heavily on the quality and scale of their pre-training datasets. Recent research shows that LMMs trained on natural documents where images and text are interleaved outperform those trained only on image–text pairs across a wide range of benchmarks, leveraging advanced pre-trained models to enforce semantic alignment, image-sequence consistency, and textual coherence. For Arabic, however, the lack of high-quality multimodal datasets that preserve document structure has limited progress. In this paper, we present our pipeline Wasm for processing the Common Crawl dataset to create a new Arabic multimodal dataset that uniquely provides markdown output. Unlike existing Arabic corpora that fo- cus solely on text extraction, our approach preserves the structural integrity of web content while maintaining flexibility for both text-only and multimodal pre-training scenarios. We provide a comprehensive comparative analysis of our data processing pipeline against those used for major existing datasets, highlighting the convergences in filtering strategies and justifying our specific design choices. To support future research, we publicly release a representative dataset dump along with the full pipeline implementation.},
  url       = {https://aclanthology.org/2026.latell-1.46}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
Author{5}{Orcid}:
Author{6}{Orcid}:
Author{7}{Orcid}:
@InProceedings{abdelfattah-EtAl:2026:latell,
  author    = {Abdelfattah, Ahmad  and  Aldallal, Zeina  and  Chrouf, Sara  and  AlModhayan, Safwan},
  title     = {Linguistic Specialization of Arabic in Text Embedding Models via Multi-Teacher Knowledge Distillation},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {459--467},
  abstract  = {This submission presents a novel approach to embedding model distillation that improves the capabilities of embedding models in Arabic using multiple teachers, without requiring a contrastive learning phase.},
  url       = {https://aclanthology.org/2026.latell-1.47}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
Author{4}{Orcid}:
@InProceedings{alharbi-rbaiti-mitkov:2026:latell,
  author    = {Alharbi, Maram I.  and  R’baiti, Jihad  and  Mitkov, Ruslan},
  title     = {Arabic Dialect-to-MSA Translation: A Comparative Evaluation of Large Language Models and Neural Machine Translation},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {30--39},
  abstract  = {Arabic dialect to Modern Standard Arabic (MSA) translation remains challenging due to the rich linguistic variation across dialects and the limited availability of parallel corpora. Recent advances in Large Language Models (LLMs) have created new potential to address these challenges. However, their effectiveness in Arabic dialect translation remains underexplored. In this paper, we present a comparative evaluation of Arabic language models, multilingual language models, and dedicated neural machine translation under zero-shot, few-shot, and fine-tuning settings on the ADOR dataset, a parallel corpus covering Saudi, Moroccan Darija, Egyptian, and Jordanian Arabic. Results show clear differences between model families, with Fanar achieving the strongest overall performance. Fine-tuning consistently improved translation quality, whereas prompting produced more variable results across dialects.},
  url       = {https://aclanthology.org/2026.latell-1.5}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{fernandes:2026:latell,
  author    = {Fernandes, Rafael Macario},
  title     = {Cross-Lingual Transfer from Portuguese to Nheengatu: Evidence for Contact-Induced Convergence as a Computational Bridge},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {40--49},
  abstract  = {Typological distance predicts that Portuguese-to-Nheengatu transfer should fail: the languages are genetically unrelated and typologically divergent (URIEL+ genetic 1.0, syntactic 0.61). We argue that 400 years of contact-induced structural convergence may provide a computational bridge invisible to these measures, so that transfer should succeed at the levels where contact has operated most deeply. Across five experiments on a parallel corpus of about 5,000 sentence pairs, we find a progressive pattern: word-level and lexical-semantic methods fail, while fine-tuned XLM-RoBERTa reaches Precision@1 of 24.7\% and MRR of 0.371 on cross-lingual sentence retrieval—more than double a BM25 character-n-gram baseline and 1.7× that of zero-shot LaBSE, the strongest off-the-shelf retriever we test. Crucially, Portuguese beats Spanish—typologically near-identical and no less represented in pretraining, but without comparable contact history—on every metric and all five seeds, and the two sources share substantial lexical overlap, so lexical overlap alone is unlikely to explain the gap. These results suggest that contact history can distinguish transfer sources that typological distance treats as equivalent.},
  url       = {https://aclanthology.org/2026.latell-1.6}
}

Author{1}{Orcid}:
@InProceedings{miyagawa:2026:latell,
  author    = {Miyagawa, So},
  title     = {Retrieval-Augmented Machine Translation for Bohairic Coptic: A Pilot Case Study},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {50--58},
  abstract  = {Coptic is extremely low-resource in natural language processing despite its importance for philology, historical linguistics, and heritage-language learning. This paper reports a pilot case study of Bohairic Coptic–English translation using one passage from the Vita Sinuthii, one published English reference, and one retained output from each of seven deployed systems. THOTH AI, a retrieval-augmented system built with Coptological lexical, corpus, and grammatical resources, obtained 40.63 BLEU, compared with 31.40 for a separately accessed Gemini 3.0 Pro baseline. Results from chrF, ROUGE, TER, and METEOR are also reported, together with an illustrative expert reading of lexical errors and unsupported generation. These figures are descriptive, not a general benchmark or a causal retrieval ablation: the systems were accessed through different deployed interfaces, exact retrieval traces were not retained, and the evaluation contains only one passage. Three outputs contained clear passage-scale fabricated material. The study documents this bounded case and the evidence needed for a larger, reproducible evaluation.},
  url       = {https://aclanthology.org/2026.latell-1.7}
}

Author{1}{Orcid}:
@InProceedings{meroni-franois-silberztein:2026:latell,
  author    = {Meroni, Fabio  and  François, Alexandre  and  Silberztein, Max},
  title     = {Automating Mwotlap morphology using formal grammar},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {59--67},
  abstract  = {This paper presents a rule-based computational formalization of vowel alternations in Mwotlap morphology using NooJ. Mwotlap displays several morphophonological processes at the prefix-root boundary, including vowel copy, epenthesis, transfer, elision, and blockage. To model these first three, which cannot be derived by simple string concatenation within NooJ's finite-state framework, we introduce a new string-edit operator which swaps adjacent segments. This operator unifies vowel copy, insertion, and transfer under a single formal mechanism while preserving linguistic generalization.},
  url       = {https://aclanthology.org/2026.latell-1.8}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
@InProceedings{guan-buzaaba-fellbaum:2026:latell,
  author    = {Guan, Kevin  and  Buzaaba, Happy  and  Fellbaum, Christiane},
  title     = {Dependency Parsing Across the Resource Spectrum: Evaluating Architectures on High and Low-Resource Languages},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {68--79},
  abstract  = {Transformer-based models achieve state-of-the-art dependency parsing for high-resource languages, yet their advantage over simpler architectures in low-resource settings remains poorly understood. We evaluate four parsers—the Biaffine LSTM, Stack-Pointer Network, AfroXLMR-large, and RemBERT—across twelve typologically diverse languages, with a focus on low-resource African languages. We find that the Biaffine LSTM consistently outperforms transformer models in low-resource regimes, with transformers recovering their advantage as training data increases. The crossover falls within a resource range typical of treebanks for under-resourced languages. Morphological complexity (measured via MATTR) emerges as a significant secondary predictor of transformers' relative disadvantage after controlling for corpus size. These results indicate that the Biaffine LSTM may be better suited for syntactic tool development in low-resource regimes until sufficient annotated data is available to leverage the representational capacity of pre-trained transformers.},
  url       = {https://aclanthology.org/2026.latell-1.9}
}

Author{1}{Orcid}:
Author{2}{Orcid}:
Author{3}{Orcid}:
