@InProceedings{alhuthali-zou-orasan:2026:NeTTIT,
  author    = {Al-Huthali, Najat B.  and  Zou, Yuan  and  Orasan, Constantin},
  title     = {Fine-tuning Arabic-English NMT Engines for Investment Laws},
  booktitle      = {Proceedings of the Conference on New Trends in Translation and Interpreting Technology 2026},
  month          = {June},
  year           = {2026},
  address        = {Dubrovnik, Croatia},
  publisher      = {INCOMA Ltd., Shoumen, Bulgaria},
  pages     = {146--156},
  abstract  = {The growing volume of legislative and regulatory translation is Saudi Arabia has created an increasing need for reliable Arabic-English machine translation (MT) solutions in the legal domain. However, Neural Machine Translation (NMT) systems trained on general-domain data perform poorly on legal texts due to terminological density, formulaic structures, and jurisdiction-specific drafting conventions. A major obstacle to domain adaptation for Arabic legal MT is the scarcity of large, high-quality, and publicly accessible parallel corpora representing national legislation. This study addresses this gap by constructing a specialised Arabic–English parallel corpus of Saudi investment laws and regulations compiled from official governmental sources. The paper documents the challenges involved in legal text extraction, OCR processing, cleaning, and alignment. Using this corpus, a pilot domain-adaptation experiment is conducted by fine-tuning the OPUS-MT Arabic–English model. Preliminary evaluation indicates that exposure to in-domain legal data improves terminology handling and structural adequacy compared to the baseline generic model.},
  url       = {https://aclanthology.org/2026.nettit-1.19}
}

