@InProceedings{lasmanis:2026:latell,
  author    = {Lasmanis, Viesturs Jūlijs},
  title     = {Improving OCR for a Latvian Pronunciation Dictionary},
  booktitle      = {Proceedings of the First International Conference on Language Technologies for Low-resource Languages (LaTeLL 2026)},
  month          = {September},
  year           = {2026},
  address        = {Fes, Morocco},
  publisher      = {Association for Computational Linguistics},
  pages     = {295--303},
  abstract  = {This paper presents evaluation and data augmentation methods for improving OCR of the Latvian Language Spelling and Pronunciation Dictionary (LVPPV), a specialised dictionary containing numerous non-standard diacritic symbols used for transcribing Latvian pronunciation. To reduce the amount of manual correction required for OCR output, two new Tesseract v5 models were trained: one using a pre-existing annotated dataset and another using the same dataset supplemented with artificially generated training data. To enable more task-oriented model evaluation, two metrics were defined based on agreement between extracted lexeme–pronunciation pairs and two Latvian pronunciation resources: the Modern Latvian Language Dictionary and a rules-based phonetic transcriber. Both newly trained models outperform the previously published OCR model on these evaluation metrics, increasing transcription match rate from 71.81\% to 87.56\%. While synthetic data augmentation yields only limited improvements in overall transcription match rate, it substantially improves recognition of placenames containing uppercase letters, resulting in an 81\% increase in correctly extracted placename pronunciations compared to the non-augmented model.},
  url       = {https://aclanthology.org/2026.latell-1.30}
}

