@inproceedings{3f37efca0f874275909b4a2645d54f42,
title = "Learning Cross-Dialectal Morphophonology with Syllable Structure Constraints",
abstract = "We investigate learning surface forms from underlying morphological forms for low-resource language varieties. We concentrate on learning explicit rules with the aid of learned syllable structure constraints, which outperforms neural methods on this small data task and provides interpretable output. Evaluating across one relatively high-resource and two related low-resource Arabic dialects, we find that a model trained only on the high-resource dialect achieves decent performance on the low-resource dialects, useful when no low-resource training data is available. The best results are obtained when our system is trained only on the low-resource dialect data without augmentation from the related higher-resource dialect. We discuss the impact of syllable structure constraints and the strengths and weaknesses of data augmentation and transfer learning from a related dialect.",
author = "Salam Khalifa and Abdelrahim Qaddoumi and Jordan Kodner and Owen Rambow",
note = "Publisher Copyright: {\textcopyright} 2025 Association for Computational Linguistics.; 12th Workshop on NLP for Similar Languages, Varieties and Dialects, VarDial 2025 - co-located with the 31st International Conference on Computational Linguistics, COLING 2025 ; Conference date: 19-01-2025",
year = "2025",
language = "English",
series = "VarDial 2025 - 12th Workshop on NLP for Similar Languages, Varieties and Dialects, Proceedings of the Workshop",
publisher = "Association for Computational Linguistics (ACL)",
pages = "157--167",
editor = "Yves Scherrer and Tommi Jauhiainen and Nikola Ljubesic and Preslav Nakov and Jorg Tiedemann and Marcos Zampieri",
booktitle = "VarDial 2025 - 12th Workshop on NLP for Similar Languages, Varieties and Dialects, Proceedings of the Workshop",
}