@inproceedings{a2ef4ee8c05b404f9c0dd1ec575d1c96,
title = "Lisan: Yemeni, Iraqi, Libyan, and Sudanese Arabic Dialect Corpora with Morphological Annotations",
abstract = "This article presents morphologically-annotated Yemeni, Sudanese, Iraqi, and Libyan Arabic dialects (Lisan) corpora. Lisan features around 1.2 million tokens. We collected the content of the corpora from several social media platforms. The Yemeni corpus ((1) over tilde .05M tokens) was collected automatically from Twitter. The corpora of the other three dialects ((5) over tilde 0K tokens each) was manually collected from Facebook and YouTube posts and comments. Thirty-five (35) annotators who are native speakers of the target dialects carried out the annotations. The annotators segmented all words in the four corpora into prefixes, stems and suffixes and labeled each with different morphological features such as part of speech, lemma, and a gloss in English. We developed the Arabic Dialect Annotation Toolkit (ADAT) to assist the annotators and to ensure compatibility with SAMA and Curras tagsets. We trained annotators on a set of guidelines and on how to use ADAT. ADAT is open source, and the four corpora are available at https://sina.birzeit.edu/currasat.",
author = "Mustafa Jarrar and Zaraket, \{Fadi A.\} and Tymaa Hammouda and Alavi, \{Daanish Masood\} and Martin Wahlisch",
note = "Publisher Copyright: {\textcopyright} 2023 IEEE.; 20th ACS/IEEE International Conference on Computer Systems and Applications, AICCSA 2023 ; Conference date: 04-12-2023 Through 07-12-2023",
year = "2023",
month = dec,
day = "7",
doi = "10.1109/AICCSA59173.2023.10479250",
language = "English",
series = "International Conference On Computer Systems And Applications",
publisher = "IEEE Computer Society",
booktitle = "2023 20th Acs/ieee International Conference On Computer Systems And Applications, Aiccsa",
address = "United States",
}