@inproceedings{aliwy-etal-2020-arabic,
title = "{A}rabic Dialects Identification for All {A}rabic countries",
author = "Aliwy, Ahmed and
Taher, Hawraa and
AboAltaheen, Zena",
editor = "Zitouni, Imed and
Abdul-Mageed, Muhammad and
Bouamor, Houda and
Bougares, Fethi and
El-Haj, Mahmoud and
Tomeh, Nadi and
Zaghouani, Wajdi",
booktitle = "Proceedings of the Fifth Arabic Natural Language Processing Workshop",
month = dec,
year = "2020",
address = "Barcelona, Spain (Online)",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/jlcl-multiple-ingestion/2020.wanlp-1.32/",
pages = "302--307",
abstract = {Arabic dialects are among of three main variant of Arabic language (Classical Arabic, modern standard Arabic and dialectal Arabic). It has many variants according to the country, city (provinces) or town. In this paper, several techniques with multiple algorithms are applied for Arabic dialects identification starting from removing noise till classification task using all Arabic countries as 21 classes. Three types of classifiers (Na{\"i}ve Bayes, Logistic Regression, and Decision Tree) are combined using voting with two different methodologies. Also clustering technique is used for decreasing the noise that result from the existing of MSA tweets in the data set for training phase. The results of f-measure were 27.17, 41.34 and 52.38 for first methodology without clustering, second methodology without clustering, and second methodology with clustering, the used data set is NADI shared task data set.}
}
Markdown (Informal)
[Arabic Dialects Identification for All Arabic countries](https://preview.aclanthology.org/jlcl-multiple-ingestion/2020.wanlp-1.32/) (Aliwy et al., WANLP 2020)
ACL