@inproceedings{6fc7f3768310478cb5e7c281882cccbf,
title = "Clustering Arabic tweets for sentiment analysis",
abstract = "The focus of this study is to evaluate the impact of linguistic preprocessing and similarity functions for clustering Arabic Twitter tweets. The experiments apply an optimized version of the standard K-Means algorithm to assign tweets into positive and negative categories. The results show that root-based stemming has a significant advantage over light stemming in all settings. The Averaged Kullback-Leibler Divergence similarity function clearly outperforms the Cosine, Pearson Correlation, Jaccard Coefficient and Euclidean functions. The combination of the Averaged Kullback-Leibler Divergence and root-based stemming achieved the highest purity of 0.764 while the second-best purity was 0.719. These results are of importance as it is contrary to normal-sized documents where, in many information retrieval applications, light stemming performs better than root-based stemming and the Cosine function is commonly used.",
keywords = "Arabic stemmers, Arabic tweets, Clustering algorithms, K-means, Sentiment analysis",
author = "DIab Abuaiadah and DIleep Rajendran and Mustafa Jarrar",
note = "Publisher Copyright: {\textcopyright} 2017 IEEE.; 14th IEEE/ACS International Conference on Computer Systems and Applications, AICCSA 2017 ; Conference date: 30-10-2017 Through 03-11-2017",
year = "2017",
month = jul,
day = "2",
doi = "10.1109/AICCSA.2017.162",
language = "English",
series = "Proceedings of IEEE/ACS International Conference on Computer Systems and Applications, AICCSA",
publisher = "IEEE Computer Society",
pages = "449--456",
booktitle = "Proceedings - 2017 IEEE/ACS 14th International Conference on Computer Systems and Applications, AICCSA 2017",
address = "United States",
}