@inproceedings{xu-etal-2023-comparative,
title = "Comparative Analysis of Anomaly Detection Algorithms in Text Data",
author = "Xu, Yizhou and
G{\'a}bor, Kata and
Milleret, J{\'e}r{\^o}me and
Segond, Fr{\'e}d{\'e}rique",
editor = "Mitkov, Ruslan and
Angelova, Galia",
booktitle = "Proceedings of the 14th International Conference on Recent Advances in Natural Language Processing",
month = sep,
year = "2023",
address = "Varna, Bulgaria",
publisher = "INCOMA Ltd., Shoumen, Bulgaria",
url = "https://preview.aclanthology.org/fix-sig-urls/2023.ranlp-1.131/",
pages = "1234--1245",
abstract = "Text anomaly detection (TAD) is a crucial task that aims to identify texts that deviate significantly from the norm within a corpus. Despite its importance in various domains, TAD remains relatively underexplored in natural language processing. This article presents a systematic evaluation of 22 TAD algorithms on 17 corpora using multiple text representations, including monolingual and multilingual SBERT. The performance of the algorithms is compared based on three criteria: degree of supervision, theoretical basis, and architecture used. The results demonstrate that semi-supervised methods utilizing weak labels outperform both unsupervised methods and semi-supervised methods using only negative samples for training. Additionally, we explore the application of TAD techniques in hate speech detection. The results provide valuable insights for future TAD research and guide the selection of suitable algorithms for detecting text anomalies in different contexts."
}
Markdown (Informal)
[Comparative Analysis of Anomaly Detection Algorithms in Text Data](https://preview.aclanthology.org/fix-sig-urls/2023.ranlp-1.131/) (Xu et al., RANLP 2023)
ACL