@inproceedings{bhadauria-etal-2024-effects,
title = "The Effects of Data Quality on Named Entity Recognition",
author = "Bhadauria, Divya and
Sierra M{\'u}nera, Alejandro and
Krestel, Ralf",
editor = {van der Goot, Rob and
Bak, JinYeong and
M{\"u}ller-Eberstein, Max and
Xu, Wei and
Ritter, Alan and
Baldwin, Tim},
booktitle = "Proceedings of the Ninth Workshop on Noisy and User-generated Text (W-NUT 2024)",
month = mar,
year = "2024",
address = "San {\.{G}}iljan, Malta",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/add-emnlp-2024-awards/2024.wnut-1.8/",
pages = "79--88",
abstract = "The extraction of valuable information from the vast amount of digital data available today has become increasingly important, making Named Entity Recognition models an essential component of information extraction tasks. This emphasizes the importance of understanding the factors that can compromise the performance of these models. Many studies have examined the impact of data annotation errors on NER models, leaving the broader implication of overall data quality on these models unexplored. In this work, we evaluate the robustness of three prominent NER models on datasets with varying amounts of textual noise types. The results show that as the noise in the dataset increases, model performance declines, with a minor impact for some noise types and a significant drop in performance for others. The findings of this research can be used as a foundation for building robust NER systems by enhancing dataset quality beforehand."
}
Markdown (Informal)
[The Effects of Data Quality on Named Entity Recognition](https://preview.aclanthology.org/add-emnlp-2024-awards/2024.wnut-1.8/) (Bhadauria et al., WNUT 2024)
ACL
- Divya Bhadauria, Alejandro Sierra Múnera, and Ralf Krestel. 2024. The Effects of Data Quality on Named Entity Recognition. In Proceedings of the Ninth Workshop on Noisy and User-generated Text (W-NUT 2024), pages 79–88, San Ġiljan, Malta. Association for Computational Linguistics.