@inproceedings{dang-etal-2026-toward,
title = "Toward Generalized Cross-Lingual Hateful Language Detection with Web-Scale Data and Ensemble {LLM} Annotations",
author = "Dang, Dang Hai and
Mitrovi{\'c}, Jelena and
Granitzer, Michael",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/paragraph-normalization/2026.lrec-1.38/",
doi = "10.63317/22xjdmqtv6mk",
pages = "549--559",
abstract = "We study whether large-scale unlabelled web data and LLM-based synthetic annotations can improve multilingual hate speech detection. Starting from texts crawled via OpenWebSearch{~}(OWS) in four languages (English, German, Spanish, Vietnamese), we pursue two complementary strategies. First, we apply continued pre-training to BERT models by continuing masked language modelling on unlabelled OWS texts before supervised fine-tuning, and show that this yields an average macro-F1 gain of approximately 3{\%} over standard baselines across sixteen benchmarks, with stronger gains in low-resource settings. Second, we use four open-source LLMs (Mistral-7B, Llama3.1-8B, Gemma2-9B, Qwen2.5-14B) to produce synthetic annotations through three ensemble strategies: mean averaging, majority voting, and a LightGBM meta-learner. The LightGBM ensemble consistently outperforms the other strategies. Fine-tuning on these synthetic labels substantially benefits a small model Llama3.2-1B: +11{\%} pooled F1), but provides only a modest gain for the larger Qwen2.5-14B (+0.6{\%}). Our results indicate that the combination of web-scale unlabelled data and LLM-ensemble annotations is most valuable for smaller models and low-data languages."
}Markdown (Informal)
[Toward Generalized Cross-Lingual Hateful Language Detection with Web-Scale Data and Ensemble LLM Annotations](https://preview.aclanthology.org/paragraph-normalization/2026.lrec-1.38/) (Dang et al., LREC 2026)
ACL