@inproceedings{rossini-van-der-plas-2026-124,
title = "From 124 Million Tokens to 1,021 Neologisms: A Large-Scale Pipeline for Automatic Neologism Detection",
author = "Rossini, Diego and
van der Plas, Lonneke",
editor = "Oleskeviciene, Giedre Valunaite and
Giouli, Voula and
Armaselu, Florentina and
Liebeskind, Chaya and
McGillivray, Barbara",
booktitle = "Proceedings of the Workshop Neology and Large Language Models",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/test-year-match/2026.neollm-1.1/",
doi = "10.63317/4o6ks86o293r",
pages = "1--15",
abstract = "We present a scalable, modular pipeline for automatic neologism detection that combines rule-based filtering with LLM classification. The pipeline is grounded in two complementary word-formation frameworks, grammatical and extra-grammatical morphology, which jointly define the scope of what counts as a neologism and inform a four-class classification scheme (NEOLOGISM, ENTITY, FOREIGN, NONE). While designed to be modular and transferable at the architectural level, the pipeline is instantiated on 527 million English-language Reddit posts spanning 2005{--}2024. From this corpus, we extract 124.6 million unique tokens and reduce them by over 99.99{\%} to yield 1,021 neologism candidates, a set small enough for manual expert verification. Multiple LLMs independently classify each candidate via majority vote, with a final verification step, revealing substantial cross-model disagreement and highlighting the challenge of operationalizing neologism detection at scale. Manual annotation of all 1,021 candidates confirms that 599 (58.7{\%}) are genuine lexical innovations."
}Markdown (Informal)
[From 124 Million Tokens to 1,021 Neologisms: A Large-Scale Pipeline for Automatic Neologism Detection](https://preview.aclanthology.org/test-year-match/2026.neollm-1.1/) (Rossini & van der Plas, NeoLLM 2026)
ACL