@inproceedings{bingert-etal-2026-benchmarking,
title = "Benchmarking {LLM}s for {ARR} Area Assignment: Evidence and Implications for Assignment Strategies",
author = "Bingert, Eileen and
Alves, Diego and
Degaetano-Ortlieb, Stefania",
editor = "Rehm, Georg and
Dietze, Stefan and
Dessi, Danilo and
Maynard, Diana and
Schimmler, Sonja",
booktitle = "Proceedings of Natural Scientific Language Processing ({NSLP}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/paragraph-normalization/2026.nslp-1.2/",
doi = "10.63317/4jbm8o3us4c8",
pages = "13--24",
abstract = "We study how large language models (LLMs) perform at assigning ACL Rolling Review (ARR) areas from paper titles/abstracts. Using 558 papers (ACL/EACL/NAACL, 2020 to 2025), we compare multiple LLMs and prompting schemes (zero/few-shot; with/without ARR keywords; each-category variants) and analyze per-area scores, error overlap, and confusion matrices. One-shot prompting (with OpenAI-gpt-oss-20b) tends to perform best, while injecting ARR keywords often lowers accuracy. Task-bounded areas (e.g., MT, IE, QA, Summarization) are predicted more reliably, whereas broad, cross-cutting labels (e.g., Resources and Evaluation, NLP Applications) are frequently conflated, indicating taxonomy ambiguity rather than solely model limitations. We recommend hierarchical or primary-plus-secondary labels to reduce ambiguity and improve reviewer matching. Our dataset, methods, and findings offer a reproducible baseline for area selection support in ACL workflows."
}Markdown (Informal)
[Benchmarking LLMs for ARR Area Assignment: Evidence and Implications for Assignment Strategies](https://preview.aclanthology.org/paragraph-normalization/2026.nslp-1.2/) (Bingert et al., NSLP 2026)
ACL