@inproceedings{degraeuwe-moerman-2026-shanel,
title = "{S}h{A}n{EL}-2: A Multilingual Benchmarking Dataset for Short-Answer Language Learning Exercises",
author = "Degraeuwe, Jasper and
Moerman, Thomas",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/cawl-year/2026.lrec-1.538/",
doi = "10.63317/3cvfqh22muoo",
pages = "6764--6771",
abstract = "Before using GenAI models as EdTech tools, their pedagogical suitability should be corroborated. In this paper, we present ShAnEL-2, a novel multilingual dataset comprising 1,185 student responses to short-answer language learning exercises corrected by teachers. We use ShAnEL-2 to establish an initial benchmark of (1) ``off-the-shelf'' GenAI models and (2) retrieval-augmented generation (RAG) techniques for the automated correction of this exercise type. With an overall accuracy of 90{\%} and recall of 95{\%}, few-shot RAG (which adds previously corrected responses to the prompt) outperforms the off-the-shelf baseline and textbook RAG setup (which adds coursebook materials) by up to 7 (accuracy) and 5 (recall) percentage points. These results confirm that LLMs learn better from examples than from analysing context and highlight GenAI{'}s particular potential as a correction assistant for teachers."
}Markdown (Informal)
[ShAnEL-2: A Multilingual Benchmarking Dataset for Short-Answer Language Learning Exercises](https://preview.aclanthology.org/cawl-year/2026.lrec-1.538/) (Degraeuwe & Moerman, LREC 2026)
ACL