@inproceedings{ye-bloem-2026-lost,
title = "Lost in Translation: Repurposing semantic similarity benchmarks for evaluating lexical-semantic consistency in {LLM}-based machine translation",
author = "Ye, Quin and
Bloem, Jelke",
editor = "Morger, Felix and
Ilinykh, Nikolai and
Scalvini, Barbara and
Dobnik, Simon and
Dann{\'e}lls, Dana",
booktitle = "Proceedings of the Fourth Workshop on the Role of Resources in the Age of Large Language Models ({RESOURCEFUL} 2026)",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/revision-workflow/2026.resourceful-4.1/",
doi = "10.63317/3n3847mvjzk2",
pages = "1--12",
abstract = "We propose and demonstrate a repurposing of the lexical similarity benchmark Multi-SimLex and the SimLex-999 family of resources for assessing the cross-lingual lexical-semantic consistency of multilingual large language models. While originally gathered for evaluating word embedding models, the parallel nature of the word pairs enables their use in machine translation settings. Using a manually verified subset of 500 word pairs from the Multi-SimLex dataset, we evaluate models' ability to assess semantic similarity and perform translation between English and Mandarin through zero-shot prompting. We compare BLOOMZ and GPT-4{'}s similarity ratings against human-annotated benchmarks and examine translation consistency using our and other metrics, with GPT-4 showing stronger human alignment. As SimLex-999 and Multi-SimLex together cover a range of at least 25 languages, this approach has the potential to be extended to many language pairs including ones that don{'}t involve English, though it requires some manual checks."
}Markdown (Informal)
[Lost in Translation: Repurposing semantic similarity benchmarks for evaluating lexical-semantic consistency in LLM-based machine translation](https://preview.aclanthology.org/revision-workflow/2026.resourceful-4.1/) (Ye & Bloem, RESOURCEFUL 2026)
ACL