@inproceedings{abuin-garcia-2025-wic,
title = "{W}i{C} Evaluation in {G}alician and {S}panish: Effects of Dataset Quality and Composition",
author = "Abu{\'i}n, Marta V{\'a}zquez and
Garcia, Marcos",
editor = "Frermann, Lea and
Stevenson, Mark",
booktitle = "Proceedings of the 14th Joint Conference on Lexical and Computational Semantics (*SEM 2025)",
month = nov,
year = "2025",
address = "Suzhou, China",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/ingest-emnlp/2025.starsem-1.13/",
pages = "172--178",
ISBN = "979-8-89176-340-1",
abstract = "This work explores the impact of dataset quality and composition on Word-in-Context performance for Galician and Spanish. We assess existing datasets, validate their test sets, and create new manually constructed evaluation data. Across five experiments with controlled variations in training and test data, we find that while the validation of test data tends to yield better model performance, evaluations on manually created datasets suggest that contextual embeddings are not sufficient on their own to reliably capture word meaning variation. Regarding training data, our results suggest that performance is influenced not only by size and human validation but also by deeper factors related to the semantic properties of the datasets. All new resources will be freely released."
}Markdown (Informal)
[WiC Evaluation in Galician and Spanish: Effects of Dataset Quality and Composition](https://preview.aclanthology.org/ingest-emnlp/2025.starsem-1.13/) (Abuín & Garcia, *SEM 2025)
ACL