@article{martin-etal-2026-clevr,
title = "{CLEVR}-3{D}-{D}e{R}ef",
author = "Martin, Mary Lynn and
Palmer, Martha and
Pacheco, Maria Leonor",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
journal = "International Conference on Language Resources and Evaluation",
volume = "main",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/ingest-lrec/2026.lrec-main.745/",
pages = "9490--9503",
abstract = "Vision-language models (VLMs) often struggle to interpret spatial referring expressions that require relational reasoning rather than reliance on surface-level cues. These models frequently identify referents through explicit visual attributes such as color or shape, rather than understanding spatial relationships (e.g., ``to the left of the red cube''). To systematically analyze these limitations, we introduce CLEVR-3D-DeRef, a synthetic and extensible benchmark dataset modeled after CLEVR-Ref+, designed to evaluate spatial reasoning in multi-modal systems. CLEVR-3D-DeRef extends the original framework by incorporating depth information for 3D spatial reasoning, introducing de-identified context-dependent referring expressions that require relational inference to disambiguate referent objects, and expanding the range of spatial relations beyond the original four. We further extend our dataset by producing expressions with and without ordinal language and diversifying the language and structure of expressions while preserving meaning."
}Markdown (Informal)
[CLEVR-3D-DeRef](https://preview.aclanthology.org/ingest-lrec/2026.lrec-main.745/) (Martin et al., LREC 2026)
ACL
- Mary Lynn Martin, Martha Palmer, and Maria Leonor Pacheco. 2026. CLEVR-3D-DeRef. International Conference on Language Resources and Evaluation, main:9490–9503.