@inproceedings{da-huang-etal-2026-magicbench,
title = "{M}agic{B}ench: Diagnosing Visual Agency Loss and Semantic Dependency in Multimodal {LLM}s",
author = "Da Huang, Tang and
Tang, Weidong and
Xu, Wen Qi and
Guo, Xianpeng",
editor = "Liakata, Maria and
Moreira, Viviane P. and
Zhang, Jiajun and
Jurgens, David",
booktitle = "Proceedings of the 64th Annual Meeting of the {A}ssociation for {C}omputational {L}inguistics (Volume 1: Long Papers)",
month = jul,
year = "2026",
address = "San Diego, California, United States",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/ingest-acl/2026.acl-long.1314/",
pages = "28493--28511",
ISBN = "979-8-89176-390-6",
abstract = "Multimodal Large Language Models typically assume linguistic context invariably enhances visual understanding. We study this assumption in semantic adversarial scenarios, specifically magic tricks, where narration deliberately diverges from physical reality. We introduce MagicBench, a diagnostic benchmark of 402 videos for evaluating MLLMs under hierarchical linguistic interference, together with a Physical Constraint Set (PCS) protocol for assessing adherence to physical laws. Evaluation uncovers a Semantic Dependency Paradox: (1) Semantic anchoring: Entity nouns act as anchors aiding localization, paradoxically boosting performance despite false predicates. (2) Visual Agency Loss: In semantic vacuums, multimodal performance collapses 12.4{\%} (p {\ensuremath{<}} 0.01) below the vision-only ``capability probe''. This gap persists under symmetric prompting, suggesting a form of functional perception suppression in which autonomous visual search may be under-utilized in multimodal settings without linguistic triggers. Causal interventions via spatial prompting and signal magnification provide evidence that internal reasoning remains functional, supporting the interpretation of a perceptual access bottleneck. Our findings suggest MLLMs function as ``language-guided passive observers'', advocating for perceptually-independent architectures that decouple sensory agency from linguistic dominance. Code and dataset are available at https://github.com/Ink-Dawn/MagicBench"
}Markdown (Informal)
[MagicBench: Diagnosing Visual Agency Loss and Semantic Dependency in Multimodal LLMs](https://preview.aclanthology.org/ingest-acl/2026.acl-long.1314/) (Da Huang et al., ACL 2026)
ACL