@inproceedings{ryu-yanaka-2026-vision,
title = "What Do Vision{--}Language Models Encode for Personalized Image Aesthetics Assessment?",
author = "Ryu, Koki and
Yanaka, Hitomi",
editor = "Liakata, Maria and
Moreira, Viviane P. and
Zhang, Jiajun and
Jurgens, David",
booktitle = "Findings of the {A}ssociation for {C}omputational {L}inguistics: {ACL} 2026",
month = jul,
year = "2026",
address = "San Diego, California, United States",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/ingest-acl/2026.findings-acl.1706/",
pages = "34146--34167",
ISBN = "979-8-89176-395-1",
abstract = "Personalized image aesthetics assessment (PIAA) is an important research problem with practical real-world applications. While methods based on vision-language models (VLMs) are promising candidates for PIAA, it remains unclear whether they internally encode rich, multi-level aesthetic attributes required for effective personalization. In this paper, we first analyze the internal representations of VLMs to examine the presence and distribution of such aesthetic attributes, and then leverage them for lightweight, individual-level personalization without model fine-tuning. Our analysis reveals that VLMs encode diverse aesthetic attributes that propagate into the language decoder layers. Building on these representations, we demonstrate that simple linear models can achieve effective personalized image aesthetics assessment. We further analyze how aesthetic information is transferred across layers in different VLM architectures and across image domains. Our findings provide insights into how VLMs can be utilized for modeling subjective, individual aesthetic preferences."
}Markdown (Informal)
[What Do Vision–Language Models Encode for Personalized Image Aesthetics Assessment?](https://preview.aclanthology.org/ingest-acl/2026.findings-acl.1706/) (Ryu & Yanaka, Findings 2026)
ACL