@inproceedings{goel-etal-2026-auditing,
title = "Auditing Language Model Unlearning via Information Decomposition",
author = "Goel, Anmol and
Ritter, Alan and
Gurevych, Iryna",
editor = "Demberg, Vera and
Inui, Kentaro and
Marquez, Llu{\'i}s",
booktitle = "Proceedings of the 19th Conference of the {E}uropean Chapter of the {A}ssociation for {C}omputational {L}inguistics (Volume 1: Long Papers)",
month = mar,
year = "2026",
address = "Rabat, Morocco",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/ingest-eacl/2026.eacl-long.35/",
pages = "808--826",
ISBN = "979-8-89176-380-7",
abstract = "We expose a critical limitation in current approaches to machine unlearning in language models: despite the apparent success of unlearning algorithms, information about the forgotten data remains linearly decodable from internal representations. To systematically assess this discrepancy, we introduce an interpretable, information-theoretic framework for auditing unlearning using Partial Information Decomposition (PID). By comparing model representations before and after unlearning, we decompose the mutual information with the forgotten data into distinct components, formalizing the notions of unlearned and residual knowledge. Our analysis reveals that redundant information, shared across both models, constitutes residual knowledge that persists post-unlearning and correlates with susceptibility to known adversarial reconstruction attacks. Leveraging these insights, we propose a representation-based risk score that can guide abstention on sensitive inputs at inference time, providing a practical mechanism to mitigate privacy leakage. Our work introduces a principled, representation-level audit for unlearning, offering theoretical insight and actionable tools for safer deployment of language models."
}Markdown (Informal)
[Auditing Language Model Unlearning via Information Decomposition](https://preview.aclanthology.org/ingest-eacl/2026.eacl-long.35/) (Goel et al., EACL 2026)
ACL
- Anmol Goel, Alan Ritter, and Iryna Gurevych. 2026. Auditing Language Model Unlearning via Information Decomposition. In Proceedings of the 19th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers), pages 808–826, Rabat, Morocco. Association for Computational Linguistics.