@inproceedings{paik-wense-2025-effectiveness,
title = "The Effectiveness of Uncased Tokeniziaion for Clinical Notes",
author = "Paik, Cory and
Wense, Katharina Von Der",
editor = "Che, Wanxiang and
Nabende, Joyce and
Shutova, Ekaterina and
Pilehvar, Mohammad Taher",
booktitle = "Findings of the Association for Computational Linguistics: ACL 2025",
month = jul,
year = "2025",
address = "Vienna, Austria",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/display_plenaries/2025.findings-acl.775/",
pages = "14986--14992",
ISBN = "979-8-89176-256-5",
abstract = "The impact of case-sensitive tokenization on clinical notes is not well understood. While clinical notes share similarities with biomedical text in terminology, they often lack the proper casing found in polished publications. Language models, unlike humans, require a fixed vocabulary and case sensitivity is a trade-off that must be considered carefully. Improper casing can lead to sub-optimal tokenization and increased sequence length, degrading downstream performance and increasing computational costs. While most recent open-domain encoder language models use uncased tokenization for all tasks, there is no clear trend in biomedical and clinical models. In this work we (1) show that uncased models exceed the performance of cased models on clinical notes, even on traditionally case-sensitive tasks such as named entity recognition and (2) introduce independent case encoding to better balance model performance on case-sensitive and improperly-cased tasks."
}
Markdown (Informal)
[The Effectiveness of Uncased Tokeniziaion for Clinical Notes](https://preview.aclanthology.org/display_plenaries/2025.findings-acl.775/) (Paik & Wense, Findings 2025)
ACL