@inproceedings{singh-etal-2026-max,
title = "{MAX}-{EVAL}-11: A Large Scale Benchmark for Evaluating Large Language Models on Full-Spectrum {ICD}-11 Medical Coding",
author = "Singh, Ujjwal and
Deshwal, Sarthak and
Dube, Nitish and
Sharma, Arjun",
editor = "Demner-Fushman, Dina and
Ananiadou, Sophia and
Roberts, Kirk and
Tsujii, Junichi",
booktitle = "{B}io{NLP} 2026",
month = jul,
year = "2026",
address = "San Diego, California",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/corrections-2026-06/2026.bionlp-1.23/",
doi = "10.18653/v1/2026.bionlp-1.23",
pages = "282--291",
ISBN = "979-8-89176-434-7",
abstract = "The global transition to the ICD-11 taxonomy demands robust automated medical coding, yet comprehensive benchmarks to evaluate Large Language Models (LLMs) on this task remain absent. We introduce MAX-EVAL-11, the first large-scale benchmark for full-spectrum ICD-11 medical coding. MAX-EVAL-11 comprises 10,000 MIMIC-III discharge summaries with mapped, expert-validated ICD-11 annotations spanning 99.87{\textbackslash}{\%} of the diagnostic taxonomy. To better reflect clinical utility, we propose a novel hierarchical evaluation framework that assigns partial credit based on ICD-11{'}s 5-level structure, addressing the brittleness of traditional exact-match metrics. Our evaluation of state-of-the-art LLMs reveals significant performance gaps. The best-performing model (Claude 4 Sonnet) achieves a weighted score of 0.433, outperforming both general-purpose peers and specialized medical models (MedCoder). Crucially, all models exhibit near-zero exact match rates (0?4.8{\textbackslash}{\%}) and rely primarily on hierarchical credit, underscoring the extreme difficulty of precise ICD-11 code generation. Furthermore, the superiority of general-purpose LLMs over legacy ICD-10 medical models (with ICD-11 codelist) suggests that broad reasoning capabilities currently outweigh domain-specific training for complex taxonomy scaling."
}Markdown (Informal)
[MAX-EVAL-11: A Large Scale Benchmark for Evaluating Large Language Models on Full-Spectrum ICD-11 Medical Coding](https://preview.aclanthology.org/corrections-2026-06/2026.bionlp-1.23/) (Singh et al., BioNLP 2026)
ACL