@inproceedings{touchent-de-la-clergerie-2026-ontobook,
title = "{O}nto{B}ook: Ontology-Grounded Synthetic Textbooks for Medical Encoder Pretraining",
author = "Touchent, Rian and
de la Clergerie, {\'E}ric",
editor = "S{\'e}rasset, Gilles and
Gkirtzou, Katerina and
Cochez, Michael and
Kalo, Jan-Christoph",
booktitle = "Proceedings of the Knowledge Graphs and Large Language Models Workshop ({KG}-{LLM}) @ {LREC}26",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/cawl-year/2026.kallm-1.2/",
doi = "10.63317/37ik5npnrvv6",
pages = "11--19",
abstract = "We present OntoBook, a method that converts medical ontology structure into pretraining signal for encoder language models. Our approach has three stages: random walks through ontology graphs capture hierarchical and causal relations between medical codes, a large language model reformulates these walks into fluent textbook-style prose, and the resulting text is used to train ModernCamemBERT, a 149M-parameter French encoder, with two objectives on the same data: masked language modeling and relation prediction between code pairs. On three French medical coding benchmarks (FRACCO, Cantemist-FR, Distemist-FR), OntoBook achieves significant improvements over MLM-only pretraining, with +2.5 micro-F1 on FRACCO and +8.0 micro-F1 on Distemist. We find that alignment between objectives is necessary: misaligned training, where each task uses different data, causes a 30-point degradation. We release 1.3 million LLM-reformulated medical textbooks across three French ontologies (CIM-10, CCAM, ATC) and pretrained model checkpoints."
}Markdown (Informal)
[OntoBook: Ontology-Grounded Synthetic Textbooks for Medical Encoder Pretraining](https://preview.aclanthology.org/cawl-year/2026.kallm-1.2/) (Touchent & de la Clergerie, KaLLM 2026)
ACL