@inproceedings{haiber-etal-2026-domain,
title = "Domain-Specific Considerations in the Preparation of Specialized Corpora: A Case Study on a Corpus of {G}erman Sermons",
author = "Haiber, Cora and
Roussel, Adam and
Dipper, Stefanie",
editor = "Hinrichs, Erhard and
Nivre, Joakim and
Osenova, Petya and
Pustejovsky, James and
Zinn, Claus",
booktitle = "Proceedings of the Workshop on Structured Linguistic Data and Evaluation ({SL}i{DE})",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/revision-previews/2026.slide-1.19/",
doi = "10.63317/5nhkpqtxbr6i",
pages = "212--223",
abstract = "We present a new corpus of contemporary German sermons and describe the steps taken in its preparation. We apply a semi-automatic approach to sentence segmentation, tokenization, and lemmatization, utilizing annotation guidelines that are specialized to this domain. In the process of preparing these data, we find that state-of-the-art tools for these tasks still make problematic errors, especially with non-standard data, despite apparently very high performance on common benchmarks. We obtain test scores of F1 = 96.69 {\%} for sentence segmentation, F1 = 99.99 {\%} for tokenization, and acc = 64.00 {\%} for lemmatization with our domain-adapted models and show that domain-adaptation improves performance over state-of-the-art models for the token and sentence segmentation tasks."
}Markdown (Informal)
[Domain-Specific Considerations in the Preparation of Specialized Corpora: A Case Study on a Corpus of German Sermons](https://preview.aclanthology.org/revision-previews/2026.slide-1.19/) (Haiber et al., SLiDE 2026)
ACL