@article{lemay-etal-2026-corpusclues,
title = "{C}orpus{C}lues: Scalable Unsupervised Similarity Search for Historical Texts Using {M}in{H}ash-{LSH}",
author = "Lemay, Paulien and
Bentein, Klaas and
Lefever, Els",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
journal = "International Conference on Language Resources and Evaluation",
volume = "main",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/ingest-lrec/2026.lrec-main.63/",
pages = "838--847",
abstract = "CorpusClues is a prototype web-based platform for large-scale, unsupervised clustering of textual data, designed to address the specific challenges of historical corpora. It leverages the well-established computational techniques of MinHash and Locality-Sensitive Hashing (LSH) at the character level in order to detect structural similarities between texts even when exact patterns diverge. This approach makes CorpusClues robust to orthographic variation, such as historical spelling differences, while remaining fast and language-agnostic, capable of processing large and heterogeneous corpora without relying on language-specific models or preprocessing. Researchers can explore resulting clusters through interactive visualizations and exportable data, gaining access to patterns that would otherwise require the slow and uncertain process of manual collation. Evaluation against labeled gold standards shows that the system consistently produces high-quality clustering, accurately reconstructing relationships between texts despite substantial orthographic variation. By combining computational efficiency with user-friendly design, CorpusClues provides an accessible yet rigorous means of uncovering formulaicity and textual transmission at scale, opening new possibilities for the study of historical textual traditions."
}Markdown (Informal)
[CorpusClues: Scalable Unsupervised Similarity Search for Historical Texts Using MinHash-LSH](https://preview.aclanthology.org/ingest-lrec/2026.lrec-main.63/) (Lemay et al., LREC 2026)
ACL