@inproceedings{koeva-stoyanova-2026-large,
title = "A Large Dataset Representing {B}ulgarian, with the {B}ulgarian National Corpus as Its Core",
author = "Koeva, Svetla Peneva and
Stoyanova, Ivelina",
editor = "Ba{\'n}ski, Piotr and
Knight, Dawn and
Kupietz, Marc and
Witt, Andreas and
Wr{\'o}blewska, Alina",
booktitle = "Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/test-year-match/2026.cmlc-1.2/",
doi = "10.63317/4gxqpsnk5jz5",
pages = "12--24",
abstract = "The paper introduces the IfGPT dataset, which integrates several Bulgarian text collections, including the Bulgarian National Corpus, and applies cleaning, deduplication, and LLM-oriented metadata such as personally identifiable information and bias scores. The composition of the IfGPT dataset is presented, along with the unified metadata schema and metadata management in a graph database, enabling efficient querying and document selection for specific tasks. The main contributions are the integration of multiple Bulgarian text collections into a unified dataset, the development of a standardised metadata schema with graph-based organisation, and the provision of efficient metadata querying mechanisms to support LLM development."
}Markdown (Informal)
[A Large Dataset Representing Bulgarian, with the Bulgarian National Corpus as Its Core](https://preview.aclanthology.org/test-year-match/2026.cmlc-1.2/) (Koeva & Stoyanova, CMLC 2026)
ACL