@inproceedings{kucera-etal-2026-towards,
title = "Towards Semantic Searching in Diverse Multimodal Collections",
author = "Ku{\v{c}}era, V{\'a}clav and
Bul{\'i}n, Martin and
{\v{S}}vec, Jan and
Ircing, Pavel",
editor = "Anuradha, Isuri and
Wynne, Martin",
booktitle = "Proceedings of The Second Workshop on Holocaust Testimonies as Language Resources ({HTR}es)",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/paragraph-normalization/2026.htres-2.2/",
doi = "10.63317/5cmrq8mhnrph",
pages = "12--19",
abstract = "Digital humanities projects increasingly rely on heterogeneous collections of multimodal data, including video testimonies, scanned documents, and photographs. Despite the growing availability of such archives, researchers face challenges in efficiently locating relevant content due to the diversity of formats and the lack of unified retrieval methods. In this work, we present a general framework for semantic search over collections of multiple modalities. The framework integrates specific parsers and transforms all inputs into textual representations leveraging services like automatic speech recognition (ASR), optical character recognition (OCR), and generative-AI-based image captioning. Text is subsequently segmented into overlapping chunks, indexed in a vector database, and enriched through an automatic question generation (AQ) pipeline to create ground-truth queries for evaluation. We evaluate the framework on a constructed dataset derived from Holocaust-related archives, comparing two retrieval strategies (pure vector search vs. hybrid semantic-lexical search) under two chunking scenarios. Results demonstrate that hybrid search consistently outperforms vector-only retrieval, achieving high recall across modalities, and that semantic search is feasible even with diverse and noisy input sources. This framework provides a robust foundation for exploring complex multimodal archives, facilitating access to content that would otherwise remain difficult to discover."
}Markdown (Informal)
[Towards Semantic Searching in Diverse Multimodal Collections](https://preview.aclanthology.org/paragraph-normalization/2026.htres-2.2/) (Kučera et al., htres 2026)
ACL
- Václav Kučera, Martin Bulín, Jan Švec, and Pavel Ircing. 2026. Towards Semantic Searching in Diverse Multimodal Collections. In Proceedings of The Second Workshop on Holocaust Testimonies as Language Resources (HTRes), pages 12–19, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).