@inproceedings{haralambous-halpern-2026-comprehensive,
title = "A Comprehensive Full-Form Lexicon for {A}rabic {NLP} and Speech Technology",
author = "Haralambous, Yannis and
Halpern, Jack",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/paragraph-normalization/2026.lrec-1.108/",
doi = "10.63317/2gbvmmu4ix5e",
pages = "1382--1393",
abstract = "Natural Language Processing (NLP) applications require morphological data with precise grammatical attributes, while speech technology requires abundant phonemic and phonetic data. This presents a challenge for Arabic due to its abundant morphological, orthographic, and phonemic ambiguity in both MSA and its various dialects. Existing systems struggle with incomplete and unstructured web data, leading to suboptimal performance in both morphological analysis and speech applications. This paper presents ArabLEX, a full-form lexicon (includes all wordforms, i.e., fully inflected/cliticized members of a lexeme class) that addresses these issues by providing a large-scale database designed to enhance NLP accuracy. It comprises approximately 570 million entries with fully inflected forms and detailed morphological, phonetic, and orthographic attributes. ArabLEX serves as a foundational framework for developing comprehensive Arabic lexical resources for NLP, particularly for speech technology, as well as dialect databases."
}Markdown (Informal)
[A Comprehensive Full-Form Lexicon for Arabic NLP and Speech Technology](https://preview.aclanthology.org/paragraph-normalization/2026.lrec-1.108/) (Haralambous & Halpern, LREC 2026)
ACL