@inproceedings{esmaeil-etal-2026-tantaarabnlp,
title = "{T}anta{A}rab{NLP} at {KSAA}-2026 Task 2: Adapting {CATT}-Whisper for {A}rabic Speech Dictation with Automatic Diacritization",
author = "Esmaeil, Nada Adel and
Elbasiony, Reda M. and
Faheem, Mohamed T.",
editor = "Al-Khalifa, Hend and
El-Haj, Mo and
Ezzini, Saad",
booktitle = "The 7th Workshop on Open-Source {A}rabic Corpora and Processing Tools ({OSACT}7) with 5 Shared Tasks",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/revision-workflow/2026.osact-1.30/",
doi = "10.63317/46cm97fcekow",
pages = "229--233",
abstract = "We present our submission to the KSAA-2026 Shared Task (Subtask 2): Automatic Diacritization of Speech Dictation. Building upon the CATT-Whisper multimodal architecture, which fuses representations from a pre-trained CATT text encoder and the Whisper speech encoder, we fine-tune the model end-to-end on the official shared task training data. To further enhance performance on speech-dictated Arabic text, we apply careful post-processing to the model outputs. Our best submission achieves a Diacritic Error Rate (DER) of 7.04, a Word Error Rate (WER) of 24.39, and a Sentence Error Rate (SER) of 71.65 on the hidden test set, securing 2nd place in the competition. These results demonstrate the effectiveness of adapting a strong multimodal baseline to the speech-aware diacritization setting and highlight the value of task-specific fine-tuning and output refinement for bridging the gap between spoken transcripts and fully diacritized Arabic text."
}Markdown (Informal)
[TantaArabNLP at KSAA-2026 Task 2: Adapting CATT-Whisper for Arabic Speech Dictation with Automatic Diacritization](https://preview.aclanthology.org/revision-workflow/2026.osact-1.30/) (Esmaeil et al., OSACT 2026)
ACL