@inproceedings{bashendy-elsayed-2026-corruption,
title = "Corruption-Based Data Augmentation for {A}rabic Essay Scoring: A Preliminary Study on the Organization Trait",
author = "Bashendy, May Saed and
Elsayed, Tamer",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/cawl-year/2026.lrec-1.825/",
doi = "10.63317/5kt6mmyaumav",
pages = "10525--10531",
abstract = "Despite significant advances in Automated Essay Scoring (AES), progress in Arabic AES remains limited by the scarcity and imbalance of publicly available datasets. Manual curation of such data is labor-intensive and lacks scalability. To address this, we introduce COrE, a corruption-based data augmentation method that targets the organization trait of Arabic essays. COrE generates synthetic essays by intentionally disrupting the organization of well-written essays through controlled, distance-aware sentence swapping. Our experiments are conducted on TAQAE, a dataset of 620 essays across 4 distinct writing prompts. We evaluate the effectiveness of COrE using two widely-adopted pre-trained models: AraBERTv2 and CAMeLBERT-mix. Both models show improved performance with COrE, achieving gains of 9-17{\%} over the no-augmentation baseline. These results highlight the potential of trait-specific augmentation to address data scarcity and enhance AES performance for low-resource languages."
}Markdown (Informal)
[Corruption-Based Data Augmentation for Arabic Essay Scoring: A Preliminary Study on the Organization Trait](https://preview.aclanthology.org/cawl-year/2026.lrec-1.825/) (Bashendy & Elsayed, LREC 2026)
ACL