@inproceedings{lalitha-devi-etal-2026-integrating,
title = "Integrating Syntactic and Discourse Signals through Multi-Encoder Fusion in {NMT} for Low-Resource {I}ndian Language Pairs",
author = "Lalitha Devi, Sobha and
Sundar Ram, Vijay and
RK Rao, Pattabhi",
editor = "Jha, Girish Nath and
Bali, Kalika and
L, Sobha and
Kumar, Devendr",
booktitle = "Proceedings of the 8th Workshop on {I}ndian Language Data: Resources and Evaluation",
month = may,
year = "2026",
address = "Palma, Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/paragraph-normalization/2026.wildre-1.13/",
doi = "10.63317/24vtvyv2iqhs",
pages = "98--103",
abstract = "Neural Machine Translation (NMT) for low-resource Indian language pairs such as Hindi{--}Tamil and Tamil{--}Malayalam remains challenging due to morphological richness, syntactic divergence, and limited availability of high-quality parallel corpora. While Transformer-based architectures achieve strong performance in high-resource settings, they often struggle to model syntactic structure and discourse-level dependencies in low-resource scenarios, resulting in errors in agreement, word order, and pronoun translation. In this work, we propose a linguistically informed multi-encoder fusion framework that explicitly incorporates syntactic and discourse signals into NMT. Experiments conducted on Hindi{--}Tamil and Tamil{--}Malayalam parallel corpora demonstrate consistent improvements over strong Transformer baselines in BLEU and ChrF scores, along with gains in pronoun translation accuracy and agreement consistency. The results highlight the effectiveness of explicit linguistic integration for improving NMT in low-resource Indian language settings."
}Markdown (Informal)
[Integrating Syntactic and Discourse Signals through Multi-Encoder Fusion in NMT for Low-Resource Indian Language Pairs](https://preview.aclanthology.org/paragraph-normalization/2026.wildre-1.13/) (Lalitha Devi et al., WILDRE 2026)
ACL