@inproceedings{poncelas-etal-2020-using,
title = "Using Multiple Subwords to Improve {E}nglish-{E}speranto Automated Literary Translation Quality",
author = "Poncelas, Alberto and
Buts, Jan and
Hadley, James and
Way, Andy",
editor = "Karakanta, Alina and
Ojha, Atul Kr. and
Liu, Chao-Hong and
Abbott, Jade and
Ortega, John and
Washington, Jonathan and
Oco, Nathaniel and
Lakew, Surafel Melaku and
Pirinen, Tommi A and
Malykh, Valentin and
Logacheva, Varvara and
Zhao, Xiaobing",
booktitle = "Proceedings of the 3rd Workshop on Technologies for MT of Low Resource Languages",
month = dec,
year = "2020",
address = "Suzhou, China",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/jlcl-multiple-ingestion/2020.loresmt-1.14/",
doi = "10.18653/v1/2020.loresmt-1.14",
pages = "108--117",
abstract = "Building Machine Translation (MT) systems for low-resource languages remains challenging. For many language pairs, parallel data are not widely available, and in such cases MT models do not achieve results comparable to those seen with high-resource languages. When data are scarce, it is of paramount importance to make optimal use of the limited material available. To that end, in this paper we propose employing the same parallel sentences multiple times, only changing the way the words are split each time. For this purpose we use several Byte Pair Encoding models, with various merge operations used in their configuration. In our experiments, we use this technique to expand the available data and improve an MT system involving a low-resource language pair, namely English-Esperanto. As an additional contribution, we made available a set of English-Esperanto parallel data in the literary domain."
}
Markdown (Informal)
[Using Multiple Subwords to Improve English-Esperanto Automated Literary Translation Quality](https://preview.aclanthology.org/jlcl-multiple-ingestion/2020.loresmt-1.14/) (Poncelas et al., LoResMT 2020)
ACL