@inproceedings{root-hopkins-2023-williams,
title = "{W}illiams College{'}s Submission for the {C}oco4{MT} 2023 Shared Task",
author = "Root, Alex and
Hopkins, Mark",
booktitle = "Proceedings of the Second Workshop on Corpus Generation and Corpus Augmentation for Machine Translation",
month = sep,
year = "2023",
address = "Macau SAR, China",
publisher = "Asia-Pacific Association for Machine Translation",
url = "https://preview.aclanthology.org/fix-sig-urls/2023.mtsummit-coco4mt.4/",
pages = "28--32",
abstract = "Professional translation is expensive. As a consequence, when developing a translation system in the absence of a pre-existing parallel corpus, it is important to strategically choose sentences to have professionally translated for the training corpus. In our contribution to the Coco4MT 2023 Shared Task, we explore how sentence embeddings can be leveraged to choose an impactful set of sentences to translate. Based on six language pairs of the JHU Bible corpus, we demonstrate that a technique based on SimCSE embeddings outperforms a competitive suite of baselines."
}
Markdown (Informal)
[Williams College’s Submission for the Coco4MT 2023 Shared Task](https://preview.aclanthology.org/fix-sig-urls/2023.mtsummit-coco4mt.4/) (Root & Hopkins, MTSummit 2023)
ACL