@inproceedings{madhyastha-espana-bonet-2017-learning,
title = "Learning Bilingual Projections of Embeddings for Vocabulary Expansion in Machine Translation",
author = "Madhyastha, Pranava Swaroop and
Espa{\~n}a-Bonet, Cristina",
editor = "Blunsom, Phil and
Bordes, Antoine and
Cho, Kyunghyun and
Cohen, Shay and
Dyer, Chris and
Grefenstette, Edward and
Hermann, Karl Moritz and
Rimell, Laura and
Weston, Jason and
Yih, Scott",
booktitle = "Proceedings of the 2nd Workshop on Representation Learning for {NLP}",
month = aug,
year = "2017",
address = "Vancouver, Canada",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/jlcl-multiple-ingestion/W17-2617/",
doi = "10.18653/v1/W17-2617",
pages = "139--145",
abstract = "We propose a simple log-bilinear softmax-based model to deal with vocabulary expansion in machine translation. Our model uses word embeddings trained on significantly large unlabelled monolingual corpora and learns over a fairly small, word-to-word bilingual dictionary. Given an out-of-vocabulary source word, the model generates a probabilistic list of possible translations in the target language using the trained bilingual embeddings. We integrate these translation options into a standard phrase-based statistical machine translation system and obtain consistent improvements in translation quality on the English{--}Spanish language pair. When tested over an out-of-domain testset, we get a significant improvement of 3.9 BLEU points."
}
Markdown (Informal)
[Learning Bilingual Projections of Embeddings for Vocabulary Expansion in Machine Translation](https://preview.aclanthology.org/jlcl-multiple-ingestion/W17-2617/) (Madhyastha & España-Bonet, RepL4NLP 2017)
ACL