@inproceedings{cho-etal-2020-open,
title = "Open {K}orean Corpora: A Practical Report",
author = "Cho, Won Ik and
Moon, Sangwhan and
Song, Youngsook",
editor = "Park, Eunjeong L. and
Hagiwara, Masato and
Milajevs, Dmitrijs and
Liu, Nelson F. and
Chauhan, Geeticka and
Tan, Liling",
booktitle = "Proceedings of Second Workshop for NLP Open Source Software (NLP-OSS)",
month = nov,
year = "2020",
address = "Online",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/jlcl-multiple-ingestion/2020.nlposs-1.12/",
doi = "10.18653/v1/2020.nlposs-1.12",
pages = "85--93",
abstract = "Korean is often referred to as a low-resource language in the research community. While this claim is partially true, it is also because the availability of resources is inadequately advertised and curated. This work curates and reviews a list of Korean corpora, first describing institution-level resource development, then further iterate through a list of current open datasets for different types of tasks. We then propose a direction on how open-source dataset construction and releases should be done for less-resourced languages to promote research."
}
Markdown (Informal)
[Open Korean Corpora: A Practical Report](https://preview.aclanthology.org/jlcl-multiple-ingestion/2020.nlposs-1.12/) (Cho et al., NLPOSS 2020)
ACL
- Won Ik Cho, Sangwhan Moon, and Youngsook Song. 2020. Open Korean Corpora: A Practical Report. In Proceedings of Second Workshop for NLP Open Source Software (NLP-OSS), pages 85–93, Online. Association for Computational Linguistics.