@inproceedings{huang-etal-2010-predicting,
title = "Predicting Morphological Types of {C}hinese Bi-Character Words by Machine Learning Approaches",
author = "Huang, Ting-Hao and
Ku, Lun-Wei and
Chen, Hsin-Hsi",
editor = "Calzolari, Nicoletta and
Choukri, Khalid and
Maegaard, Bente and
Mariani, Joseph and
Odijk, Jan and
Piperidis, Stelios and
Rosner, Mike and
Tapias, Daniel",
booktitle = "Proceedings of the Seventh International Conference on Language Resources and Evaluation ({LREC}'10)",
month = may,
year = "2010",
address = "Valletta, Malta",
publisher = "European Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/test-year-match/L10-1274/",
abstract = {This paper presented an overview of Chinese bi-character words' morphological types, and proposed a set of features for machine learning approaches to predict these types based on composite characters' information. First, eight morphological types were defined, and 6,500 Chinese bi-character words were annotated with these types. After pre-processing, 6,178 words were selected to construct a corpus named Reduced Set. We analyzed Reduced Set and conducted the inter-annotator agreement test. The average kappa value of 0.67 indicates a substantial agreement. Second, Bi-character words' morphological types are considered strongly related with the composite characters' parts of speech in this paper, so we proposed a set of features which can simply be extracted from dictionaries to indicate the characters' ``tendency'' of parts of speech. Finally, we used these features and adopted three machine learning algorithms, SVM, CRF, and Na{\"i}ve Bayes, to predict the morphological types. On the average, the best algorithm CRF achieved 75{\%} of the annotators' performance.}
}Markdown (Informal)
[Predicting Morphological Types of Chinese Bi-Character Words by Machine Learning Approaches](https://preview.aclanthology.org/test-year-match/L10-1274/) (Huang et al., LREC 2010)
ACL