@inproceedings{taguchi-etal-2022-universal,
title = "{U}niversal {D}ependencies Treebank for {T}atar: Incorporating Intra-Word Code-Switching Information",
author = "Taguchi, Chihiro and
Iwata, Sei and
Watanabe, Taro",
editor = "Ojha, Atul Kr. and
Ahmadi, Sina and
Liu, Chao-Hong and
McCrae, John P.",
booktitle = "Proceedings of the Workshop on Resources and Technologies for Indigenous, Endangered and Lesser-resourced Languages in Eurasia within the 13th Language Resources and Evaluation Conference",
month = jun,
year = "2022",
address = "Marseille, France",
publisher = "European Language Resources Association",
url = "https://preview.aclanthology.org/jlcl-multiple-ingestion/2022.eurali-1.17/",
pages = "95--104",
abstract = "This paper introduces a new Universal Dependencies treebank for the Tatar language named NMCTT. A significant feature of the corpus is that it includes code-switching (CS) information at a morpheme level, given the fact that Tatar texts contain intra-word CS between Tatar and Russian. We first outline NMCTT with a focus on differences from other treebanks of Turkic languages. Then, to evaluate the merit of the CS annotation, this study concisely reports the results of a language identification task implemented with Conditional Random Fields that considers POS tag information, which is readily available in treebanks in the CoNLL-U format. Experimenting on NMCTT and the Turkish-German CS treebank (SAGT), we demonstrate that the proposed annotation scheme introduced in NMCTT can improve the performance of the subword-level language identification. This annotation scheme for CS is not only universally applicable to languages with CS, but also shows a possibility to employ morphosyntactic information for CS-related downstream tasks."
}
Markdown (Informal)
[Universal Dependencies Treebank for Tatar: Incorporating Intra-Word Code-Switching Information](https://preview.aclanthology.org/jlcl-multiple-ingestion/2022.eurali-1.17/) (Taguchi et al., EURALI 2022)
ACL