@inproceedings{stein-2014-parsing,
title = "Parsing Heterogeneous Corpora with a Rich Dependency Grammar",
author = "Stein, Achim",
editor = "Calzolari, Nicoletta and
Choukri, Khalid and
Declerck, Thierry and
Loftsson, Hrafn and
Maegaard, Bente and
Mariani, Joseph and
Moreno, Asuncion and
Odijk, Jan and
Piperidis, Stelios",
booktitle = "Proceedings of the Ninth International Conference on Language Resources and Evaluation ({LREC}'14)",
month = may,
year = "2014",
address = "Reykjavik, Iceland",
publisher = "European Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/fix-sig-urls/L14-1227/",
pages = "2879--2886",
abstract = "Grammar models conceived for parsing purposes are often poorer than models that are motivated linguistically. We present a grammar model which is linguistically satisfactory and based on the principles of traditional dependency grammar. We show how a state-of-the-art dependency parser (mate tools) performs with this model, trained on the Syntactic Reference Corpus of Medieval French (SRCMF), a manually annotated corpus of medieval (Old French) texts. We focus on the problems caused by small and heterogeneous training sets typical for corpora of older periods. The result is the first publicly available dependency parser for Old French. On a 90/10 training/evaluation split of eleven OF texts (206000 words), we obtained an UAS of 89.68{\%} and a LAS of 82.62{\%}. Three experiments showed how heterogeneity, typical of medieval corpora, affects the parsing results: (a) a `one-on-one' cross evaluation for individual texts, (b) a `leave-one-out' cross evaluation, and (c) a prose/verse cross evaluation."
}
Markdown (Informal)
[Parsing Heterogeneous Corpora with a Rich Dependency Grammar](https://preview.aclanthology.org/fix-sig-urls/L14-1227/) (Stein, LREC 2014)
ACL