@inproceedings{kumar-2014-developing,
title = "Developing Politeness Annotated Corpus of {H}indi Blogs",
author = "Kumar, Ritesh",
editor = "Calzolari, Nicoletta and
Choukri, Khalid and
Declerck, Thierry and
Loftsson, Hrafn and
Maegaard, Bente and
Mariani, Joseph and
Moreno, Asuncion and
Odijk, Jan and
Piperidis, Stelios",
booktitle = "Proceedings of the Ninth International Conference on Language Resources and Evaluation ({LREC}'14)",
month = may,
year = "2014",
address = "Reykjavik, Iceland",
publisher = "European Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/landing_page/L14-1480/",
pages = "1275--1280",
abstract = "In this paper I discuss the creation and annotation of a corpus of Hindi blogs. The corpus consists of a total of over 479,000 blog posts and blog comments. It is annotated with the information about the politeness level of each blog post and blog comment. The annotation is carried out using four levels of politeness {\textemdash} neutral, appropriate, polite and impolite. For the annotation, three classifiers {\textemdash} were trained and tested maximum entropy (MaxEnt), Support Vector Machines (SVM) and C4.5 - using around 30,000 manually annotated texts. Among these, C4.5 gave the best accuracy. It achieved an accuracy of around 78{\%} which is within 2{\%} of the human accuracy during annotation. Consequently this classifier is used to annotate the rest of the corpus"
}
Markdown (Informal)
[Developing Politeness Annotated Corpus of Hindi Blogs](https://preview.aclanthology.org/landing_page/L14-1480/) (Kumar, LREC 2014)
ACL