@inproceedings{carik-yeniterzi-2022-twitter,
title = "A {T}witter Corpus for Named Entity Recognition in {T}urkish",
author = "{\c{C}}ar{\i}k, Buse and
Yeniterzi, Reyyan",
editor = "Calzolari, Nicoletta and
B{\'e}chet, Fr{\'e}d{\'e}ric and
Blache, Philippe and
Choukri, Khalid and
Cieri, Christopher and
Declerck, Thierry and
Goggi, Sara and
Isahara, Hitoshi and
Maegaard, Bente and
Mariani, Joseph and
Mazo, H{\'e}l{\`e}ne and
Odijk, Jan and
Piperidis, Stelios",
booktitle = "Proceedings of the Thirteenth Language Resources and Evaluation Conference",
month = jun,
year = "2022",
address = "Marseille, France",
publisher = "European Language Resources Association",
url = "https://preview.aclanthology.org/add-emnlp-2024-awards/2022.lrec-1.484/",
pages = "4546--4551",
abstract = "This paper introduces a new Turkish Twitter Named Entity Recognition dataset. The dataset, which consists of 5000 tweets from a year-long period, was labeled by multiple annotators with a high agreement score. The dataset is also diverse in terms of the named entity types as it contains not only person, organization, and location but also time, money, product, and tv-show categories. Our initial experiments with pretrained language models (like BertTurk) over this dataset returned F1 scores of around 80{\%}. We share this dataset publicly."
}
Markdown (Informal)
[A Twitter Corpus for Named Entity Recognition in Turkish](https://preview.aclanthology.org/add-emnlp-2024-awards/2022.lrec-1.484/) (Çarık & Yeniterzi, LREC 2022)
ACL