@inproceedings{mieskes-benz-2023-h,
title = "h{\_}da@{R}epro{H}umn {--} Reproduction of Human Evaluation and Technical Pipeline",
author = "Mieskes, Margot and
Benz, Jacob Georg",
editor = "Belz, Anya and
Popovi{\'c}, Maja and
Reiter, Ehud and
Thomson, Craig and
Sedoc, Jo{\~a}o",
booktitle = "Proceedings of the 3rd Workshop on Human Evaluation of NLP Systems",
month = sep,
year = "2023",
address = "Varna, Bulgaria",
publisher = "INCOMA Ltd., Shoumen, Bulgaria",
url = "https://preview.aclanthology.org/fix-sig-urls/2023.humeval-1.11/",
pages = "130--135",
abstract = "How reliable are human evaluation results? Is it possible to replicate human evaluation? This work takes a closer look at the evaluation of the output of a Text-to-Speech (TTS) system. Unfortunately, our results indicate that human evaluation is not as straightforward to replicate as expected. Additionally, we also present results on reproducing the technical background of the TTS system and discuss potential reasons for the reproduction failure."
}
Markdown (Informal)
[h_da@ReproHumn – Reproduction of Human Evaluation and Technical Pipeline](https://preview.aclanthology.org/fix-sig-urls/2023.humeval-1.11/) (Mieskes & Benz, HumEval 2023)
ACL