@inproceedings{jaf-2016-simple,
title = "A Simple Approach to Unifying Ambiguously Encoded {K}urdish Characters",
author = "Jaf, Sardar",
booktitle = "Proceedings of the Second International Conference on Computational Linguistics in Bulgaria (CLIB 2016)",
month = sep,
year = "2016",
address = "Sofia, Bulgaria",
publisher = "Department of Computational Linguistics, Institute for Bulgarian Language, Bulgarian Academy of Sciences",
url = "https://preview.aclanthology.org/jlcl-multiple-ingestion/2016.clib-1.11/",
pages = "86--94",
abstract = "In this study we outline a potential problem in the normalisation stage of processing texts that are based on a modified version of the Arabic alphabet. The main source of resources available for processing resource-scarce languages is raw text. We have identified an interesting challenge that must be addressed when normalising certain natural language texts. Many less-resourced languages, such as Kurdish, Farsi, Urdu, Pashtu, etc., use a modified version of the Arabic writing system. Many characters in harvested data from the Internet may have exactly the same form but encoded with different Unicode values (ambiguous characters). It is important to identify ambiguous characters during the normalisation stage of most text processing tasks. We will demonstrate cases related to ambiguous Kurdish and Farsi characters and propose a semi-automatic approach to identifying and unifying ambiguously encoded characters."
}
Markdown (Informal)
[A Simple Approach to Unifying Ambiguously Encoded Kurdish Characters](https://preview.aclanthology.org/jlcl-multiple-ingestion/2016.clib-1.11/) (Jaf, CLIB 2016)
ACL