@InProceedings{fernando-EtAl:2016:WSSANLP2016,
  author    = {Fernando, Sandareka  and  Ranathunga, Surangika  and  Jayasena, Sanath  and  Dias, Gihan},
  title     = {Comprehensive Part-Of-Speech Tag Set and SVM based POS Tagger for Sinhala},
  booktitle = {Proceedings of the 6th Workshop on South and Southeast Asian Natural Language Processing (WSSANLP2016)},
  month     = {December},
  year      = {2016},
  address   = {Osaka, Japan},
  publisher = {The COLING 2016 Organizing Committee},
  pages     = {173--182},
  abstract  = {This paper presents a new comprehensive multi-level Part-Of-Speech tag set and
	a Support Vector Machine based Part-Of-Speech tagger for the Sinhala language.
	The currently available tag set for Sinhala has two limitations: the
	unavailability of tags to represent some word classes and the lack of tags to
	capture inflection based grammatical variations of words. The new tag set,
	presented in this paper overcomes both of these limitations. The accuracy of
	available Sinhala Part-Of-Speech taggers, which are based on Hidden Markov
	Models, still falls far behind state of the art. Our Support Vector Machine
	based tagger achieved an overall accuracy of 84.68% with 59.86% accuracy for
	unknown words and 87.12% for known words, when the test set contains 10% of
	unknown words.},
  url       = {http://aclweb.org/anthology/W16-3718}
}

