@inproceedings{cfe98c65e42f4487b9feab610a30dd4f,
title = "Towards named entity annotation of latvian national library corpus",
abstract = "The paper describes a work in progress of building a catalogue of named entities-people, places and organizations-based on a recently digitized large (4.5 billion tokens) Latvian corpus. The authors propose an annotation standard for markup of named entities within Latvian corpus, according to which a representative set of documents (150 000 words) are manually annotated. This corpus is used for training and evaluation of an automated named entity recognition system based on Stanford CRF classifier, achieving an F-score of up to 81\%. The named entities indexed within the Latvian National Library corpus and the annnotated documents are publicly available for linguistic and historical research online.",
keywords = "Latvian, NER, corpus indexing, named entity recognition",
author = "Peteris Paikens and Ilze Auzina and Ginta Garkaje and Madara Paegle",
year = "2012",
doi = "10.3233/978-1-61499-133-5-169",
language = "English",
isbn = "9781614991328",
series = "Frontiers in Artificial Intelligence and Applications",
publisher = "IOS Press BV",
pages = "169--175",
booktitle = "Human Language Technologies - The Baltic Perspective. Proceedings of the Fifth International Conference Baltic HLT 2012",
address = "Netherlands",
note = "5th International Conference on Human Language Technologies - The Baltic Perspective, Baltic HLT 2012 ; Conference date: 04-10-2012 Through 05-10-2012",
}