@inbook{8a2d513fd32a46e48e6c2496fdd0ea6b,
title = "Developing a robust part-of-speech tagger for biomedical text",
abstract = "This paper presents a part-of-speech tagger which is specifically tuned for biomedical text. We have built the tagger with maximum entropy modeling and a state-of-the-art tagging algorithm. The tagger was trained on a corpus containing newspaper articles said biomedical documents so that it would work well on various types of biomedical text. Experimental results on the Wall Street Journal corpus, the GENIA corpus, and the PennBioIE corpus revealed that adding training data from a different domain does not hurt the performance of a tagger, and our tagger exhibits very good precision (97% to 98%) on all these corpora. We also evaluated the robustness of the tagger using recent MEDLINE articles. {\textcopyright} Springer-Verlag Berlin Heidelberg 2005.",
author = "Yoshimasa Tsuruoka and Yuka Tateishi and Kim, {Jin Dong} and Tomoko Ohta and John McNaught and Sophia Ananiadou and Jun'ichi Tsujii",
year = "2005",
doi = "10.1007/11573036_36",
language = "English",
isbn = "3540296735",
volume = "3746",
series = "Advances in Informatics - 10th Panhellenic Conference on Informatics",
pages = "382--392",
booktitle = "Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)|Lect. Notes Comput. Sci.",
note = "10th Panhellenic Conference on Informatics, PCI 2005 ; Conference date: 01-07-2005",
}