@inproceedings{76aed9e2f7104b449fb8fe9244b96b72,
title = "PengYuan@PKU: Extracting infrequent sense instance with the same N-gram pattern for the SemEval-2010 task 15",
abstract = "This paper describes our infrequent sense identification system participating in the SemEval-2010 task 15 on Infrequent Sense Identification for Mandarin Text to Speech Systems. The core system is a supervised system based on the ensembles of Na{\"i}ve Bayesian classifiers. In order to solve the problem of unbalanced sense distribution, we intentionally extract only instances of infrequent sense with the same N-gram pattern as the complement training data from an untagged Chinese corpus - People's Daily of the year 2001. At the same time, we adjusted the prior probability to adapt to the distribution of the test data and tuned the smoothness coefficient to take the data sparseness into account. Official result shows that, our system ranked the first with the best Macro Accuracy 0.952. We briefly describe this system, its configuration options and the features used for this task and present some discussion of the results.",
author = "Liu, \{Peng Yuan\} and Shui Liu and Yu, \{Shi Wen\} and Zhao, \{Tie Jun\}",
note = "Publisher Copyright: {\textcopyright} 2010 Association for Computational Linguistics.; 5th International Workshop on Semantic Evaluation, SemEval 2010 ; Conference date: 15-07-2010 Through 16-07-2010",
year = "2010",
language = "英语",
series = "ACL 2010 - SemEval 2010 - 5th International Workshop on Semantic Evaluation, Proceedings",
publisher = "Association for Computational Linguistics (ACL)",
pages = "371--374",
booktitle = "ACL 2010 - SemEval 2010 - 5th International Workshop on Semantic Evaluation, Proceedings",
address = "澳大利亚",
}