@inproceedings{9bdb1f771f0b4ee1b724e4f0fc9ded0b,
title = "The research of the maximum length n-grams priority Chinese word segmentation method based on corpus type frequency information",
abstract = "In order to solve the difficulties to extract words in particular domain, we formulate a method of automatic word segmentation in Chinese based on corpus type frequency information. This method can effectively extract n-gram words that are not predefined in a lexicon by setting the maximum length (n) of the n-gram word we want to extract from a sentence and the minimum threshold frequency the n-gram word appears in corpus. When the real frequency the n-gram appears in corpus is above the threshold, the n-gram word will be extracted. If there are two or more n-grams have the same length, the higher frequency one will be chosen, and then the next higher frequency one if any of its characters are not in previous one.",
keywords = "Corpus type frequency information, N-gram word, Word frequency, Word segmentation",
author = "Pengyu Lu and Lijun Jin and Bin Jiang",
year = "2012",
doi = "10.2991/citcs.2012.111",
language = "英语",
isbn = "9789491216381",
series = "Proceedings of the 2012 National Conference on Information Technology and Computer Science, CITCS 2012",
publisher = "Atlantis Press",
pages = "71--74",
booktitle = "Proceedings of the 2012 National Conference on Information Technology and Computer Science, CITCS 2012",
note = "2012 National Conference on Information Technology and Computer Science, CITCS 2012 ; Conference date: 16-11-2012 Through 18-11-2012",
}