@inproceedings{66a15409576345d5b2717fcb53ee419f,
title = "An information extraction system for heterogeneous Web source",
abstract = "Information Extraction is the task of identifying information in texts and converting it into a predefined format. In this paper, we build an information integration system which focuses on the information of computer science teachers in Chinese universities. The target of the system is to automatically extract the useful information from heterogeneous sources and re-organize them into structured format. The system includes 4 main modules: web pages retrieval module, web pages' structure classification module, information extraction module and information updating module. We have successfully applied the system to deal with 107 universities in China which shows the effect of the proposed system.",
keywords = "Information Extraction, Topical crawler, Web mining, Web page structure classification",
author = "Ting Zhou and Sun, \{Cheng Jie\} and Lei Lin and Liu, \{Bing Quan\}",
year = "2010",
doi = "10.1109/ICMLC.2010.5580698",
language = "英语",
isbn = "9781424465262",
series = "2010 International Conference on Machine Learning and Cybernetics, ICMLC 2010",
publisher = "IEEE Computer Society",
pages = "3287--3292",
booktitle = "2010 International Conference on Machine Learning and Cybernetics, ICMLC 2010",
address = "美国",
}