@inbook{3bd7777016a1472ca057cdc1d59f09ce,
title = "Advanced techniques in web data pre-processing and cleaning",
abstract = "Central to successful e-business is the construction of web sites that attract users, capture user preferences, and entice them into making a purchase. Web mining is diverse data mining applied to categorize both the content and structure of web sites with the goal of aiding e-business. Web mining requires knowledge of the web site structure (hyperlink graph), the web content (vector model) and user sessions (the sequence of pages visited by each user to a site). Much of the data for web mining can be noisy. The origin of the noise comes from many sources, for example, undocumented changes to the web site structure and content, a different understanding of the text and media semantic, and web logs without individual user identification. There may not be any record of the number of times a specific page has been visited in a session as page is stored on a proxy or web browser cache. Such noise presents a challenge for web mining. This chapter presents issues with and approaches for cleaning web data in preparation for web mining analysis.",
author = "Rom{\'a}n, \{Pablo E.\} and Dell, \{Robert F.\} and Vel{\'a}squez, \{Juan D.\}",
year = "2010",
doi = "10.1007/978-3-642-14461-5\_2",
language = "English",
isbn = "9783642144608",
series = "Studies in Computational Intelligence",
pages = "19--48",
editor = "Juan Velasquez and Lakhmi Jain",
booktitle = "Advanced Techniques in Web Intelligence - 1",
}