@inbook{a340b876882c427d9c6f87a600d48d1e,
title = "Data collection from the web for informetric purposes",
abstract = "This chapter reviews the development of data collection procedures on the web with an emphasis on current practices, data cleansing and matching, data quality and transparency. There are several issues to be considered when collecting data from the web. Transparency is essential to know what is included in the data source, how recent and comprehensive the data are, what timeframe is covered etc. Data quality relates to reliability and accuracy. Mistakes are inevitable, data providers, aggregators, and researchers all make mistakes, but these mistakes should be reduced to a minimum so that meaningful conclusions may be reached from the data analysis. Extensive data cleansing before starting the analysis is needed to try to correct mistakes in the data. When several data sources are used, data from different sources should be matched, and duplicates should be removed.",
keywords = "altmetrics, data analysis, data cleansing, link analysis, search engines, webometrics, world wide web",
author = "Judit Bar-Ilan",
note = "Publisher Copyright: {\textcopyright} Springer Nature Switzerland AG 2019.",
year = "2019",
doi = "10.1007/978-3-030-02511-3_30",
language = "English",
series = "Springer Handbooks",
publisher = "Springer",
pages = "781--800",
booktitle = "Springer Handbooks",
address = "Germany",
}