@inproceedings{6b4d1d617fe544208f62730bf452f3ae,
title = "A two-step classification approach to unsupervised record linkage",
abstract = "Linking or matching databases is becoming increasingly important in many data mining projects, as linked data can contain information that is not available otherwise, or that would be too expensive to collect manually. A main challenge when linking large databases is the classification of the compared record pairs into matches and non-matches. In traditional record linkage, classification thresholds have to be set either manually or using an EM-based approach. More recently developed classification methods are mainly based on supervised machine learning techniques and thus require training data, which is often not available in real world situations or has to be prepared manually. In this paper, a novel two-step approach to record pair classification is presented. In a first step, example training data of high quality is generated automatically, and then used in a second step to train a supervised classifier. Initial experimental results on both real and synthetic data show that this approach can outperform traditional unsupervised clustering, and even achieve linkage quality almost as good as fully supervised techniques.",
keywords = "Clustering, Data linkage, Data matching, Deduplication, Entity resolution, Quality measures, Support vector machines",
author = "Peter Christen",
year = "2007",
language = "English",
isbn = "9781920682514",
series = "Conferences in Research and Practice in Information Technology Series",
pages = "111--119",
booktitle = "Data Mining and Analytics 2007 - 6th Australasian Data Mining Conference, AusDM 2007, Proceedings",
note = "6th Australasian Data Mining Conference, AusDM 2007 ; Conference date: 03-12-2007 Through 04-12-2007",
}