@inproceedings{b6bc1dabcb164bf2b864fa8e4041ed34,
title = "Semantic-aware blocking for entity resolution",
abstract = "In this work we propose a semantic-aware blocking framework for entity resolution (ER). The proposed framework is built using locality-sensitive hashing (LSH) techniques to efficiently unify both textual and semantic features into an ER blocking process. In order to understand how similarity metrics may affect the effectiveness of ER blocking we study the robustness of similarity metrics and their properties in terms of LSH families. We further discuss how the semantic similarity of records can be captured, measured, and integrated with LSH techniques over multiple similarity spaces. We have evaluated our proposed framework over two real-world data sets, and compared it with the state-of-the-art blocking techniques. The experimental study shows that using a combination of semantic features and textual features can considerably improve the quality of blocking. Due to the probabilistic nature of LSH, this semantic-aware blocking framework also enables us to build fast and reliable blocking for performing entity resolution tasks in a large-scale data environment.",
author = "Qing Wang and Mingyuan Cui and Huizhi Liang",
note = "Publisher Copyright: {\textcopyright} 2016 IEEE.; 32nd IEEE International Conference on Data Engineering, ICDE 2016 ; Conference date: 16-05-2016 Through 20-05-2016",
year = "2016",
month = jun,
day = "22",
doi = "10.1109/ICDE.2016.7498378",
language = "English",
series = "2016 IEEE 32nd International Conference on Data Engineering, ICDE 2016",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "1468--1469",
booktitle = "2016 IEEE 32nd International Conference on Data Engineering, ICDE 2016",
address = "United States",
}