Entity Resolution is a core task for merging data collections. Due to its quadratic complexity, it typically scales to large volumes of data through blocking: similar entities are clustered into blocks and pair-wise comparisons are executed only between co-occurring entities, at the cost of some missed matches. There are numerous blocking methods, and the aim of this work is to offer a comprehensive empirical survey, extending the dimensions of comparison beyond what is commonly available in the literature. We consider 17 state-of-the-art blocking methods and use 6 popular real datasets to examine the robustness of their internal configurations and their relative balance between effectiveness and time efficiency. We also investigate their scalability over a corpus of 7 established synthetic datasets that range from 10,000 to 2 million entities.
%0 Journal Article
%1 papadakis2016comparative
%A Papadakis, George
%A Svirsky, Jonathan
%A Gal, Avigdor
%A Palpanas, Themis
%D 2016
%I VLDB Endowment
%J Proc. VLDB Endow.
%K blocking deduplication entity linkage resolution
%N 9
%P 684–695
%R 10.14778/2947618.2947624
%T Comparative analysis of approximate blocking techniques for entity resolution
%U https://doi.org/10.14778/2947618.2947624
%V 9
%X Entity Resolution is a core task for merging data collections. Due to its quadratic complexity, it typically scales to large volumes of data through blocking: similar entities are clustered into blocks and pair-wise comparisons are executed only between co-occurring entities, at the cost of some missed matches. There are numerous blocking methods, and the aim of this work is to offer a comprehensive empirical survey, extending the dimensions of comparison beyond what is commonly available in the literature. We consider 17 state-of-the-art blocking methods and use 6 popular real datasets to examine the robustness of their internal configurations and their relative balance between effectiveness and time efficiency. We also investigate their scalability over a corpus of 7 established synthetic datasets that range from 10,000 to 2 million entities.
@article{papadakis2016comparative,
abstract = {Entity Resolution is a core task for merging data collections. Due to its quadratic complexity, it typically scales to large volumes of data through blocking: similar entities are clustered into blocks and pair-wise comparisons are executed only between co-occurring entities, at the cost of some missed matches. There are numerous blocking methods, and the aim of this work is to offer a comprehensive empirical survey, extending the dimensions of comparison beyond what is commonly available in the literature. We consider 17 state-of-the-art blocking methods and use 6 popular real datasets to examine the robustness of their internal configurations and their relative balance between effectiveness and time efficiency. We also investigate their scalability over a corpus of 7 established synthetic datasets that range from 10,000 to 2 million entities.},
added-at = {2026-08-05T14:27:20.000+0200},
author = {Papadakis, George and Svirsky, Jonathan and Gal, Avigdor and Palpanas, Themis},
biburl = {https://www.bibsonomy.org/bibtex/2095e55a44544ddf6f08218b768b81eb4/jaeschke},
doi = {10.14778/2947618.2947624},
interhash = {f60cf47ec73ee9a390d7a138b9b18e6d},
intrahash = {095e55a44544ddf6f08218b768b81eb4},
issn = {2150-8097},
issue_date = {May 2016},
journal = {Proc. VLDB Endow.},
keywords = {blocking deduplication entity linkage resolution},
month = may,
number = 9,
numpages = {12},
pages = {684–695},
publisher = {VLDB Endowment},
timestamp = {2026-08-05T14:27:20.000+0200},
title = {Comparative analysis of approximate blocking techniques for entity resolution},
url = {https://doi.org/10.14778/2947618.2947624},
volume = 9,
year = 2016
}