@inproceedings{4d4e315b9d3443518fef8f8cc962c40d,
title = "NokeaRM: employing non-key attributes in record matching",
abstract = "Record Matching (RM) aims at finding out pairs of instances referring to the same entity between relational tables. Existing RM methods mainly work on key attribute values, but neglect the possible effectiveness of non-key attribute values in RM. As a result, when two instances referring to the same entity do not have similar key attribute values, they are unlikely to be linked as an instance pair. On the other hand, the two instances may share some important non-key attribute values which can also help us identify the relationship between them. With this intuition, we propose to employ non-key attributes in RM. Basically, we propose a rule-based algorithm based on a tree-like structure, which can not only deal with noisy and missing values, but also greatly improve the efficiency of the method by finding out matched instances or filtering unmatched instances as early as possible. The experimental results based on several data sets demonstrate that our method outperforms existing RM methods by reaching a higher precision and recall. Besides, the proposed techniques can greatly improve the efficiency of a baseline.",
keywords = "Algorithm, Non-key attribute, Record matching",
author = "Qiang Yang and Zhixu Li and Jun Jiang and Pengpeng Zhao and Guanfeng Liu and An Liu and Jia Zhu",
year = "2015",
doi = "10.1007/978-3-319-21042-1\_36",
language = "English",
isbn = "9783319210414",
series = "Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)",
publisher = "Springer, Springer Nature",
pages = "438--442",
editor = "Dong, \{Xin Luna\} and Xiaohui Yu and Jian Li and Yizhou Sun",
booktitle = "Web-Age Information Management",
address = "United States",
note = "16th International Conference on Web-Age Information Management, WAIM 2015 ; Conference date: 08-06-2015 Through 10-06-2015",
}