Merge pull request #1289 from sirphilliptubell/perf/fix1

perf: fix O(n^2) -> O(n) LSH neighbor lookup in dedup
This commit is contained in:
Safi
2026-06-12 20:15:20 +01:00
committed by GitHub
+14 -7
View File
@@ -237,19 +237,26 @@ def deduplicate_entities(
if len(candidates) >= 2:
lsh = MinHashLSH(threshold=_LSH_THRESHOLD, num_perm=_NUM_PERM)
minhashes: dict[str, MinHash] = {}
# Pre-build O(1) lookup structures so the query loop below doesn't scan
# the candidates list linearly for every LSH neighbor (was O(n²×B)).
candidates_by_id: dict[str, dict] = {}
norm_cache: dict[str, str] = {}
for node in candidates:
norm_label = _norm(node.get("label", node.get("id", "")))
m = _make_minhash(norm_label)
minhashes[node["id"]] = m
node_id = node["id"]
candidates_by_id[node_id] = node
nl = _norm(node.get("label", node.get("id", "")))
norm_cache[node_id] = nl
m = _make_minhash(nl)
minhashes[node_id] = m
try:
lsh.insert(node["id"], m)
lsh.insert(node_id, m)
except ValueError:
pass # duplicate key in LSH — already inserted
for node in candidates:
node_id = node["id"]
norm_label = _norm(node.get("label", node.get("id", "")))
norm_label = norm_cache[node_id]
neighbors = lsh.query(minhashes[node_id])
for neighbor_id in neighbors:
@@ -258,11 +265,11 @@ def deduplicate_entities(
if uf.find(node_id) == uf.find(neighbor_id):
continue
neighbor = next((n for n in candidates if n["id"] == neighbor_id), None)
neighbor = candidates_by_id.get(neighbor_id)
if neighbor is None:
continue
neighbor_norm = _norm(neighbor.get("label", neighbor.get("id", "")))
neighbor_norm = norm_cache.get(neighbor_id) or _norm(neighbor.get("label", neighbor.get("id", "")))
score = JaroWinkler.normalized_similarity(norm_label, neighbor_norm) * 100
if _is_variant_pair(norm_label, neighbor_norm):