diff --git a/graphify/serve.py b/graphify/serve.py index 28cb466..6d8b369 100644 --- a/graphify/serve.py +++ b/graphify/serve.py @@ -664,13 +664,20 @@ def _find_node(G: nx.Graph, label: str) -> list[str]: term = " ".join(_search_tokens(label)) if not term: return [] + # Punctuation-preserving normalized query. `term` tokenizes on \w+ (so + # "blockStream.ts" -> "blockstream ts", space where the '.' was), but a node's + # stored `norm_label` keeps punctuation ("blockstream.ts"). Matching only via + # `term`/`label_tokens` works when the node label tokenizes the same way, but is + # fragile if `label` and `norm_label` diverge. `norm_query` matches `norm_label` + # symmetrically so an exactly-typed punctuated label always resolves (#1704). + norm_query = _strip_diacritics(str(label)).lower().strip() source_exact: list[str] = [] exact: list[str] = [] prefix: list[str] = [] substring: list[str] = [] # Trigram prefilter (graph-iteration order preserved so exact/prefix/substring # ordering — and thus matches[0] — is byte-identical to the full scan). - candidate_ids = _trigram_candidates(G, [term]) + candidate_ids = _trigram_candidates(G, [term, norm_query]) node_iter = ( G.nodes(data=True) if candidate_ids is None else ((nid, G.nodes[nid]) for nid in candidate_ids) @@ -683,16 +690,21 @@ def _find_node(G: nx.Graph, label: str) -> list[str]: nid_lower = nid.lower() if term == source_tokens: source_exact.append(nid) - elif term == norm_label or term == bare_label or term == label_tokens or term == nid_lower: + elif ( + term == norm_label or term == bare_label or term == label_tokens or term == nid_lower + or norm_query == norm_label or norm_query == bare_label + ): exact.append(nid) elif ( norm_label.startswith(term) or bare_label.startswith(term) or label_tokens.startswith(term) or nid_lower.startswith(term) + or norm_label.startswith(norm_query) + or bare_label.startswith(norm_query) ): prefix.append(nid) - elif term in norm_label or term in label_tokens: + elif term in norm_label or term in label_tokens or norm_query in norm_label: substring.append(nid) if source_exact: diff --git a/tests/test_serve.py b/tests/test_serve.py index 2647aa1..9bf9aa6 100644 --- a/tests/test_serve.py +++ b/tests/test_serve.py @@ -135,6 +135,30 @@ def test_find_node_matches_full_punctuated_unicode_label(): assert _find_node(G, "Skill /auditar — Auditoría inquisitiva de enlaces") == ["n1"] +def test_find_node_matches_punctuated_file_label_exactly(): + # #1704: an exactly-typed punctuated file label must resolve through explain, + # just like it does through path/query. + G = nx.Graph() + G.add_node("f1", label="blockStream.ts", norm_label="blockstream.ts", + source_file="lib/blockStream.ts", source_location="L1") + G.add_node("f2", label="blockStream.test.ts", norm_label="blockstream.test.ts", + source_file="lib/blockStream.test.ts", source_location="L1") + assert _find_node(G, "blockStream.ts")[0] == "f1" + assert _find_node(G, "blockStream.test.ts")[0] == "f2" + + +def test_find_node_resolves_when_label_and_norm_label_diverge(): + # #1704 hardening: the tokenized-label tier only rescues the match by + # coincidence (label tokenizes the same as the query). When `label` and + # `norm_label` diverge, only the symmetric `norm_query == norm_label` match + # resolves it. Here label tokenizes to "blockstream" but norm_label is + # "blockstream.ts" — this fails without the norm_query path. + G = nx.Graph() + G.add_node("n1", label="BlockStream", norm_label="blockstream.ts", + source_file="lib/x.ts", source_location="L1") + assert _find_node(G, "blockStream.ts") == ["n1"] + + # --- trigram candidate prefilter (the trigram index that shrinks the O(N) scan) ---