fix(serve): match a punctuated label against norm_label symmetrically in _find_node (#1704)

`_find_node` built its search term with `_search_tokens` (\w+ tokenization), so
"blockStream.ts" became "blockstream ts" (space where the '.' was) while the
node's stored `norm_label` keeps punctuation ("blockstream.ts"). The verbatim
case is already rescued by the `term == label_tokens` tier (the node label
tokenizes the same way), but that is a coincidence: if `label` and `norm_label`
diverge, an exactly-typed punctuated label fails to resolve through `explain`
even though `path`/`query` find it.

Add a punctuation-preserving `norm_query` (`_strip_diacritics(label).lower()`)
matched against `norm_label`/`bare_label` across the exact/prefix/substring tiers
(and fed to the trigram prefilter so candidates are not missed). Purely additive,
symmetric with how norm_label is stored. Regression tests cover the verbatim
file-label case and the label/norm_label divergence case that only norm_query
resolves.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-07-06 23:28:32 +01:00
co-authored by Claude Opus 4.8
parent aba1232b66
commit d1d1f412b2
2 changed files with 39 additions and 3 deletions
+15 -3
View File
@@ -664,13 +664,20 @@ def _find_node(G: nx.Graph, label: str) -> list[str]:
term = " ".join(_search_tokens(label))
if not term:
return []
# Punctuation-preserving normalized query. `term` tokenizes on \w+ (so
# "blockStream.ts" -> "blockstream ts", space where the '.' was), but a node's
# stored `norm_label` keeps punctuation ("blockstream.ts"). Matching only via
# `term`/`label_tokens` works when the node label tokenizes the same way, but is
# fragile if `label` and `norm_label` diverge. `norm_query` matches `norm_label`
# symmetrically so an exactly-typed punctuated label always resolves (#1704).
norm_query = _strip_diacritics(str(label)).lower().strip()
source_exact: list[str] = []
exact: list[str] = []
prefix: list[str] = []
substring: list[str] = []
# Trigram prefilter (graph-iteration order preserved so exact/prefix/substring
# ordering — and thus matches[0] — is byte-identical to the full scan).
candidate_ids = _trigram_candidates(G, [term])
candidate_ids = _trigram_candidates(G, [term, norm_query])
node_iter = (
G.nodes(data=True) if candidate_ids is None
else ((nid, G.nodes[nid]) for nid in candidate_ids)
@@ -683,16 +690,21 @@ def _find_node(G: nx.Graph, label: str) -> list[str]:
nid_lower = nid.lower()
if term == source_tokens:
source_exact.append(nid)
elif term == norm_label or term == bare_label or term == label_tokens or term == nid_lower:
elif (
term == norm_label or term == bare_label or term == label_tokens or term == nid_lower
or norm_query == norm_label or norm_query == bare_label
):
exact.append(nid)
elif (
norm_label.startswith(term)
or bare_label.startswith(term)
or label_tokens.startswith(term)
or nid_lower.startswith(term)
or norm_label.startswith(norm_query)
or bare_label.startswith(norm_query)
):
prefix.append(nid)
elif term in norm_label or term in label_tokens:
elif term in norm_label or term in label_tokens or norm_query in norm_label:
substring.append(nid)
if source_exact:
+24
View File
@@ -135,6 +135,30 @@ def test_find_node_matches_full_punctuated_unicode_label():
assert _find_node(G, "Skill /auditar — Auditoría inquisitiva de enlaces") == ["n1"]
def test_find_node_matches_punctuated_file_label_exactly():
# #1704: an exactly-typed punctuated file label must resolve through explain,
# just like it does through path/query.
G = nx.Graph()
G.add_node("f1", label="blockStream.ts", norm_label="blockstream.ts",
source_file="lib/blockStream.ts", source_location="L1")
G.add_node("f2", label="blockStream.test.ts", norm_label="blockstream.test.ts",
source_file="lib/blockStream.test.ts", source_location="L1")
assert _find_node(G, "blockStream.ts")[0] == "f1"
assert _find_node(G, "blockStream.test.ts")[0] == "f2"
def test_find_node_resolves_when_label_and_norm_label_diverge():
# #1704 hardening: the tokenized-label tier only rescues the match by
# coincidence (label tokenizes the same as the query). When `label` and
# `norm_label` diverge, only the symmetric `norm_query == norm_label` match
# resolves it. Here label tokenizes to "blockstream" but norm_label is
# "blockstream.ts" — this fails without the norm_query path.
G = nx.Graph()
G.add_node("n1", label="BlockStream", norm_label="blockstream.ts",
source_file="lib/x.ts", source_location="L1")
assert _find_node(G, "blockStream.ts") == ["n1"]
# --- trigram candidate prefilter (the trigram index that shrinks the O(N) scan) ---