fix(serve): match a punctuated label against norm_label symmetrically in _find_node (#1704)
`_find_node` built its search term with `_search_tokens` (\w+ tokenization), so
"blockStream.ts" became "blockstream ts" (space where the '.' was) while the
node's stored `norm_label` keeps punctuation ("blockstream.ts"). The verbatim
case is already rescued by the `term == label_tokens` tier (the node label
tokenizes the same way), but that is a coincidence: if `label` and `norm_label`
diverge, an exactly-typed punctuated label fails to resolve through `explain`
even though `path`/`query` find it.
Add a punctuation-preserving `norm_query` (`_strip_diacritics(label).lower()`)
matched against `norm_label`/`bare_label` across the exact/prefix/substring tiers
(and fed to the trigram prefilter so candidates are not missed). Purely additive,
symmetric with how norm_label is stored. Regression tests cover the verbatim
file-label case and the label/norm_label divergence case that only norm_query
resolves.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
aba1232b66
commit
d1d1f412b2
+15
-3
@@ -664,13 +664,20 @@ def _find_node(G: nx.Graph, label: str) -> list[str]:
|
||||
term = " ".join(_search_tokens(label))
|
||||
if not term:
|
||||
return []
|
||||
# Punctuation-preserving normalized query. `term` tokenizes on \w+ (so
|
||||
# "blockStream.ts" -> "blockstream ts", space where the '.' was), but a node's
|
||||
# stored `norm_label` keeps punctuation ("blockstream.ts"). Matching only via
|
||||
# `term`/`label_tokens` works when the node label tokenizes the same way, but is
|
||||
# fragile if `label` and `norm_label` diverge. `norm_query` matches `norm_label`
|
||||
# symmetrically so an exactly-typed punctuated label always resolves (#1704).
|
||||
norm_query = _strip_diacritics(str(label)).lower().strip()
|
||||
source_exact: list[str] = []
|
||||
exact: list[str] = []
|
||||
prefix: list[str] = []
|
||||
substring: list[str] = []
|
||||
# Trigram prefilter (graph-iteration order preserved so exact/prefix/substring
|
||||
# ordering — and thus matches[0] — is byte-identical to the full scan).
|
||||
candidate_ids = _trigram_candidates(G, [term])
|
||||
candidate_ids = _trigram_candidates(G, [term, norm_query])
|
||||
node_iter = (
|
||||
G.nodes(data=True) if candidate_ids is None
|
||||
else ((nid, G.nodes[nid]) for nid in candidate_ids)
|
||||
@@ -683,16 +690,21 @@ def _find_node(G: nx.Graph, label: str) -> list[str]:
|
||||
nid_lower = nid.lower()
|
||||
if term == source_tokens:
|
||||
source_exact.append(nid)
|
||||
elif term == norm_label or term == bare_label or term == label_tokens or term == nid_lower:
|
||||
elif (
|
||||
term == norm_label or term == bare_label or term == label_tokens or term == nid_lower
|
||||
or norm_query == norm_label or norm_query == bare_label
|
||||
):
|
||||
exact.append(nid)
|
||||
elif (
|
||||
norm_label.startswith(term)
|
||||
or bare_label.startswith(term)
|
||||
or label_tokens.startswith(term)
|
||||
or nid_lower.startswith(term)
|
||||
or norm_label.startswith(norm_query)
|
||||
or bare_label.startswith(norm_query)
|
||||
):
|
||||
prefix.append(nid)
|
||||
elif term in norm_label or term in label_tokens:
|
||||
elif term in norm_label or term in label_tokens or norm_query in norm_label:
|
||||
substring.append(nid)
|
||||
|
||||
if source_exact:
|
||||
|
||||
@@ -135,6 +135,30 @@ def test_find_node_matches_full_punctuated_unicode_label():
|
||||
assert _find_node(G, "Skill /auditar — Auditoría inquisitiva de enlaces") == ["n1"]
|
||||
|
||||
|
||||
def test_find_node_matches_punctuated_file_label_exactly():
|
||||
# #1704: an exactly-typed punctuated file label must resolve through explain,
|
||||
# just like it does through path/query.
|
||||
G = nx.Graph()
|
||||
G.add_node("f1", label="blockStream.ts", norm_label="blockstream.ts",
|
||||
source_file="lib/blockStream.ts", source_location="L1")
|
||||
G.add_node("f2", label="blockStream.test.ts", norm_label="blockstream.test.ts",
|
||||
source_file="lib/blockStream.test.ts", source_location="L1")
|
||||
assert _find_node(G, "blockStream.ts")[0] == "f1"
|
||||
assert _find_node(G, "blockStream.test.ts")[0] == "f2"
|
||||
|
||||
|
||||
def test_find_node_resolves_when_label_and_norm_label_diverge():
|
||||
# #1704 hardening: the tokenized-label tier only rescues the match by
|
||||
# coincidence (label tokenizes the same as the query). When `label` and
|
||||
# `norm_label` diverge, only the symmetric `norm_query == norm_label` match
|
||||
# resolves it. Here label tokenizes to "blockstream" but norm_label is
|
||||
# "blockstream.ts" — this fails without the norm_query path.
|
||||
G = nx.Graph()
|
||||
G.add_node("n1", label="BlockStream", norm_label="blockstream.ts",
|
||||
source_file="lib/x.ts", source_location="L1")
|
||||
assert _find_node(G, "blockStream.ts") == ["n1"]
|
||||
|
||||
|
||||
# --- trigram candidate prefilter (the trigram index that shrinks the O(N) scan) ---
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user