fix: three correctness bugs — cycle hang, label token budget, fuzzy dedup prefix merge
- analyze.py: pass length_bound=max_cycle_length to nx.simple_cycles() so networkx prunes during enumeration instead of post-filtering; drops report generation from never-returns to ~0.1s on dense graphs (#1196) - llm.py: replace hardcoded min(40+16*n,4096) label_communities token budget with _resolve_max_tokens(min(64+24*n,8192)) — 24 tok/community covers 5-word JSON entries; 8192 cap fits 16k-context models; env var now honoured (#1200) - dedup.py: add prefix-extension guard in Pass 2 and _llm_tiebreak — skip merge when one normalised label is a strict prefix of the other (getActiveSession / getActiveSessions, parseConfig / parseConfigFile). Option (a) rejected: dropping the >=12 early-out from _short_label_blocked breaks test_typo_merged (#1201) - tests/test_dedup.py: two new regression tests verifying prefix guard fires for extension pairs and does not fire for same-length typo pairs Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 4.6
parent
3602c8031e
commit
e477825a97
+4
-1
@@ -686,8 +686,11 @@ def find_import_cycles(
|
||||
return []
|
||||
|
||||
# Step 2: Find simple cycles, bounded by length.
|
||||
# Pass length_bound so networkx prunes during enumeration rather than
|
||||
# enumerating all elementary cycles and post-filtering — avoids exponential
|
||||
# blowup on dense graphs with many long cycles (#1196).
|
||||
cycles: list[list[str]] = []
|
||||
for cycle in nx.simple_cycles(file_graph):
|
||||
for cycle in nx.simple_cycles(file_graph, length_bound=max_cycle_length):
|
||||
if len(cycle) <= max_cycle_length:
|
||||
cycles.append(cycle)
|
||||
if len(cycles) >= top_n * 10:
|
||||
|
||||
@@ -245,6 +245,13 @@ def deduplicate_entities(
|
||||
continue
|
||||
if _short_label_blocked(norm_label, neighbor_norm, score):
|
||||
continue
|
||||
# Prefix-extension pairs (getActiveSession / getActiveSessions,
|
||||
# parseConfig / parseConfigFile) are almost never duplicates —
|
||||
# one is a strict suffix-extension of the other. Block the merge
|
||||
# regardless of JW score (#1201).
|
||||
_lo, _hi = sorted((norm_label, neighbor_norm), key=len)
|
||||
if _hi.startswith(_lo) and _hi != _lo:
|
||||
continue
|
||||
|
||||
c1 = communities.get(node_id)
|
||||
c2 = communities.get(neighbor_id)
|
||||
@@ -372,6 +379,9 @@ def _llm_tiebreak(
|
||||
continue
|
||||
if _short_label_blocked(norm_i, norm_j, score):
|
||||
continue
|
||||
_lo, _hi = sorted((norm_i, norm_j), key=len)
|
||||
if _hi.startswith(_lo) and _hi != _lo:
|
||||
continue
|
||||
c1 = communities.get(node["id"])
|
||||
c2 = communities.get(neighbor["id"])
|
||||
if (c1 is not None and c2 is not None and c1 == c2
|
||||
|
||||
+4
-1
@@ -1890,7 +1890,10 @@ def label_communities(
|
||||
"Respond ONLY with a JSON object mapping the community id (as a string) to "
|
||||
"its name - no prose, no markdown fences.\n\n" + "\n".join(batch_lines)
|
||||
)
|
||||
max_tokens = min(40 + 16 * len(batch_cids), 4096)
|
||||
# 24 tok/community covers 2-5 word JSON entries including id, quotes,
|
||||
# and punctuation. Cap at 8192 for 16k-context models. Wrapped in
|
||||
# _resolve_max_tokens so GRAPHIFY_MAX_OUTPUT_TOKENS applies here too (#1200).
|
||||
max_tokens = _resolve_max_tokens(min(64 + 24 * len(batch_cids), 8192))
|
||||
try:
|
||||
text = _call_llm(prompt, backend=backend, max_tokens=max_tokens)
|
||||
parsed = _parse_label_response(text, batch_cids)
|
||||
|
||||
@@ -177,3 +177,64 @@ def test_variant_pair_helper():
|
||||
assert _is_variant_pair("cortex a55", "cortex a55x")
|
||||
assert not _is_variant_pair("graphextractor", "graphextracter")
|
||||
assert not _is_variant_pair("foo", "foo")
|
||||
|
||||
|
||||
def test_prefix_extension_symbols_not_merged():
|
||||
"""Distinct symbols whose name is a strict prefix-extension of another must not
|
||||
be merged (#1201). getActiveSession / getActiveSessions score ~98.82 JW but are
|
||||
different functions; parseConfig / parseConfigFile likewise."""
|
||||
import networkx as nx
|
||||
from graphify.dedup import deduplicate_entities
|
||||
|
||||
pairs = [
|
||||
("getActiveSession", "getActiveSessions"),
|
||||
("parseConfig", "parseConfigFile"),
|
||||
("load", "loadAll"),
|
||||
("handleRequest", "handleRequestTimeout"),
|
||||
]
|
||||
for a, b in pairs:
|
||||
nodes = [
|
||||
{"id": f"{a}_id", "label": a, "type": "CODE", "src_file": "api.py"},
|
||||
{"id": f"{b}_id", "label": b, "type": "CODE", "src_file": "api.py"},
|
||||
]
|
||||
edges = [{"src": f"{a}_id", "tgt": f"{b}_id", "relation": "calls",
|
||||
"c": 1.0, "weight": 1.0}]
|
||||
out_nodes, _ = deduplicate_entities(
|
||||
nodes, edges, communities={f"{a}_id": 0, f"{b}_id": 0}
|
||||
)
|
||||
labels = {n["label"] for n in out_nodes}
|
||||
assert a in labels and b in labels, (
|
||||
f"#1201 regression: '{a}' and '{b}' were merged — they are distinct symbols"
|
||||
)
|
||||
|
||||
|
||||
def test_prefix_guard_does_not_block_same_length_typos():
|
||||
"""The prefix-extension guard must not fire for same-length pairs — only strict
|
||||
prefix-extensions (one is a substring of the other) should be blocked (#1201).
|
||||
graphextractor / graphextractar have the same length, so neither starts-with the
|
||||
other, and the guard must not fire."""
|
||||
from graphify.dedup import _norm
|
||||
a = _norm("GraphExtractor") # "graphextractor" — 14 chars
|
||||
b = _norm("GraphExtractar") # "graphextractar" — 14 chars
|
||||
lo, hi = sorted((a, b), key=len)
|
||||
# Same-length pair: startswith only holds when strings are identical
|
||||
assert not (hi.startswith(lo) and hi != lo), (
|
||||
f"Prefix guard fires on same-length pair ({a!r}, {b!r}) — should not"
|
||||
)
|
||||
|
||||
|
||||
def test_prefix_guard_fires_for_extension_pairs():
|
||||
"""The prefix-extension guard must fire for pairs where one is a strict prefix
|
||||
of the other, preventing false merges (#1201)."""
|
||||
from graphify.dedup import _norm
|
||||
pairs = [
|
||||
("getActiveSession", "getActiveSessions"),
|
||||
("parseConfig", "parseConfigFile"),
|
||||
("load", "loadAll"),
|
||||
]
|
||||
for a_raw, b_raw in pairs:
|
||||
a, b = _norm(a_raw), _norm(b_raw)
|
||||
lo, hi = sorted((a, b), key=len)
|
||||
assert hi.startswith(lo) and hi != lo, (
|
||||
f"Prefix guard should fire for ({a!r}, {b!r}) but did not"
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user