fix: three correctness bugs — cycle hang, label token budget, fuzzy dedup prefix merge

- analyze.py: pass length_bound=max_cycle_length to nx.simple_cycles() so
  networkx prunes during enumeration instead of post-filtering; drops report
  generation from never-returns to ~0.1s on dense graphs (#1196)

- llm.py: replace hardcoded min(40+16*n,4096) label_communities token budget
  with _resolve_max_tokens(min(64+24*n,8192)) — 24 tok/community covers 5-word
  JSON entries; 8192 cap fits 16k-context models; env var now honoured (#1200)

- dedup.py: add prefix-extension guard in Pass 2 and _llm_tiebreak — skip merge
  when one normalised label is a strict prefix of the other (getActiveSession /
  getActiveSessions, parseConfig / parseConfigFile). Option (a) rejected: dropping
  the >=12 early-out from _short_label_blocked breaks test_typo_merged (#1201)

- tests/test_dedup.py: two new regression tests verifying prefix guard fires for
  extension pairs and does not fire for same-length typo pairs

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Safi
2026-06-08 23:49:16 +01:00
co-authored by Claude Sonnet 4.6
parent 3602c8031e
commit e477825a97
4 changed files with 79 additions and 2 deletions
+4 -1
View File
@@ -686,8 +686,11 @@ def find_import_cycles(
return []
# Step 2: Find simple cycles, bounded by length.
# Pass length_bound so networkx prunes during enumeration rather than
# enumerating all elementary cycles and post-filtering — avoids exponential
# blowup on dense graphs with many long cycles (#1196).
cycles: list[list[str]] = []
for cycle in nx.simple_cycles(file_graph):
for cycle in nx.simple_cycles(file_graph, length_bound=max_cycle_length):
if len(cycle) <= max_cycle_length:
cycles.append(cycle)
if len(cycles) >= top_n * 10:
+10
View File
@@ -245,6 +245,13 @@ def deduplicate_entities(
continue
if _short_label_blocked(norm_label, neighbor_norm, score):
continue
# Prefix-extension pairs (getActiveSession / getActiveSessions,
# parseConfig / parseConfigFile) are almost never duplicates —
# one is a strict suffix-extension of the other. Block the merge
# regardless of JW score (#1201).
_lo, _hi = sorted((norm_label, neighbor_norm), key=len)
if _hi.startswith(_lo) and _hi != _lo:
continue
c1 = communities.get(node_id)
c2 = communities.get(neighbor_id)
@@ -372,6 +379,9 @@ def _llm_tiebreak(
continue
if _short_label_blocked(norm_i, norm_j, score):
continue
_lo, _hi = sorted((norm_i, norm_j), key=len)
if _hi.startswith(_lo) and _hi != _lo:
continue
c1 = communities.get(node["id"])
c2 = communities.get(neighbor["id"])
if (c1 is not None and c2 is not None and c1 == c2
+4 -1
View File
@@ -1890,7 +1890,10 @@ def label_communities(
"Respond ONLY with a JSON object mapping the community id (as a string) to "
"its name - no prose, no markdown fences.\n\n" + "\n".join(batch_lines)
)
max_tokens = min(40 + 16 * len(batch_cids), 4096)
# 24 tok/community covers 2-5 word JSON entries including id, quotes,
# and punctuation. Cap at 8192 for 16k-context models. Wrapped in
# _resolve_max_tokens so GRAPHIFY_MAX_OUTPUT_TOKENS applies here too (#1200).
max_tokens = _resolve_max_tokens(min(64 + 24 * len(batch_cids), 8192))
try:
text = _call_llm(prompt, backend=backend, max_tokens=max_tokens)
parsed = _parse_label_response(text, batch_cids)
+61
View File
@@ -177,3 +177,64 @@ def test_variant_pair_helper():
assert _is_variant_pair("cortex a55", "cortex a55x")
assert not _is_variant_pair("graphextractor", "graphextracter")
assert not _is_variant_pair("foo", "foo")
def test_prefix_extension_symbols_not_merged():
"""Distinct symbols whose name is a strict prefix-extension of another must not
be merged (#1201). getActiveSession / getActiveSessions score ~98.82 JW but are
different functions; parseConfig / parseConfigFile likewise."""
import networkx as nx
from graphify.dedup import deduplicate_entities
pairs = [
("getActiveSession", "getActiveSessions"),
("parseConfig", "parseConfigFile"),
("load", "loadAll"),
("handleRequest", "handleRequestTimeout"),
]
for a, b in pairs:
nodes = [
{"id": f"{a}_id", "label": a, "type": "CODE", "src_file": "api.py"},
{"id": f"{b}_id", "label": b, "type": "CODE", "src_file": "api.py"},
]
edges = [{"src": f"{a}_id", "tgt": f"{b}_id", "relation": "calls",
"c": 1.0, "weight": 1.0}]
out_nodes, _ = deduplicate_entities(
nodes, edges, communities={f"{a}_id": 0, f"{b}_id": 0}
)
labels = {n["label"] for n in out_nodes}
assert a in labels and b in labels, (
f"#1201 regression: '{a}' and '{b}' were merged — they are distinct symbols"
)
def test_prefix_guard_does_not_block_same_length_typos():
"""The prefix-extension guard must not fire for same-length pairs — only strict
prefix-extensions (one is a substring of the other) should be blocked (#1201).
graphextractor / graphextractar have the same length, so neither starts-with the
other, and the guard must not fire."""
from graphify.dedup import _norm
a = _norm("GraphExtractor") # "graphextractor" — 14 chars
b = _norm("GraphExtractar") # "graphextractar" — 14 chars
lo, hi = sorted((a, b), key=len)
# Same-length pair: startswith only holds when strings are identical
assert not (hi.startswith(lo) and hi != lo), (
f"Prefix guard fires on same-length pair ({a!r}, {b!r}) — should not"
)
def test_prefix_guard_fires_for_extension_pairs():
"""The prefix-extension guard must fire for pairs where one is a strict prefix
of the other, preventing false merges (#1201)."""
from graphify.dedup import _norm
pairs = [
("getActiveSession", "getActiveSessions"),
("parseConfig", "parseConfigFile"),
("load", "loadAll"),
]
for a_raw, b_raw in pairs:
a, b = _norm(a_raw), _norm(b_raw)
lo, hi = sorted((a, b), key=len)
assert hi.startswith(lo) and hi != lo, (
f"Prefix guard should fire for ({a!r}, {b!r}) but did not"
)