diff --git a/graphify/analyze.py b/graphify/analyze.py index 5f28179..8aaf6c1 100644 --- a/graphify/analyze.py +++ b/graphify/analyze.py @@ -686,8 +686,11 @@ def find_import_cycles( return [] # Step 2: Find simple cycles, bounded by length. + # Pass length_bound so networkx prunes during enumeration rather than + # enumerating all elementary cycles and post-filtering — avoids exponential + # blowup on dense graphs with many long cycles (#1196). cycles: list[list[str]] = [] - for cycle in nx.simple_cycles(file_graph): + for cycle in nx.simple_cycles(file_graph, length_bound=max_cycle_length): if len(cycle) <= max_cycle_length: cycles.append(cycle) if len(cycles) >= top_n * 10: diff --git a/graphify/dedup.py b/graphify/dedup.py index 913975a..e37b4c7 100644 --- a/graphify/dedup.py +++ b/graphify/dedup.py @@ -245,6 +245,13 @@ def deduplicate_entities( continue if _short_label_blocked(norm_label, neighbor_norm, score): continue + # Prefix-extension pairs (getActiveSession / getActiveSessions, + # parseConfig / parseConfigFile) are almost never duplicates — + # one is a strict suffix-extension of the other. Block the merge + # regardless of JW score (#1201). + _lo, _hi = sorted((norm_label, neighbor_norm), key=len) + if _hi.startswith(_lo) and _hi != _lo: + continue c1 = communities.get(node_id) c2 = communities.get(neighbor_id) @@ -372,6 +379,9 @@ def _llm_tiebreak( continue if _short_label_blocked(norm_i, norm_j, score): continue + _lo, _hi = sorted((norm_i, norm_j), key=len) + if _hi.startswith(_lo) and _hi != _lo: + continue c1 = communities.get(node["id"]) c2 = communities.get(neighbor["id"]) if (c1 is not None and c2 is not None and c1 == c2 diff --git a/graphify/llm.py b/graphify/llm.py index 93725f2..513c1e5 100644 --- a/graphify/llm.py +++ b/graphify/llm.py @@ -1890,7 +1890,10 @@ def label_communities( "Respond ONLY with a JSON object mapping the community id (as a string) to " "its name - no prose, no markdown fences.\n\n" + "\n".join(batch_lines) ) - max_tokens = min(40 + 16 * len(batch_cids), 4096) + # 24 tok/community covers 2-5 word JSON entries including id, quotes, + # and punctuation. Cap at 8192 for 16k-context models. Wrapped in + # _resolve_max_tokens so GRAPHIFY_MAX_OUTPUT_TOKENS applies here too (#1200). + max_tokens = _resolve_max_tokens(min(64 + 24 * len(batch_cids), 8192)) try: text = _call_llm(prompt, backend=backend, max_tokens=max_tokens) parsed = _parse_label_response(text, batch_cids) diff --git a/tests/test_dedup.py b/tests/test_dedup.py index 293d2a8..419f009 100644 --- a/tests/test_dedup.py +++ b/tests/test_dedup.py @@ -177,3 +177,64 @@ def test_variant_pair_helper(): assert _is_variant_pair("cortex a55", "cortex a55x") assert not _is_variant_pair("graphextractor", "graphextracter") assert not _is_variant_pair("foo", "foo") + + +def test_prefix_extension_symbols_not_merged(): + """Distinct symbols whose name is a strict prefix-extension of another must not + be merged (#1201). getActiveSession / getActiveSessions score ~98.82 JW but are + different functions; parseConfig / parseConfigFile likewise.""" + import networkx as nx + from graphify.dedup import deduplicate_entities + + pairs = [ + ("getActiveSession", "getActiveSessions"), + ("parseConfig", "parseConfigFile"), + ("load", "loadAll"), + ("handleRequest", "handleRequestTimeout"), + ] + for a, b in pairs: + nodes = [ + {"id": f"{a}_id", "label": a, "type": "CODE", "src_file": "api.py"}, + {"id": f"{b}_id", "label": b, "type": "CODE", "src_file": "api.py"}, + ] + edges = [{"src": f"{a}_id", "tgt": f"{b}_id", "relation": "calls", + "c": 1.0, "weight": 1.0}] + out_nodes, _ = deduplicate_entities( + nodes, edges, communities={f"{a}_id": 0, f"{b}_id": 0} + ) + labels = {n["label"] for n in out_nodes} + assert a in labels and b in labels, ( + f"#1201 regression: '{a}' and '{b}' were merged — they are distinct symbols" + ) + + +def test_prefix_guard_does_not_block_same_length_typos(): + """The prefix-extension guard must not fire for same-length pairs — only strict + prefix-extensions (one is a substring of the other) should be blocked (#1201). + graphextractor / graphextractar have the same length, so neither starts-with the + other, and the guard must not fire.""" + from graphify.dedup import _norm + a = _norm("GraphExtractor") # "graphextractor" — 14 chars + b = _norm("GraphExtractar") # "graphextractar" — 14 chars + lo, hi = sorted((a, b), key=len) + # Same-length pair: startswith only holds when strings are identical + assert not (hi.startswith(lo) and hi != lo), ( + f"Prefix guard fires on same-length pair ({a!r}, {b!r}) — should not" + ) + + +def test_prefix_guard_fires_for_extension_pairs(): + """The prefix-extension guard must fire for pairs where one is a strict prefix + of the other, preventing false merges (#1201).""" + from graphify.dedup import _norm + pairs = [ + ("getActiveSession", "getActiveSessions"), + ("parseConfig", "parseConfigFile"), + ("load", "loadAll"), + ] + for a_raw, b_raw in pairs: + a, b = _norm(a_raw), _norm(b_raw) + lo, hi = sorted((a, b), key=len) + assert hi.startswith(lo) and hi != lo, ( + f"Prefix guard should fire for ({a!r}, {b!r}) but did not" + )