From cce26730212baab6d92ce5f390ef5aae56268f31 Mon Sep 17 00:00:00 2001 From: Safi Date: Thu, 11 Jun 2026 12:03:59 +0100 Subject: [PATCH] fix: LLM calls-edge direction reversal and ghost-node merge MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit llm.py: add explicit edge direction rule to extraction system prompt — source = actor (caller/importer/subclass), target = acted-upon (callee/ imported/base). LLM was systematically emitting callee->caller for calls edges because the schema never stated direction semantics. build.py: extend ghost-node merge to catch LLM nodes that populate source_location (bypassing the old None check). Now uses _origin=="ast" as the canonical signal — AST nodes always win; any non-AST node sharing (basename, label) with an AST node is collapsed into the AST canonical. Fixes LLM bare-stem IDs (bpe_get_pairs) surviving alongside AST parent-qualified IDs (mingpt_bpe_get_pairs) and carrying reversed edges. Co-Authored-By: Claude Sonnet 4.6 --- graphify/build.py | 32 +++++++++++++++++++++----------- graphify/llm.py | 5 +++++ 2 files changed, 26 insertions(+), 11 deletions(-) diff --git a/graphify/build.py b/graphify/build.py index 5893ef2..d8a69ac 100644 --- a/graphify/build.py +++ b/graphify/build.py @@ -157,14 +157,17 @@ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Pat G.add_node(node["id"], **{k: v for k, v in node.items() if k != "id"}) node_set = set(G.nodes()) - # #1145: merge semantic ghost-duplicate nodes into AST nodes. - # When AST and semantic extractors emit different IDs for the same symbol - # (one has source_location=L, the other has source_location=None), find - # pairs that share (source_file basename, label) and collapse the semantic - # copy into the AST copy so edges re-point to a single node. - # Two passes: first collect all AST (located) nodes, then find ghosts. - _loc_nodes: dict[tuple[str, str], str] = {} # (basename, label) -> AST node id - _noloc_nodes: dict[tuple[str, str], str] = {} # (basename, label) -> semantic node id + # #1145 (extended): merge LLM ghost-duplicate nodes into AST canonical nodes. + # Original bug: AST uses parent-qualified IDs (mingpt_bpe_get_pairs) while LLM + # uses bare-stem IDs (bpe_get_pairs) — different IDs, same symbol. + # Original fix only caught LLM nodes with source_location=None; LLM now + # populates source_location, so those ghosts survived. Extended fix: use + # _origin=="ast" as the canonical signal. AST nodes always win; any non-AST + # node sharing (basename, label) with an AST node is a ghost. + _loc_nodes: dict[tuple[str, str], str] = {} # (basename, label) -> canonical node id + _noloc_nodes: dict[tuple[str, str], str] = {} # (basename, label) -> ghost node id + + # Pass 1: collect canonical nodes — AST-origin nodes take precedence over LLM nodes. for nid in node_set: attrs = G.nodes[nid] label = str(attrs.get("label", "")).strip() @@ -172,14 +175,21 @@ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Pat basename = Path(sf).name if sf else "" if not label or not basename: continue - if attrs.get("source_location"): - _loc_nodes[(basename, label)] = nid + if attrs.get("source_location") or attrs.get("_origin") == "ast": + key = (basename, label) + # AST-origin nodes always overwrite; non-AST only written if key unseen. + if attrs.get("_origin") == "ast" or key not in _loc_nodes: + _loc_nodes[key] = nid + + # Pass 2: find ghosts — non-AST nodes that have an AST canonical twin. for nid in node_set: attrs = G.nodes[nid] + if attrs.get("_origin") == "ast": + continue # AST nodes are never ghosts label = str(attrs.get("label", "")).strip() sf = str(attrs.get("source_file", "")) basename = Path(sf).name if sf else "" - if not label or not basename or attrs.get("source_location"): + if not label or not basename: continue key = (basename, label) if key in _loc_nodes and _loc_nodes[key] != nid: diff --git a/graphify/llm.py b/graphify/llm.py index c6ceabd..d29857a 100644 --- a/graphify/llm.py +++ b/graphify/llm.py @@ -354,6 +354,11 @@ by these rules. Node ID format: lowercase, only [a-z0-9_], no dots or slashes. Format: {stem}_{entity} where stem = filename without extension, entity = symbol name (both normalised). +Edge direction rule — source is always the ACTOR, target is the ACTED-UPON: +- calls: source = the function/method that CONTAINS the call site; target = the function/method BEING CALLED. Never reverse this. +- imports/references: source = the file/entity that imports or references; target = the thing imported or referenced. +- implements/inherits: source = the subclass/implementor; target = the base class/interface. + Output exactly this schema: {"nodes":[{"id":"stem_entity","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"relative/path","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"relative/path","source_location":null,"weight":1.0}],"hyperedges":[],"input_tokens":0,"output_tokens":0} """