diff --git a/graphify/analyze.py b/graphify/analyze.py index a436ca0..43948bf 100644 --- a/graphify/analyze.py +++ b/graphify/analyze.py @@ -380,6 +380,9 @@ def suggest_questions( Based on: AMBIGUOUS edges, bridge nodes, underexplored god nodes, isolated nodes. Each question has a 'type', 'question', and 'why' field. """ + if community_labels: + community_labels = {int(k) if isinstance(k, str) else k: v for k, v in community_labels.items()} + questions = [] node_community = _node_community_map(communities) diff --git a/graphify/build.py b/graphify/build.py index a0dbcae..12accd7 100644 --- a/graphify/build.py +++ b/graphify/build.py @@ -295,11 +295,12 @@ def build_merge( all_chunks = base + list(new_chunks) G = build(all_chunks, directed=directed, dedup=dedup, dedup_llm_backend=dedup_llm_backend) - # Prune nodes from deleted source files + # Prune nodes and edges from deleted source files if prune_sources: + prune_set = set(prune_sources) to_remove = [ n for n, d in G.nodes(data=True) - if d.get("source_file") in prune_sources + if d.get("source_file") in prune_set ] G.remove_nodes_from(to_remove) n_files = len(prune_sources) @@ -309,10 +310,22 @@ def build_merge( f"[graphify] Pruned {n_nodes} node(s) from {n_files} deleted source file(s).", file=sys.stderr, ) - else: + + edges_to_remove = [ + (u, v) for u, v, d in G.edges(data=True) + if d.get("source_file") in prune_set + ] + if edges_to_remove: + G.remove_edges_from(edges_to_remove) + print( + f"[graphify] Pruned {len(edges_to_remove)} edge(s) from deleted source file(s).", + file=sys.stderr, + ) + + if not n_nodes and not edges_to_remove: print( f"[graphify] {n_files} source file(s) deleted since last run — " - f"no matching nodes in graph, already clean.", + f"no matching nodes or edges in graph, already clean.", file=sys.stderr, ) diff --git a/graphify/extract.py b/graphify/extract.py index c1db3e4..bc7ea87 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -2681,20 +2681,51 @@ def extract_sql(path: Path) -> dict: # Foreign key REFERENCES for col in node.children: if col.type == "column_definitions": + has_error = any(cd.type == "ERROR" for cd in col.children) + seen_refs: set[str] = set() for cd in col.children: - if cd.type != "column_definition": - continue - ref_name: str | None = None - found_ref = False - for cc in cd.children: - if cc.type == "keyword_references": - found_ref = True - elif found_ref and cc.type == "object_reference": - ref_name = _read(cc) - break - if ref_name: - ref_nid = _make_id(stem, ref_name) - _add_edge(nid, ref_nid, "references", line) + if cd.type == "column_definition": + # Inline column-level REFERENCES + ref_name: str | None = None + found_ref = False + for cc in cd.children: + if cc.type == "keyword_references": + found_ref = True + elif found_ref and cc.type == "object_reference": + ref_name = _read(cc) + break + if ref_name: + ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name) + _add_edge(nid, ref_nid, "references", line) + seen_refs.add(ref_name.lower()) + elif cd.type == "constraints": + # Table-level FOREIGN KEY ... REFERENCES ... constraints + for constraint in cd.children: + if constraint.type != "constraint": + continue + ref_name = None + found_ref = False + for cc in constraint.children: + if cc.type == "keyword_references": + found_ref = True + elif found_ref and cc.type == "object_reference": + ref_name = _read(cc) + break + if ref_name: + ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name) + _add_edge(nid, ref_nid, "references", line) + seen_refs.add(ref_name.lower()) + if has_error: + # Dialect-specific syntax (e.g. Firebird COMPUTED BY) causes ERROR + # nodes that make the parser drop the trailing constraints block. + # Regex-scan the raw column_definitions text as fallback. + col_text = _read(col) + for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", col_text, re.IGNORECASE): + ref_name = rm.group(1) + if ref_name.lower() not in seen_refs: + ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name) + _add_edge(nid, ref_nid, "references", line) + seen_refs.add(ref_name.lower()) elif t == "create_view": name = _obj_name(node) @@ -2746,6 +2777,64 @@ def extract_sql(path: Path) -> dict: ref_nid = _make_id(stem, ref_name) _add_edge(src_nid, ref_nid, "references", line) + elif t == "create_trigger": + trig_name: str | None = None + tbl_name: str | None = None + after_trigger = False + after_for = False + for c in node.children: + if c.type == "keyword_trigger": + after_trigger = True + elif after_trigger and not trig_name and c.type == "object_reference": + trig_name = _read(c) + elif c.type == "keyword_for": + after_for = True + elif after_for and not tbl_name and c.type == "object_reference": + tbl_name = _read(c) + if trig_name: + trig_nid = _make_id(stem, trig_name) + _add_node(trig_nid, trig_name, line) + if tbl_name: + tbl_nid = table_nids.get(tbl_name.lower()) or _make_id(stem, tbl_name) + _add_edge(trig_nid, tbl_nid, "triggers", line) + + elif t == "fb_proc_or_trigger": + text = _read(node) + m = re.match( + r"CREATE\s+(?:OR\s+(?:REPLACE|ALTER)\s+)?" + r"(PROCEDURE|TRIGGER|FUNCTION)\s+([\w$]+)", + text, re.IGNORECASE, + ) + if m: + obj_type = m.group(1).upper() + obj_name = m.group(2) + obj_nid = _make_id(stem, obj_name) + label = obj_name if obj_type == "TRIGGER" else f"{obj_name}()" + _add_node(obj_nid, label, line) + if obj_type == "TRIGGER": + fm = re.search(r"\bFOR\s+([\w$]+)", text, re.IGNORECASE) + if fm: + tbl = fm.group(1) + tbl_nid = table_nids.get(tbl.lower()) or _make_id(stem, tbl) + _add_edge(obj_nid, tbl_nid, "triggers", line) + _NON_TABLES = { + "select", "where", "set", "dual", "null", "true", "false", + "first", "skip", "rows", "next", "only", "lateral", + } + seen_tbls: set[str] = set() + for rm in re.finditer(r"\b(?:FROM|JOIN|INTO)\s+([\w$]+)", text, re.IGNORECASE): + tbl = rm.group(1) + if tbl.lower() not in _NON_TABLES and tbl.lower() not in seen_tbls: + seen_tbls.add(tbl.lower()) + tbl_nid = table_nids.get(tbl.lower()) or _make_id(stem, tbl) + _add_edge(obj_nid, tbl_nid, "reads_from", line) + for rm in re.finditer(r"\bUPDATE\s+([\w$]+)", text, re.IGNORECASE): + tbl = rm.group(1) + if tbl.lower() not in _NON_TABLES and tbl.lower() not in seen_tbls: + seen_tbls.add(tbl.lower()) + tbl_nid = table_nids.get(tbl.lower()) or _make_id(stem, tbl) + _add_edge(obj_nid, tbl_nid, "reads_from", line) + for child in node.children: walk(child) @@ -2767,6 +2856,29 @@ def extract_sql(path: Path) -> dict: if stmt.type == "statement": for child in stmt.children: walk(child) + elif stmt.type in ("fb_proc_or_trigger", "set_term", "declare_external_function"): + walk(stmt) + + # Global regex fallback: catch any REFERENCES missed due to ERROR nodes in the parse tree + # (e.g. Firebird COMPUTED BY columns push constraints out of the tree entirely). + # Snapshot after tree walk so we don't re-emit edges already captured above. + emitted = {(e["source"], e["target"]) for e in edges if e["relation"] == "references"} + src_text = source.decode("utf-8", errors="replace") + for m in re.finditer(r"CREATE\s+TABLE\s+([\w$]+)\s*\(", src_text, re.IGNORECASE): + tbl_name = m.group(1) + tbl_nid = table_nids.get(tbl_name.lower()) + if tbl_nid is None: + continue + tbl_line = src_text[: m.start()].count("\n") + 1 + tail = src_text[m.start():] + end = re.search(r"(?:^|\n)(?:CREATE|SET\s+TERM|ALTER)\s", tail[1:], re.IGNORECASE) + block = tail[: end.start() + 1] if end else tail + for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", block, re.IGNORECASE): + ref_name = rm.group(1) + ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name) + if (tbl_nid, ref_nid) not in emitted: + _add_edge(tbl_nid, ref_nid, "references", tbl_line) + emitted.add((tbl_nid, ref_nid)) return {"nodes": nodes, "edges": edges} diff --git a/graphify/report.py b/graphify/report.py index b6f45a8..fed26fa 100644 --- a/graphify/report.py +++ b/graphify/report.py @@ -28,6 +28,10 @@ def generate( ) -> str: today = date.today().isoformat() + # JSON deserialization produces string keys; normalize to int so .get(cid) works. + if community_labels: + community_labels = {int(k) if isinstance(k, str) else k: v for k, v in community_labels.items()} + confidences = [d.get("confidence", "EXTRACTED") for _, _, d in G.edges(data=True)] total = len(confidences) or 1 ext_pct = round(confidences.count("EXTRACTED") / total * 100) diff --git a/graphify/skill-aider.md b/graphify/skill-aider.md index eeb4517..7146e87 100644 --- a/graphify/skill-aider.md +++ b/graphify/skill-aider.md @@ -716,10 +716,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) Path('.graphify_incremental.json').write_text(json.dumps(result)) -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -743,6 +747,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/graphify/skill-claw.md b/graphify/skill-claw.md index cc86d5c..f5186c3 100644 --- a/graphify/skill-claw.md +++ b/graphify/skill-claw.md @@ -716,10 +716,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) Path('.graphify_incremental.json').write_text(json.dumps(result)) -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -743,6 +747,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/graphify/skill-codex.md b/graphify/skill-codex.md index c04e6a0..d046a03 100644 --- a/graphify/skill-codex.md +++ b/graphify/skill-codex.md @@ -775,10 +775,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) Path('.graphify_incremental.json').write_text(json.dumps(result)) -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -802,6 +806,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/graphify/skill-copilot.md b/graphify/skill-copilot.md index f25f7af..a9baf9d 100644 --- a/graphify/skill-copilot.md +++ b/graphify/skill-copilot.md @@ -793,10 +793,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) Path('graphify-out/.graphify_incremental.json').write_text(json.dumps(result)) -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -820,6 +824,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/graphify/skill-droid.md b/graphify/skill-droid.md index 5dbf50b..33970a5 100644 --- a/graphify/skill-droid.md +++ b/graphify/skill-droid.md @@ -772,10 +772,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) Path('.graphify_incremental.json').write_text(json.dumps(result)) -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -799,6 +803,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/graphify/skill-kiro.md b/graphify/skill-kiro.md index fe8638a..952a255 100644 --- a/graphify/skill-kiro.md +++ b/graphify/skill-kiro.md @@ -715,10 +715,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) Path('.graphify_incremental.json').write_text(json.dumps(result)) -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -742,6 +746,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/graphify/skill-opencode.md b/graphify/skill-opencode.md index c679d72..22a5351 100644 --- a/graphify/skill-opencode.md +++ b/graphify/skill-opencode.md @@ -825,10 +825,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) Path('graphify-out/.graphify_incremental.json').write_text(json.dumps(result)) -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -852,6 +856,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/graphify/skill-pi.md b/graphify/skill-pi.md index 1905e6c..4e4dd9e 100644 --- a/graphify/skill-pi.md +++ b/graphify/skill-pi.md @@ -715,10 +715,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) Path('.graphify_incremental.json').write_text(json.dumps(result)) -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -742,6 +746,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/graphify/skill-trae.md b/graphify/skill-trae.md index 018aea1..2e2dc9a 100644 --- a/graphify/skill-trae.md +++ b/graphify/skill-trae.md @@ -750,10 +750,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2)) Path('.graphify_incremental.json').write_text(json.dumps(result)) -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -777,6 +781,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/graphify/skill-windows.md b/graphify/skill-windows.md index 59c9f64..2c9a034 100644 --- a/graphify/skill-windows.md +++ b/graphify/skill-windows.md @@ -863,10 +863,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2, ensure_ascii=False)) Path('graphify-out/.graphify_incremental.json').write_text(json.dumps(result, ensure_ascii=False), encoding="utf-8") -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') '@ | Out-File -FilePath graphify-out\.graphify_step_for_update_incremental_re_extracti_19.py -Encoding utf8 & (Get-Content graphify-out\.graphify_python) graphify-out\.graphify_step_for_update_incremental_re_extracti_19.py Remove-Item -ErrorAction SilentlyContinue graphify-out\.graphify_step_for_update_incremental_re_extracti_19.py @@ -894,6 +898,23 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```powershell +if (-not (Test-Path graphify-out\.graphify_extract.json)) { + Write-Host '[graphify update] Only deletions -- creating empty extraction for merge.' + @' +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +'@ | Out-File -FilePath graphify-out\.graphify_step_empty_extract.py -Encoding utf8 + & (Get-Content graphify-out\.graphify_python) graphify-out\.graphify_step_empty_extract.py + Remove-Item -ErrorAction SilentlyContinue graphify-out\.graphify_step_empty_extract.py +} +``` + + Then: ```powershell diff --git a/graphify/skill.md b/graphify/skill.md index 89b8a85..9401f2f 100644 --- a/graphify/skill.md +++ b/graphify/skill.md @@ -769,10 +769,14 @@ result = detect_incremental(Path('INPUT_PATH')) new_total = result.get('new_total', 0) print(json.dumps(result, indent=2, ensure_ascii=False)) Path('graphify-out/.graphify_incremental.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\") -if new_total == 0: +deleted = list(result.get('deleted_files', [])) +if new_total == 0 and not deleted: print('No files changed since last run. Nothing to update.') raise SystemExit(0) -print(f'{new_total} new/changed file(s) to re-extract.') +if deleted: + print(f'{len(deleted)} deleted file(s) to prune.') +if new_total > 0: + print(f'{new_total} new/changed file(s) to re-extract.') " ``` @@ -814,6 +818,21 @@ If `code_only` is True: print `[graphify update] Code-only changes detected - sk If `code_only` is False (any changed file is a doc/paper/image): run the full Steps 3A–3C pipeline as normal. + +If no new files exist (only deletions), create an empty extraction so the merge step can prune: + +```bash +if [ ! -f graphify-out/.graphify_extract.json ]; then + echo '[graphify update] Only deletions -- creating empty extraction for merge.' + $(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +fi +``` + + Then: ```bash diff --git a/pyproject.toml b/pyproject.toml index 151b73b..59e759c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" -version = "0.8.3" +version = "0.8.4" description = "AI coding assistant skill (Claude Code, Codex, OpenCode, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Google Antigravity) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph" readme = "README.md" license = { file = "LICENSE" }