From 07b9143d4b90b1e1cb88dc71423f742a501efd29 Mon Sep 17 00:00:00 2001 From: safishamsi Date: Wed, 5 Aug 2026 22:08:29 +0100 Subject: [PATCH] fix(hyperedge,skill): merge/load hyperedge integrity + community labels; bump to 0.9.34 #2486 (thanks @adminwat): normalize dict-shaped hyperedge members to ids (or drop with a warning) so a malformed hyperedge can't abort a completed merge with a TypeError. #2484 (thanks @sortakool; approach from @oleksii-tumanov's #1691): merge-graphs relabels hyperedge member ids and ids with the repo prefix, unions both inputs' hyperedges instead of clobbering, and writes both persistence slots. #2485 (thanks @sortakool): build_from_json reads hyperedges from the top-level and nested slots; a full validation wipeout is reported loudly. #2490 (thanks @PapiScholz): the skill Step-5 flow passes curated community_labels to to_json, so graph.json ships community_name. Co-Authored-By: Claude Opus 4.8 (1M context) --- CHANGELOG.md | 11 +- graphify/build.py | 138 +++++++++++++++--- graphify/cli.py | 89 +++++++++-- graphify/export.py | 28 ++++ graphify/llm.py | 12 ++ graphify/skill-agents.md | 8 + graphify/skill-aider.md | 9 ++ graphify/skill-amp.md | 8 + graphify/skill-claw.md | 8 + graphify/skill-codex.md | 8 + graphify/skill-copilot.md | 8 + graphify/skill-devin.md | 9 ++ graphify/skill-droid.md | 8 + graphify/skill-kilo.md | 8 + graphify/skill-kiro.md | 8 + graphify/skill-opencode.md | 8 + graphify/skill-pi.md | 8 + graphify/skill-trae.md | 8 + graphify/skill-vscode.md | 8 + graphify/skill-windows.md | 8 + graphify/skill.md | 8 + pyproject.toml | 2 +- tests/test_cli_export.py | 109 ++++++++++++++ tests/test_community_labels_skill.py | 131 +++++++++++++++++ tests/test_hyperedge_member_shapes.py | 93 ++++++++++++ tests/test_hyperedge_roundtrip.py | 91 ++++++++++++ tests/test_merge_graphs_cli.py | 80 ++++++++++ .../expected/graphify__skill-agents.md | 8 + .../expected/graphify__skill-aider.md | 9 ++ .../skillgen/expected/graphify__skill-amp.md | 8 + .../skillgen/expected/graphify__skill-claw.md | 8 + .../expected/graphify__skill-codex.md | 8 + .../expected/graphify__skill-copilot.md | 8 + .../expected/graphify__skill-devin.md | 9 ++ .../expected/graphify__skill-droid.md | 8 + .../skillgen/expected/graphify__skill-kilo.md | 8 + .../skillgen/expected/graphify__skill-kiro.md | 8 + .../expected/graphify__skill-opencode.md | 8 + tools/skillgen/expected/graphify__skill-pi.md | 8 + .../skillgen/expected/graphify__skill-trae.md | 8 + .../expected/graphify__skill-vscode.md | 8 + .../expected/graphify__skill-windows.md | 8 + tools/skillgen/expected/graphify__skill.md | 8 + tools/skillgen/fragments/core/aider.md | 9 ++ tools/skillgen/fragments/core/core.md | 8 + tools/skillgen/fragments/core/devin.md | 9 ++ tools/skillgen/gen.py | 26 ++++ 47 files changed, 1062 insertions(+), 34 deletions(-) create mode 100644 tests/test_community_labels_skill.py create mode 100644 tests/test_hyperedge_member_shapes.py create mode 100644 tests/test_hyperedge_roundtrip.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 148fd60..03d264a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,7 +2,16 @@ Full release notes with details on each version: [GitHub Releases](https://github.com/safishamsi/graphify/releases) -## 0.9.33 (unreleased) +## 0.9.34 (unreleased) + +- Fix: C# receiver typing no longer drops a true call when a same-named variable is declared untypeably elsewhere in the method (#2472, thanks @JensD-git). Receiver types are now tracked per lexical declaration scope and resolved by the call's position, so a typed `static` local-function parameter keeps resolving even when an `out var` reuses the name in the enclosing body. This fixes a regression from 0.9.32 (#2346). Cross-method independence (#2299) and field-conflict poisoning are unchanged; an `out var` receiver itself remains untyped. +- Fix: `graphify path` (and the MCP `shortest_path` tool) now respect edge direction by default instead of running on an undirected view, so a returned path no longer traverses edges backwards (#2487, thanks @luliaz0601). Direction is recovered from the stored `_src`/`_tgt` markers. Pass `--undirected` (CLI) or `undirected=true` (MCP) to search ignoring direction; when no directed path exists the command says so instead of silently returning a reversed one. +- Fix: semantic extraction no longer aborts at merge with a `TypeError` when a hyperedge carries dict-shaped members (#2486, thanks @adminwat). Members are normalized to ids (or dropped with a warning) so a malformed hyperedge can no longer destroy a completed extraction. +- Fix: `graphify merge-graphs` no longer drops hyperedges (#2484, thanks @sortakool, and @oleksii-tumanov for the approach in #1691). Hyperedge member ids and ids are now relabeled with the per-repo prefix, both inputs' hyperedges are unioned instead of one clobbering the other, and they are written to both the top-level and nested slots. +- Fix: `build_from_json` now reads hyperedges from both the top-level and nested `graph` slots, so label and re-cluster runs no longer silently empty a graph's hyperedge set (#2485, thanks @sortakool); a full validation wipeout is now reported loudly. +- Fix: the skill flow now passes the curated community labels to `to_json`, so `graph.json` ships with `community_name` on nodes instead of dropping it (#2490, thanks @PapiScholz). + +## 0.9.33 (2026-08-05) - Fix: the C# `partial class` merge (#2332) no longer conflates two same-named classes that live in different assemblies (#2411, thanks @JensD-git). The merge now keys on assembly (nearest ancestor directory containing a `.csproj`/`.fsproj`/`.vbproj`) in addition to namespace and name, so genuine partial halves within one project still merge while same-name types in separate projects stay distinct. A corpus with no project file keeps merging by namespace and name as before. - Fix: `graphify update` no longer drops member-call and `indirect_call` edges from a changed file into an unchanged target (#2437, #2438, thanks @aryanbonigala). Incremental re-resolution now sees the unchanged corpus (its nodes, `contains`/`method` edges, and the `_callable` markers, which now persist to `graph.json` like `_origin`), so cross-file calls survive an incremental rebuild while edges to a genuinely removed target are still evicted. diff --git a/graphify/build.py b/graphify/build.py index a082197..6931b7b 100644 --- a/graphify/build.py +++ b/graphify/build.py @@ -99,36 +99,68 @@ _FILE_TYPE_SYNONYMS = { _HE_MEMBER_ALIASES = ("members", "node_ids") +def _coerce_hyperedge_member_refs(he: dict, members: list) -> list: + """Coerce a hyperedge member list to hashable scalar ids, deduped in order. + + LLM/subagent drift sometimes emits a member as an object (``{"id": "a_ts"}``) + instead of a bare id string. Left uncoerced, the dict member is unhashable, + so the semantic-rekey pass's ``_rekey.get(n, n)`` raised ``TypeError`` and + aborted the whole merge — destroying a completed extraction (#2486). Object + members collapse to their non-empty ``id`` (numeric ids str-coerced via + ``_coerce_id``, matching #2326); members with no usable id are dropped with + a stderr WARNING naming the hyperedge, never a crash. Hashable scalar refs + pass through unchanged. A hyperedge that loses every member this way falls + to the existing no-valid-members drop-with-warning in ``build_from_json``. + """ + seen: set = set() + coerced: list = [] + for ref in members: + if isinstance(ref, dict): + inner = _coerce_id(ref.get("id")) + if inner in (None, "") or not _hashable(inner): + print( + f"[graphify] WARNING: hyperedge " + f"'{he.get('id', '?')}' has a member object with no usable " + f"'id' ({ref!r}); dropping that member.", + file=sys.stderr, + ) + continue + ref = inner + elif not _hashable(ref): + print( + f"[graphify] WARNING: hyperedge " + f"'{he.get('id', '?')}' has an unusable member reference " + f"{ref!r}; dropping that member.", + file=sys.stderr, + ) + continue + if ref in seen: + continue + seen.add(ref) + coerced.append(ref) + return coerced + + def _normalize_hyperedge_members(he: object) -> None: """Canonicalize a hyperedge's member list onto the `nodes` key, in place. If `nodes` is already a list it wins (canonical), and only stray alias keys are dropped. Otherwise the first alias (`members`, then `node_ids`) that is a - list is moved to `nodes`, deduped preserving order, with a single stderr - WARNING naming the hyperedge id and alias used. Leftover alias keys are - always removed so downstream code never re-reads them. + list is moved to `nodes`, with a single stderr WARNING naming the hyperedge + id and alias used. Leftover alias keys are always removed so downstream code + never re-reads them. Whichever branch supplied the list, member VALUES are + coerced to hashable scalar ids and deduped preserving order (#2486) — see + ``_coerce_hyperedge_member_refs``. """ if not isinstance(he, dict): return - if not isinstance(he.get("nodes"), list): + if isinstance(he.get("nodes"), list): + he["nodes"] = _coerce_hyperedge_member_refs(he, he["nodes"]) + else: for alias in _HE_MEMBER_ALIASES: val = he.get(alias) if isinstance(val, list): - seen: set = set() - deduped: list = [] - for ref in val: - try: - is_dupe = ref in seen - except TypeError: - is_dupe = False # unhashable ref: keep it, validator flags it - if is_dupe: - continue - try: - seen.add(ref) - except TypeError: - pass - deduped.append(ref) - he["nodes"] = deduped + he["nodes"] = _coerce_hyperedge_member_refs(he, val) print( f"[graphify] WARNING: hyperedge " f"'{he.get('id', '?')}' uses field '{alias}' instead of " @@ -185,6 +217,16 @@ def _coerce_id(value: object) -> object: return str(value) +def _hashable(value: object) -> bool: + """True when value can be a dict key / set member (same probe as the + inline ``try: hash(m)`` in build_from_json's hyperedge revalidation).""" + try: + hash(value) + except TypeError: + return False + return True + + def _coerce_non_string_ids(extraction: dict) -> None: """Coerce numeric node ids and edge/hyperedge references to str, in place (#2326). @@ -622,6 +664,17 @@ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Pat if "edges" not in extraction and "links" in extraction: extraction = dict(extraction, edges=extraction["links"]) + # Hyperedge persistence is dual-slot (#2485): to_json writes BOTH a + # top-level `hyperedges` key AND the nested `graph.hyperedges` (node_link + # graph attrs), but node_link_data-only writers emit just the nested slot. + # Fold the nested slot onto the top-level key ONCE, so every downstream + # pass (_coerce_non_string_ids, _normalize_hyperedge_members, the member + # revalidation before G.graph["hyperedges"] is set) reads one location. + if "hyperedges" not in extraction and isinstance( + (extraction.get("graph") or {}).get("hyperedges"), list + ): + extraction = dict(extraction, hyperedges=extraction["graph"]["hyperedges"]) + # Numeric ids from a loose backend become str before anything keys on them # (#2326) — after the links remap so aliased edges are covered too. _coerce_non_string_ids(extraction) @@ -718,7 +771,12 @@ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Pat edge["target"] = _rekey[edge["target"]] for he in extraction.get("hyperedges", []) or []: if isinstance(he, dict) and isinstance(he.get("nodes"), list): - he["nodes"] = [_rekey.get(n, n) for n in he["nodes"]] + # Guard on hashability (#2486): _normalize_hyperedge_members + # has already coerced members above, but a still-unhashable ref + # must pass through rather than abort the merge on dict.get. + he["nodes"] = [ + _rekey.get(n, n) if _hashable(n) else n for n in he["nodes"] + ] # Merge markdown quick-scan bare doc nodes into their semantic `_doc` twin # for the same file, so a document is one node regardless of which pipeline @@ -745,7 +803,10 @@ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Pat extraction["edges"] = _new_edges for he in extraction.get("hyperedges", []) or []: if isinstance(he, dict) and isinstance(he.get("nodes"), list): - he["nodes"] = [_doc_remap.get(n, n) for n in he["nodes"]] + # Same hashability guard as the _rekey pass above (#2486). + he["nodes"] = [ + _doc_remap.get(n, n) if _hashable(n) else n for n in he["nodes"] + ] G: nx.Graph = nx.DiGraph() if directed else nx.Graph() for node in extraction.get("nodes", []): @@ -1092,6 +1153,19 @@ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Pat kept_hyperedges.append(he) if kept_hyperedges: G.graph["hyperedges"] = kept_hyperedges + else: + # Full wipeout (#2485): every incoming hyperedge failed member + # revalidation. Store an EXPLICIT empty list — distinct from + # "this graph never carried hyperedge metadata" — and say loudly + # that the persisted set is about to be emptied, so the per-edge + # warnings above can't scroll past unnoticed. + G.graph["hyperedges"] = [] + print( + f"[graphify] WARNING: all {len(hyperedges)} hyperedge(s) were " + f"dropped by member revalidation; graph.json's hyperedge set " + f"will be emptied on the next export.", + file=sys.stderr, + ) # Runs LAST, after the alias-competition above (which relies on file-node # labels still being bare basenames): give colliding-basename file nodes a # directory-qualified display label so lookup/discovery can disambiguate @@ -1605,6 +1679,28 @@ def prefix_graph_for_global(G: nx.Graph, repo_tag: str) -> nx.Graph: data["_src"] = relabel[data["_src"]] if "_tgt" in data and data["_tgt"] in relabel: data["_tgt"] = relabel[data["_tgt"]] + # Out-of-band hyperedges must be relabeled with the nodes (#2484, after + # @oleksii-tumanov's diagnosis in PR #1691): relabel_nodes copies graph + # attrs by reference, so member ids kept their pre-prefix form and dangled + # after a cross-repo merge. Rebuild the list (fresh dicts — the input + # graph's list is shared with H) with member ids mapped through the same + # relabel table, and prefix the hyperedge id itself so same-named + # hyperedges from different repos cannot collide when merged. + hyperedges = H.graph.get("hyperedges") + if isinstance(hyperedges, list): + rewritten = [] + for he in hyperedges: + if isinstance(he, dict): + he = dict(he) + if isinstance(he.get("nodes"), list): + he["nodes"] = [ + relabel.get(m, m) if _hashable(m) else m + for m in he["nodes"] + ] + if he.get("id"): + he["id"] = f"{repo_tag}::{he['id']}" + rewritten.append(he) + H.graph["hyperedges"] = rewritten return H diff --git a/graphify/cli.py b/graphify/cli.py index c7d9a87..534bed6 100644 --- a/graphify/cli.py +++ b/graphify/cli.py @@ -1175,7 +1175,8 @@ def dispatch_command(cmd: str) -> None: elif cmd == "path": if len(sys.argv) < 4: print( - 'Usage: graphify path "" "" [--graph path]', + 'Usage: graphify path "" "" [--graph path] ' + "[--directed|--undirected]", file=sys.stderr, ) sys.exit(1) @@ -1187,9 +1188,30 @@ def dispatch_command(cmd: str) -> None: target_label = sys.argv[3] graph_path = _default_graph_path() args = sys.argv[4:] + direction_flag = None for i, a in enumerate(args): if a == "--graph" and i + 1 < len(args): graph_path = args[i + 1] + elif a == "--directed": + if direction_flag == "undirected": + print( + "error: --directed and --undirected are mutually exclusive", + file=sys.stderr, + ) + sys.exit(1) + direction_flag = "directed" + elif a == "--undirected": + if direction_flag == "directed": + print( + "error: --directed and --undirected are mutually exclusive", + file=sys.stderr, + ) + sys.exit(1) + direction_flag = "undirected" + # Directed by default (#2487): direction truth exists in every + # graph.json (arc order on post-#563 files, _src/_tgt markers on legacy + # canonicalized files), so respect it unless the caller opts out. + undirected = direction_flag == "undirected" gp = Path(graph_path).resolve() if not gp.exists(): print(f"error: graph file not found: {gp}", file=sys.stderr) @@ -1244,18 +1266,36 @@ def dispatch_command(cmd: str) -> None: f"(top score {_top:g}, runner-up {_runner:g})", file=sys.stderr, ) - # Deterministic shortest path (#2074): to_undirected(as_view=True) - # iterates neighbors via a hash-seeded set union, so among equal-length - # paths BFS returned an arbitrary route that varied per process. Build a - # sorted, materialized undirected graph so neighbor order — and thus the - # chosen path — is canonical for a given graph.json. - _und = _nx.Graph() - _und.add_nodes_from(sorted(G.nodes)) - _und.add_edges_from(sorted((min(u, v), max(u, v)) for u, v in G.edges())) + # Deterministic shortest path (#2074): hash-seeded neighbor views + # returned an arbitrary route among equal-length paths that varied per + # process. Build a sorted, materialized graph so neighbor order — and + # thus the chosen path — is canonical for a given graph.json. try: - path_nodes = _nx.shortest_path(_und, src_nid, tgt_nid) + if undirected: + _und = _nx.Graph() + _und.add_nodes_from(sorted(G.nodes)) + _und.add_edges_from(sorted((min(u, v), max(u, v)) for u, v in G.edges())) + path_nodes = _nx.shortest_path(_und, src_nid, tgt_nid) + else: + # Directed by default (#2487). True direction is NOT raw arc + # order: legacy canonicalized files persist a flipped arc with + # _src/_tgt markers (#2309), so build the digraph from _src/_tgt + # (falling back to the loaded arc) rather than to_directed(). + _dg = _nx.DiGraph() + _dg.add_nodes_from(sorted(G.nodes)) + _dg.add_edges_from(sorted( + (d.get("_src", u), d.get("_tgt", v)) for u, v, d in G.edges(data=True) + )) + path_nodes = _nx.shortest_path(_dg, src_nid, tgt_nid) except (_nx.NetworkXNoPath, _nx.NodeNotFound): - print(f"No path found between '{source_label}' and '{target_label}'.") + if undirected: + print(f"No path found between '{source_label}' and '{target_label}'.") + else: + print( + f"No directed path found between '{source_label}' and " + f"'{target_label}'. Re-run with --undirected to search " + "ignoring edge direction." + ) sys.exit(0) hops = len(path_nodes) - 1 segments = [] @@ -2145,6 +2185,12 @@ def dispatch_command(cmd: str) -> None: G = _jg.node_link_graph(data, edges="links") except TypeError: G = _jg.node_link_graph(data) + # node_link_graph restores only the nested `graph.hyperedges` slot; + # a graph.json whose hyperedges live only at the top level (the + # other half of to_json's dual-slot shape, #2485) would silently + # lose them here. Fall back to the top-level key (#2484). + if "hyperedges" not in G.graph and isinstance(data.get("hyperedges"), list): + G.graph["hyperedges"] = data["hyperedges"] graphs.append(G) # nx.compose requires all graphs to be the same type. When input graphs # come from different sources (e.g. an AST-only run vs a full LLM run) one @@ -2170,9 +2216,25 @@ def dispatch_command(cmd: str) -> None: if len(set(naive_tags)) != len(naive_tags): print(f" note: repo dir names collide; using distinct tags: {', '.join(repo_tags)}") merged = _nx.Graph() + # nx.compose merges graph attrs with dict.update, so each iteration + # CLOBBERED the previously accumulated hyperedge list — only the last + # input's hyperedges survived (#2484, after @oleksii-tumanov's + # diagnosis in PR #1691). Collect every input's prefixed hyperedges + # and re-attach the union after composing. + collected_hyperedges: list = [] for G, repo_tag in zip(graphs, repo_tags): prefixed = _to_simple(_prefix(G, repo_tag)) + hes = prefixed.graph.get("hyperedges") + if isinstance(hes, list): + collected_hyperedges.extend(h for h in hes if isinstance(h, dict)) merged = _nx.compose(merged, prefixed) + # Drop whatever compose left behind (the last input's list, possibly + # with internal duplicates) so attach_hyperedges dedups the full + # collection by id from a clean slate. + merged.graph.pop("hyperedges", None) + if collected_hyperedges: + from graphify.export import attach_hyperedges as _attach + _attach(merged, collected_hyperedges) try: out_data = _jg.node_link_data(merged, edges="links") except TypeError: @@ -2184,6 +2246,11 @@ def dispatch_command(cmd: str) -> None: if tsrc is not None and ttgt is not None: link["source"] = tsrc link["target"] = ttgt + # Persist BOTH hyperedge slots (#2484): node_link_data only nests graph + # attrs under `graph`, so without this line the union would survive + # solely in the slot historic readers ignored (#2485). Mirror to_json's + # dual-slot shape so every writer agrees. + out_data["hyperedges"] = merged.graph.get("hyperedges", []) out_path.parent.mkdir(parents=True, exist_ok=True) from graphify.paths import write_json_atomic as _wja _wja(out_path, out_data, indent=2) diff --git a/graphify/export.py b/graphify/export.py index 328b525..a732a8a 100644 --- a/graphify/export.py +++ b/graphify/export.py @@ -311,6 +311,34 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, if true_src is not None and true_tgt is not None: link["source"] = true_src link["target"] = true_tgt + if "hyperedges" not in getattr(G, "graph", {}): + # Hardening (#2485): a graph with NO hyperedges key at all was built by + # a path that never engaged hyperedge metadata — distinct from an + # intentional empty set ([], which build_from_json now stores + # explicitly after a full-wipeout revalidation). If the file on disk + # already holds a non-empty set, emptying it without a trace is silent + # data loss; warn loudly so the wipeout is attributable. We still write + # the graph's truth rather than preserving the stale set — resurrecting + # hyperedges whose members may no longer exist would reintroduce the + # dangling-member shape #1916 removed. + _prev_hyperedges = None + try: + if existing_path.exists(): + from graphify.security import check_graph_file_size_cap + check_graph_file_size_cap(existing_path) + _prev = json.loads(existing_path.read_text(encoding="utf-8")) + if isinstance(_prev, dict): + _prev_hyperedges = _prev.get("hyperedges") + except Exception: + _prev_hyperedges = None + if _prev_hyperedges: + print( + f"[graphify] WARNING: graph carries no hyperedge metadata but " + f"{existing_path} already holds {len(_prev_hyperedges)} " + f"hyperedge(s); writing an empty set. Rebuild from the original " + f"extraction if this is unexpected.", + file=sys.stderr, + ) data["hyperedges"] = getattr(G, "graph", {}).get("hyperedges", []) commit = built_at_commit if built_at_commit is not None else _git_head() if commit: diff --git a/graphify/llm.py b/graphify/llm.py index 30d7a6d..606d373 100644 --- a/graphify/llm.py +++ b/graphify/llm.py @@ -961,6 +961,18 @@ def _sanitize_fragment(parsed: dict) -> dict: parsed[key] = [] continue parsed[key] = [entry for entry in value if isinstance(entry, dict)] + # Coerce hyperedge member refs to hashable scalar ids (#2486): a model can + # emit a member as an object ({"id": "a_ts"}) instead of a bare id. The + # per-entry filter above only checks the hyperedge dicts themselves, so the + # bad member shape used to persist into the semantic cache and crash + # build_from_json's rekey pass much later (a dict is unhashable). Applying + # the shared coercion at this parse chokepoint keeps the cache clean. + hyperedges = parsed.get("hyperedges") + if hyperedges: + from graphify.build import _coerce_hyperedge_member_refs + for he in hyperedges: + if isinstance(he.get("nodes"), list): + he["nodes"] = _coerce_hyperedge_member_refs(he, he["nodes"]) return parsed diff --git a/graphify/skill-agents.md b/graphify/skill-agents.md index afb4ecc..b174e6d 100644 --- a/graphify/skill-agents.md +++ b/graphify/skill-agents.md @@ -486,6 +486,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -507,6 +508,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-aider.md b/graphify/skill-aider.md index 4f03ccb..4996beb 100644 --- a/graphify/skill-aider.md +++ b/graphify/skill-aider.md @@ -458,6 +458,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) @@ -478,6 +479,12 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) Path('.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') print('Report updated with community labels') " ``` @@ -896,6 +903,8 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, '.') Path('graphify-out/GRAPH_REPORT.md').write_text(report) +# No community_labels here - 'labels' are still placeholders at this point; +# Step 5 re-exports graph.json with the curated names (#2490). to_json(G, communities, 'graphify-out/graph.json') analysis = { diff --git a/graphify/skill-amp.md b/graphify/skill-amp.md index afb4ecc..b174e6d 100644 --- a/graphify/skill-amp.md +++ b/graphify/skill-amp.md @@ -486,6 +486,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -507,6 +508,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-claw.md b/graphify/skill-claw.md index d98865c..cef4d5c 100644 --- a/graphify/skill-claw.md +++ b/graphify/skill-claw.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-codex.md b/graphify/skill-codex.md index 0c821a2..bd8a020 100644 --- a/graphify/skill-codex.md +++ b/graphify/skill-codex.md @@ -486,6 +486,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -507,6 +508,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-copilot.md b/graphify/skill-copilot.md index d98865c..cef4d5c 100644 --- a/graphify/skill-copilot.md +++ b/graphify/skill-copilot.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-devin.md b/graphify/skill-devin.md index e3e6d2d..f9be846 100644 --- a/graphify/skill-devin.md +++ b/graphify/skill-devin.md @@ -523,6 +523,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) @@ -543,6 +544,12 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') print('Report updated with community labels') " ``` @@ -1032,6 +1039,8 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, '.') Path('graphify-out/GRAPH_REPORT.md').write_text(report) +# No community_labels here - 'labels' are still placeholders at this point; +# Step 5 re-exports graph.json with the curated names (#2490). to_json(G, communities, 'graphify-out/graph.json') analysis = { diff --git a/graphify/skill-droid.md b/graphify/skill-droid.md index c3815d5..370e0d6 100644 --- a/graphify/skill-droid.md +++ b/graphify/skill-droid.md @@ -486,6 +486,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -507,6 +508,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-kilo.md b/graphify/skill-kilo.md index dbb4658..f83cb06 100644 --- a/graphify/skill-kilo.md +++ b/graphify/skill-kilo.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-kiro.md b/graphify/skill-kiro.md index d98865c..cef4d5c 100644 --- a/graphify/skill-kiro.md +++ b/graphify/skill-kiro.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-opencode.md b/graphify/skill-opencode.md index cf5dae4..c121617 100644 --- a/graphify/skill-opencode.md +++ b/graphify/skill-opencode.md @@ -481,6 +481,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -502,6 +503,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-pi.md b/graphify/skill-pi.md index d98865c..cef4d5c 100644 --- a/graphify/skill-pi.md +++ b/graphify/skill-pi.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-trae.md b/graphify/skill-trae.md index b0cbeb1..98deab2 100644 --- a/graphify/skill-trae.md +++ b/graphify/skill-trae.md @@ -487,6 +487,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -508,6 +509,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-vscode.md b/graphify/skill-vscode.md index 3e6bc6b..efb2401 100644 --- a/graphify/skill-vscode.md +++ b/graphify/skill-vscode.md @@ -485,6 +485,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -506,6 +507,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill-windows.md b/graphify/skill-windows.md index 5743845..c48d874 100644 --- a/graphify/skill-windows.md +++ b/graphify/skill-windows.md @@ -511,6 +511,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -532,6 +533,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/graphify/skill.md b/graphify/skill.md index d98865c..cef4d5c 100644 --- a/graphify/skill.md +++ b/graphify/skill.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/pyproject.toml b/pyproject.toml index 5168ebd..7b8e715 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" -version = "0.9.33" +version = "0.9.34" description = "AI coding assistant skill (Claude Code, CodeBuddy, Codex, OpenCode, Kilo Code, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Devin CLI, Google Antigravity) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph" readme = "README.md" license = "Apache-2.0" diff --git a/tests/test_cli_export.py b/tests/test_cli_export.py index 6173a61..a8502c8 100644 --- a/tests/test_cli_export.py +++ b/tests/test_cli_export.py @@ -252,6 +252,115 @@ def test_path_uses_graphify_out_env(tmp_path): assert r.returncode == 0, r.stderr +# ── graphify path direction (#2487) ───────────────────────────────────────── + +def _write_path_graph(tmp_path: Path, nodes: list[str], links: list[dict]) -> Path: + """Write a minimal hand-rolled directed graph.json for path-direction tests.""" + out = tmp_path / "graphify-out" + out.mkdir() + (out / "graph.json").write_text(json.dumps({ + "directed": True, + "multigraph": False, + "graph": {}, + "nodes": [{"id": n, "label": n} for n in nodes], + "links": links, + })) + return out + + +def _calls(src: str, tgt: str) -> dict: + return {"source": src, "target": tgt, "relation": "calls"} + + +def test_path_directed_respects_direction(tmp_path): + _write_path_graph( + tmp_path, ["alpha", "beta", "gamma"], + [_calls("alpha", "beta"), _calls("beta", "gamma")], + ) + r = _run(["path", "alpha", "gamma"], tmp_path) + assert r.returncode == 0, r.stderr + assert r.stdout.count("-->") == 2 + assert "<--" not in r.stdout + # Explicit --directed is the same as the default. + r2 = _run(["path", "alpha", "gamma", "--directed"], tmp_path) + assert r2.returncode == 0, r2.stderr + assert r2.stdout == r.stdout + + +def test_path_directed_backwards_is_no_path(tmp_path): + # Default-change guard (#2487): a plain `path` with no flag is directed, + # so walking the chain backwards must report no directed path. + _write_path_graph( + tmp_path, ["alpha", "beta", "gamma"], + [_calls("alpha", "beta"), _calls("beta", "gamma")], + ) + r = _run(["path", "gamma", "alpha"], tmp_path) + assert r.returncode == 0, r.stderr + assert "No directed path found" in r.stdout + assert "--undirected" in r.stdout + assert "-->" not in r.stdout + assert "<--" not in r.stdout + + +def test_path_undirected_flag_opt_in(tmp_path): + _write_path_graph( + tmp_path, ["alpha", "beta", "gamma"], + [_calls("alpha", "beta"), _calls("beta", "gamma")], + ) + r = _run(["path", "gamma", "alpha", "--undirected"], tmp_path) + assert r.returncode == 0, r.stderr + assert "Shortest path (2 hops)" in r.stdout + assert r.stdout.count("<--calls--") == 2 + assert "-->" not in r.stdout + + +def test_path_directed_legacy_markers(tmp_path): + # Legacy canonicalized file: the persisted arc is flipped (beta->alpha) but + # the _src/_tgt markers carry the true direction alpha->beta. Direction + # truth must come from the markers, not the raw arc order (#2309/#2487). + _write_path_graph( + tmp_path, ["alpha", "beta"], + [{"source": "beta", "target": "alpha", + "_src": "alpha", "_tgt": "beta", "relation": "calls"}], + ) + r = _run(["path", "alpha", "beta"], tmp_path) + assert r.returncode == 0, r.stderr + assert "Shortest path (1 hops)" in r.stdout + assert "-->" in r.stdout + assert "<--" not in r.stdout + r2 = _run(["path", "beta", "alpha"], tmp_path) + assert r2.returncode == 0, r2.stderr + assert "No directed path found" in r2.stdout + + +def test_path_directed_deterministic(tmp_path): + # Diamond with two equal-length directed routes: the chosen route must not + # depend on the process hash seed (#2074 discipline for the digraph too). + _write_path_graph( + tmp_path, ["start", "left", "right", "goal"], + [_calls("start", "left"), _calls("left", "goal"), + _calls("start", "right"), _calls("right", "goal")], + ) + outputs = [] + for seed in ("0", "1"): + env = os.environ.copy() + env["PYTHONHASHSEED"] = seed + r = _run(["path", "start", "goal"], tmp_path, env=env) + assert r.returncode == 0, r.stderr + assert "-->" in r.stdout + outputs.append(r.stdout) + assert outputs[0] == outputs[1] + + +def test_path_flags_mutually_exclusive(tmp_path): + _write_path_graph( + tmp_path, ["alpha", "beta"], [_calls("alpha", "beta")], + ) + r = _run(["path", "alpha", "beta", "--directed", "--undirected"], tmp_path) + assert r.returncode != 0 + assert "mutually exclusive" in r.stderr + + # ── graphify explain ───────────────────────────────────────────────────────── def test_explain_runs_without_error(tmp_path): diff --git a/tests/test_community_labels_skill.py b/tests/test_community_labels_skill.py new file mode 100644 index 0000000..5184190 --- /dev/null +++ b/tests/test_community_labels_skill.py @@ -0,0 +1,131 @@ +"""Curated community labels must reach the persisted graph.json (#2490). + +Two guards: + +1. The ``to_json`` export gate: nodes get ``community_name`` only when the + ``community_labels`` kwarg is passed, so any Step-5 flow that curates labels + but omits the kwarg ships a graph.json without community names. + +2. A template lint over the generated ``graphify/skill*.md`` bodies (and the + fragments they render from): the Step-5 / post-labels code block — the one + that builds the curated ``labels = LABELS_DICT`` dict — must re-export + ``graphify-out/graph.json`` with ``community_labels=labels``. This locks the + #2490 fix so a future template edit cannot silently drop the kwarg again. +""" +from __future__ import annotations + +import json +import tempfile +from pathlib import Path + +import networkx as nx +import pytest + +from graphify.export import to_json + +REPO_ROOT = Path(__file__).resolve().parent.parent +FRAGMENTS_DIR = REPO_ROOT / "tools" / "skillgen" / "fragments" / "core" + + +def _two_community_graph() -> tuple[nx.Graph, dict[int, list[str]]]: + G = nx.Graph() + G.add_node("n1", label="Database", community=0, source_file="app/db.py", type="code") + G.add_node("n2", label="Server", community=0, source_file="app/srv.py", type="code") + G.add_node("n3", label="Cache", community=1, source_file="infra/cache.py", type="code") + G.add_edge("n1", "n2", relation="calls") + communities = {0: ["n1", "n2"], 1: ["n3"]} + return G, communities + + +def test_to_json_community_labels_kwarg_writes_community_name(): + """Passing community_labels stamps community_name on that community's nodes.""" + G, communities = _two_community_graph() + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "graph.json" + assert to_json(G, communities, str(out), community_labels={0: "X"}) + data = json.loads(out.read_text()) + by_id = {n["id"]: n for n in data["nodes"]} + assert by_id["n1"]["community_name"] == "X" + assert by_id["n2"]["community_name"] == "X" + # A community missing from a non-empty labels dict gets the placeholder. + assert by_id["n3"]["community_name"] == "Community 1" + + +def test_to_json_without_labels_kwarg_writes_no_community_name(): + """Omitting the kwarg is the #2490 bug shape: no node carries community_name.""" + G, communities = _two_community_graph() + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "graph.json" + assert to_json(G, communities, str(out)) + data = json.loads(out.read_text()) + assert all("community_name" not in n for n in data["nodes"]) + + +# --- template lint ----------------------------------------------------------- + +def _code_blocks(markdown: str) -> list[str]: + """Fenced code blocks of a markdown body, fence lines excluded.""" + blocks: list[str] = [] + current: list[str] | None = None + for line in markdown.splitlines(): + if line.lstrip().startswith("```"): + if current is None: + current = [] + else: + blocks.append("\n".join(current)) + current = None + continue + if current is not None: + current.append(line) + return blocks + + +def _skill_bodies() -> list[Path]: + paths = sorted(REPO_ROOT.glob("graphify/skill*.md")) + assert paths, "no generated graphify/skill*.md found" + return paths + + +@pytest.mark.parametrize("path", _skill_bodies(), ids=lambda p: p.name) +def test_skill_step5_reexports_graph_json_with_curated_labels(path: Path): + """Every post-labels (LABELS_DICT) block re-exports graph.json with the kwarg. + + The Step-5 block is the only place the curated labels dict exists, so it is + the block that must call ``to_json(..., community_labels=labels)``. Any + ``to_json(...graphify-out/graph.json...)`` call in that block without the + kwarg would ship graph.json nodes with no ``community_name`` (#2490). + """ + text = path.read_text(encoding="utf-8") + post_labels_blocks = [ + b for b in _code_blocks(text) if "labels = LABELS_DICT" in b + ] + assert post_labels_blocks, f"{path.name}: no Step-5 (LABELS_DICT) code block found" + for block in post_labels_blocks: + assert "to_json(G, communities, 'graphify-out/graph.json', community_labels=labels)" in block, ( + f"{path.name}: the post-labels Step-5 block must re-export " + f"graphify-out/graph.json with community_labels=labels (#2490)" + ) + # No label-less graph.json export may coexist in the post-labels block. + for line in block.splitlines(): + if "to_json(" in line and "graphify-out/graph.json" in line: + assert "community_labels=labels" in line, ( + f"{path.name}: post-labels to_json call is missing " + f"community_labels=labels: {line.strip()!r}" + ) + + +@pytest.mark.parametrize( + "fragment", sorted(FRAGMENTS_DIR.glob("*.md")), ids=lambda p: p.name +) +def test_core_fragments_step5_reexport_with_curated_labels(fragment: Path): + """Same lint at the source of truth: the core fragments skillgen renders from.""" + text = fragment.read_text(encoding="utf-8") + post_labels_blocks = [ + b for b in _code_blocks(text) if "labels = LABELS_DICT" in b + ] + assert post_labels_blocks, f"{fragment.name}: no Step-5 (LABELS_DICT) code block found" + for block in post_labels_blocks: + assert "to_json(G, communities, 'graphify-out/graph.json', community_labels=labels)" in block, ( + f"{fragment.name}: the post-labels Step-5 block must re-export " + f"graphify-out/graph.json with community_labels=labels (#2490)" + ) diff --git a/tests/test_hyperedge_member_shapes.py b/tests/test_hyperedge_member_shapes.py new file mode 100644 index 0000000..678bfa5 --- /dev/null +++ b/tests/test_hyperedge_member_shapes.py @@ -0,0 +1,93 @@ +"""Dict-shaped hyperedge member refs must never abort a build (#2486). + +LLM/subagent drift sometimes emits a hyperedge member as an object +(``{"id": "a_ts"}``) instead of a bare id string. A dict is unhashable, so the +semantic-rekey pass's ``_rekey.get(n, n)`` used to raise ``TypeError`` and +abort the whole merge — destroying a completed extraction. The fix coerces +member values at ingest (``_normalize_hyperedge_members``) and at the LLM +parse chokepoint (``_sanitize_fragment``), dropping only the individual +unusable member with a stderr WARNING. +""" +from __future__ import annotations + +from graphify.build import build_from_json +from graphify.llm import _sanitize_fragment + + +def _node(nid: str) -> dict: + return {"id": nid, "label": nid, "file_type": "code", "source_file": f"{nid}.ts"} + + +def test_dict_members_coerced_via_canonical_nodes_key(capsys): + extraction = { + "nodes": [_node("a_ts"), _node("b_ts")], + "edges": [], + "hyperedges": [ + # the #2486 repro shape: object members mixed with bare ids, + # including a duplicate that must dedupe after coercion + {"id": "h_flow", "nodes": [{"id": "a_ts"}, "b_ts", {"id": "a_ts"}]}, + ], + } + G = build_from_json(extraction, directed=True) # must not raise + assert set(G.nodes()) == {"a_ts", "b_ts"} + assert G.graph["hyperedges"][0]["nodes"] == ["a_ts", "b_ts"] + + +def test_dict_members_coerced_via_members_alias(capsys): + extraction = { + "nodes": [_node("a_ts"), _node("c_ts")], + "edges": [], + "hyperedges": [ + {"id": "h_alias", "members": ["a_ts", {"id": "c_ts"}]}, + ], + } + G = build_from_json(extraction, directed=True) + (he,) = G.graph["hyperedges"] + assert "members" not in he, "alias key must be folded onto nodes" + assert he["nodes"] == ["a_ts", "c_ts"] + + +def test_member_object_without_id_dropped_with_one_warning(capsys): + extraction = { + "nodes": [_node("a_ts"), _node("b_ts")], + "edges": [], + "hyperedges": [ + {"id": "h_partial", "nodes": [{"label": "no id here"}, "b_ts"]}, + ], + } + G = build_from_json(extraction, directed=True) + assert G.graph["hyperedges"][0]["nodes"] == ["b_ts"] + err = capsys.readouterr().err + warnings = [ + line for line in err.splitlines() + if "no usable 'id'" in line and "h_partial" in line + ] + assert len(warnings) == 1, f"expected exactly one warning, got: {err!r}" + + +def test_hyperedge_losing_all_members_is_dropped_not_fatal(capsys): + extraction = { + "nodes": [_node("a_ts")], + "edges": [], + "hyperedges": [ + {"id": "h_empty", "nodes": [{"label": "no id"}, {"nested": True}]}, + {"id": "h_ok", "nodes": ["a_ts"]}, + ], + } + G = build_from_json(extraction, directed=True) # must not raise + assert [he["id"] for he in G.graph["hyperedges"]] == ["h_ok"] + assert "h_empty" in capsys.readouterr().err + + +def test_sanitize_fragment_coerces_dict_members_to_strings(capsys): + frag = { + "nodes": [], + "edges": [], + "hyperedges": [ + {"id": "h", "nodes": [{"id": "x"}, "y", {"id": 3}, {"label": "no id"}]}, + ], + } + out = _sanitize_fragment(frag) + members = out["hyperedges"][0]["nodes"] + assert members == ["x", "y", "3"], "dict members collapse to their id" + assert all(isinstance(m, str) for m in members) diff --git a/tests/test_hyperedge_roundtrip.py b/tests/test_hyperedge_roundtrip.py new file mode 100644 index 0000000..6f0e9a8 --- /dev/null +++ b/tests/test_hyperedge_roundtrip.py @@ -0,0 +1,91 @@ +"""Hyperedges must survive the dual-slot persistence round-trip (#2485). + +to_json writes hyperedges to BOTH a top-level ``hyperedges`` key and the +nested ``graph.hyperedges`` (node_link_data graph attrs), but build_from_json +used to read only the top-level slot — a nested-only graph.json silently lost +its whole hyperedge set, and export then persisted the wipeout as a durable +``[]``. build_from_json now folds the nested slot onto the top-level key, and +a full member-revalidation wipeout announces itself with one aggregate WARNING +instead of vanishing quietly. +""" +from __future__ import annotations + +import json + +from graphify.build import build_from_json +from graphify.export import to_json + + +def _node(nid: str) -> dict: + return {"id": nid, "label": nid, "file_type": "code", "source_file": f"{nid}.py"} + + +def _roundtrip(G, tmp_path): + out = tmp_path / "graph.json" + assert to_json(G, {}, str(out)) is True + return json.loads(out.read_text(encoding="utf-8")) + + +def test_nested_only_slot_is_read_and_reexported_to_both_slots(tmp_path): + # node_link_data-only writers emit hyperedges solely under graph attrs. + extraction = { + "directed": True, + "multigraph": False, + "graph": {"hyperedges": [{"id": "h1", "nodes": ["a", "b"]}]}, + "nodes": [_node("a"), _node("b")], + "links": [], + } + G = build_from_json(extraction, directed=True) + assert G.graph["hyperedges"] == [{"id": "h1", "nodes": ["a", "b"]}] + + data = _roundtrip(G, tmp_path) + assert data["hyperedges"] == [{"id": "h1", "nodes": ["a", "b"]}] + assert data["graph"]["hyperedges"] == data["hyperedges"], ( + "re-export must carry the set in BOTH slots" + ) + # Full round-trip: rebuilding from the exported file preserves the set exactly. + G2 = build_from_json(json.loads(json.dumps(data)), directed=True) + assert G2.graph["hyperedges"] == [{"id": "h1", "nodes": ["a", "b"]}] + + +def test_top_level_slot_roundtrips_unchanged(tmp_path): + # Control arm: the canonical to_json shape keeps working as before. + extraction = { + "nodes": [_node("a"), _node("b")], + "edges": [], + "hyperedges": [{"id": "h_top", "nodes": ["a", "b"]}], + } + G = build_from_json(extraction, directed=True) + assert G.graph["hyperedges"] == [{"id": "h_top", "nodes": ["a", "b"]}] + + data = _roundtrip(G, tmp_path) + assert data["hyperedges"] == [{"id": "h_top", "nodes": ["a", "b"]}] + assert data["graph"]["hyperedges"] == data["hyperedges"] + G2 = build_from_json(json.loads(json.dumps(data)), directed=True) + assert G2.graph["hyperedges"] == [{"id": "h_top", "nodes": ["a", "b"]}] + + +def test_full_wipeout_emits_one_aggregate_warning(tmp_path, capsys): + # Every member dangles, so the #1916 revalidation drops every hyperedge. + # The wipeout must be loud (one aggregate warning naming the count) and + # explicit (an empty list, not a missing key). + extraction = { + "nodes": [_node("a")], + "edges": [], + "hyperedges": [ + {"id": "h1", "nodes": ["ghost1"]}, + {"id": "h2", "nodes": ["ghost2"]}, + ], + } + G = build_from_json(extraction, directed=True) + err = capsys.readouterr().err + aggregate = [ + line for line in err.splitlines() + if "all 2 hyperedge(s)" in line and "emptied" in line + ] + assert len(aggregate) == 1, f"expected one aggregate warning, got: {err!r}" + assert G.graph["hyperedges"] == [] + + data = _roundtrip(G, tmp_path) + assert data["hyperedges"] == [] + assert data["graph"]["hyperedges"] == [] diff --git a/tests/test_merge_graphs_cli.py b/tests/test_merge_graphs_cli.py index 491919f..e0203a5 100644 --- a/tests/test_merge_graphs_cli.py +++ b/tests/test_merge_graphs_cli.py @@ -167,3 +167,83 @@ def test_merge_graphs_preserves_import_edge_direction(tmp_path): assert repo2_link["source"] == "repo2::main" assert repo2_link["target"] == "repo2::utils" + +def _write_with_hyperedges(p: Path, node_ids: list[str], hyperedges: list[dict], + *, top_level_only: bool = False): + # Mirrors to_json's dual-slot shape: hyperedges live top-level AND under + # the node_link graph attrs. top_level_only drops the nested slot to model + # older writers (#2485). + p.parent.mkdir(parents=True, exist_ok=True) + data = { + "directed": False, "multigraph": False, + "graph": {} if top_level_only else {"hyperedges": hyperedges}, + "nodes": [{"id": n} for n in node_ids], "links": [], + "hyperedges": hyperedges, + } + p.write_text(json.dumps(data)) + + +def test_merge_graphs_carries_hyperedges_from_all_inputs(tmp_path): + # #2484: prefix_graph_for_global never rewrote G.graph["hyperedges"], and + # nx.compose's dict.update graph-attr merge clobbered each prior input's + # list, so at best the LAST graph's hyperedges survived — with stale, + # unprefixed member ids. Both inputs' hyperedges must reach the output, + # relabeled to the prefixed node ids, in BOTH persistence slots. + a = tmp_path / "alpha" / "graphify-out" / "graph.json" + b = tmp_path / "beta" / "graphify-out" / "graph.json" + _write_with_hyperedges(a, ["x", "y"], [{"id": "h_alpha", "nodes": ["x", "y"]}]) + _write_with_hyperedges(b, ["p", "q"], [{"id": "h_beta", "nodes": ["p", "q"]}]) + out = tmp_path / "merged.json" + + r = _run(["merge-graphs", str(a), str(b), "--out", str(out)], tmp_path) + assert r.returncode == 0, r.stderr + data = json.loads(out.read_text()) + + hyperedges = data.get("hyperedges") + assert isinstance(hyperedges, list), "top-level hyperedges slot must be written" + assert {h["id"] for h in hyperedges} == {"alpha::h_alpha", "beta::h_beta"} + assert len(hyperedges) == 2 + + # every member id must resolve in the merged (prefixed) node set + node_ids = {n["id"] for n in data["nodes"]} + for h in hyperedges: + assert set(h["nodes"]) <= node_ids, f"dangling members in {h}" + + # nested slot mirrors the top-level one (to_json's dual-slot shape) + assert data["graph"]["hyperedges"] == hyperedges + + +def test_merge_graphs_hyperedges_dedup_on_shared_prefixed_id(tmp_path): + # Idempotence: a duplicated hyperedge id within an input must not produce + # duplicate entries in the merged output (attach_hyperedges dedups by id). + a = tmp_path / "alpha" / "graphify-out" / "graph.json" + b = tmp_path / "beta" / "graphify-out" / "graph.json" + he = {"id": "h_alpha", "nodes": ["x"]} + _write_with_hyperedges(a, ["x"], [he, dict(he)]) + _write_with_hyperedges(b, ["p"], [{"id": "h_beta", "nodes": ["p"]}]) + out = tmp_path / "merged.json" + + r = _run(["merge-graphs", str(a), str(b), "--out", str(out)], tmp_path) + assert r.returncode == 0, r.stderr + data = json.loads(out.read_text()) + ids = [h["id"] for h in data["hyperedges"]] + assert sorted(ids) == ["alpha::h_alpha", "beta::h_beta"], f"dup survived: {ids}" + + +def test_merge_graphs_reads_top_level_only_hyperedges(tmp_path): + # #2485 skew on the input side: node_link_graph restores only the nested + # graph-attrs slot, so an input whose hyperedges live only at the top + # level used to lose them entirely. + a = tmp_path / "alpha" / "graphify-out" / "graph.json" + b = tmp_path / "beta" / "graphify-out" / "graph.json" + _write_with_hyperedges(a, ["x"], [{"id": "h_top", "nodes": ["x"]}], + top_level_only=True) + _write_with_hyperedges(b, ["p"], []) + out = tmp_path / "merged.json" + + r = _run(["merge-graphs", str(a), str(b), "--out", str(out)], tmp_path) + assert r.returncode == 0, r.stderr + data = json.loads(out.read_text()) + assert [h["id"] for h in data["hyperedges"]] == ["alpha::h_top"] + assert data["hyperedges"][0]["nodes"] == ["alpha::x"] + diff --git a/tools/skillgen/expected/graphify__skill-agents.md b/tools/skillgen/expected/graphify__skill-agents.md index afb4ecc..b174e6d 100644 --- a/tools/skillgen/expected/graphify__skill-agents.md +++ b/tools/skillgen/expected/graphify__skill-agents.md @@ -486,6 +486,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -507,6 +508,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-aider.md b/tools/skillgen/expected/graphify__skill-aider.md index 4f03ccb..4996beb 100644 --- a/tools/skillgen/expected/graphify__skill-aider.md +++ b/tools/skillgen/expected/graphify__skill-aider.md @@ -458,6 +458,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) @@ -478,6 +479,12 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) Path('.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') print('Report updated with community labels') " ``` @@ -896,6 +903,8 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, '.') Path('graphify-out/GRAPH_REPORT.md').write_text(report) +# No community_labels here - 'labels' are still placeholders at this point; +# Step 5 re-exports graph.json with the curated names (#2490). to_json(G, communities, 'graphify-out/graph.json') analysis = { diff --git a/tools/skillgen/expected/graphify__skill-amp.md b/tools/skillgen/expected/graphify__skill-amp.md index afb4ecc..b174e6d 100644 --- a/tools/skillgen/expected/graphify__skill-amp.md +++ b/tools/skillgen/expected/graphify__skill-amp.md @@ -486,6 +486,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -507,6 +508,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-claw.md b/tools/skillgen/expected/graphify__skill-claw.md index d98865c..cef4d5c 100644 --- a/tools/skillgen/expected/graphify__skill-claw.md +++ b/tools/skillgen/expected/graphify__skill-claw.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-codex.md b/tools/skillgen/expected/graphify__skill-codex.md index 0c821a2..bd8a020 100644 --- a/tools/skillgen/expected/graphify__skill-codex.md +++ b/tools/skillgen/expected/graphify__skill-codex.md @@ -486,6 +486,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -507,6 +508,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-copilot.md b/tools/skillgen/expected/graphify__skill-copilot.md index d98865c..cef4d5c 100644 --- a/tools/skillgen/expected/graphify__skill-copilot.md +++ b/tools/skillgen/expected/graphify__skill-copilot.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-devin.md b/tools/skillgen/expected/graphify__skill-devin.md index e3e6d2d..f9be846 100644 --- a/tools/skillgen/expected/graphify__skill-devin.md +++ b/tools/skillgen/expected/graphify__skill-devin.md @@ -523,6 +523,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) @@ -543,6 +544,12 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') print('Report updated with community labels') " ``` @@ -1032,6 +1039,8 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, '.') Path('graphify-out/GRAPH_REPORT.md').write_text(report) +# No community_labels here - 'labels' are still placeholders at this point; +# Step 5 re-exports graph.json with the curated names (#2490). to_json(G, communities, 'graphify-out/graph.json') analysis = { diff --git a/tools/skillgen/expected/graphify__skill-droid.md b/tools/skillgen/expected/graphify__skill-droid.md index c3815d5..370e0d6 100644 --- a/tools/skillgen/expected/graphify__skill-droid.md +++ b/tools/skillgen/expected/graphify__skill-droid.md @@ -486,6 +486,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -507,6 +508,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-kilo.md b/tools/skillgen/expected/graphify__skill-kilo.md index dbb4658..f83cb06 100644 --- a/tools/skillgen/expected/graphify__skill-kilo.md +++ b/tools/skillgen/expected/graphify__skill-kilo.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-kiro.md b/tools/skillgen/expected/graphify__skill-kiro.md index d98865c..cef4d5c 100644 --- a/tools/skillgen/expected/graphify__skill-kiro.md +++ b/tools/skillgen/expected/graphify__skill-kiro.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-opencode.md b/tools/skillgen/expected/graphify__skill-opencode.md index cf5dae4..c121617 100644 --- a/tools/skillgen/expected/graphify__skill-opencode.md +++ b/tools/skillgen/expected/graphify__skill-opencode.md @@ -481,6 +481,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -502,6 +503,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-pi.md b/tools/skillgen/expected/graphify__skill-pi.md index d98865c..cef4d5c 100644 --- a/tools/skillgen/expected/graphify__skill-pi.md +++ b/tools/skillgen/expected/graphify__skill-pi.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-trae.md b/tools/skillgen/expected/graphify__skill-trae.md index b0cbeb1..98deab2 100644 --- a/tools/skillgen/expected/graphify__skill-trae.md +++ b/tools/skillgen/expected/graphify__skill-trae.md @@ -487,6 +487,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -508,6 +509,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-vscode.md b/tools/skillgen/expected/graphify__skill-vscode.md index 3e6bc6b..efb2401 100644 --- a/tools/skillgen/expected/graphify__skill-vscode.md +++ b/tools/skillgen/expected/graphify__skill-vscode.md @@ -485,6 +485,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -506,6 +507,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill-windows.md b/tools/skillgen/expected/graphify__skill-windows.md index 5743845..c48d874 100644 --- a/tools/skillgen/expected/graphify__skill-windows.md +++ b/tools/skillgen/expected/graphify__skill-windows.md @@ -511,6 +511,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -532,6 +533,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/expected/graphify__skill.md b/tools/skillgen/expected/graphify__skill.md index d98865c..cef4d5c 100644 --- a/tools/skillgen/expected/graphify__skill.md +++ b/tools/skillgen/expected/graphify__skill.md @@ -489,6 +489,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -510,6 +511,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/fragments/core/aider.md b/tools/skillgen/fragments/core/aider.md index 4f03ccb..4996beb 100644 --- a/tools/skillgen/fragments/core/aider.md +++ b/tools/skillgen/fragments/core/aider.md @@ -458,6 +458,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('.graphify_extract.json').read_text()) @@ -478,6 +479,12 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) Path('.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') print('Report updated with community labels') " ``` @@ -896,6 +903,8 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, '.') Path('graphify-out/GRAPH_REPORT.md').write_text(report) +# No community_labels here - 'labels' are still placeholders at this point; +# Step 5 re-exports graph.json with the curated names (#2490). to_json(G, communities, 'graphify-out/graph.json') analysis = { diff --git a/tools/skillgen/fragments/core/core.md b/tools/skillgen/fragments/core/core.md index e289107..e9a4c1b 100644 --- a/tools/skillgen/fragments/core/core.md +++ b/tools/skillgen/fragments/core/core.md @@ -424,6 +424,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) @@ -445,6 +446,13 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') print('Report updated with community labels') " ``` diff --git a/tools/skillgen/fragments/core/devin.md b/tools/skillgen/fragments/core/devin.md index e3e6d2d..f9be846 100644 --- a/tools/skillgen/fragments/core/devin.md +++ b/tools/skillgen/fragments/core/devin.md @@ -523,6 +523,7 @@ from graphify.build import build_from_json from graphify.cluster import score_all from graphify.analyze import god_nodes, surprising_connections, suggest_questions from graphify.report import generate +from graphify.export import to_json from pathlib import Path extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text()) @@ -543,6 +544,12 @@ questions = suggest_questions(G, communities, labels) report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) Path('graphify-out/GRAPH_REPORT.md').write_text(report) Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()})) +# Re-export so graph.json nodes carry the curated community_name (#2490). +# Same extraction as Step 4, so the #479 shrink-guard passes on node count; +# if it still refuses, surface the guard message - do not force past it. +wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels) +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (fewer nodes than the existing graph). Run a full rebuild to be safe.') print('Report updated with community labels') " ``` @@ -1032,6 +1039,8 @@ labels = {cid: 'Community ' + str(cid) for cid in communities} report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, '.') Path('graphify-out/GRAPH_REPORT.md').write_text(report) +# No community_labels here - 'labels' are still placeholders at this point; +# Step 5 re-exports graph.json with the curated names (#2490). to_json(G, communities, 'graphify-out/graph.json') analysis = { diff --git a/tools/skillgen/gen.py b/tools/skillgen/gen.py index 732215e..4141a02 100644 --- a/tools/skillgen/gen.py +++ b/tools/skillgen/gen.py @@ -954,6 +954,31 @@ def _is_semantic_cache_scope_fix_line(line: str) -> bool: ) or stripped.startswith("saved = save_semantic_cache(") +def _is_community_label_export_fix_line(line: str) -> bool: + """Whether a line is part of the Step-5 community_name re-export fix (#2490). + + Step 5 curated the community labels but never re-exported graph.json, so the + persisted nodes shipped without ``community_name`` (only the + ``.graphify_labels.json`` sidecar carried the names). Step 5 now imports + ``to_json`` and re-exports with ``community_labels=labels`` after the curated + dict exists, honoring (not forcing past) the #479 shrink-guard — the ``if not + wrote:`` / refused-to-shrink lines are already sanctioned by the #1392 + zero-node-guard predicate. The --cluster-only runbook keeps its label-less + export (its ``labels`` are placeholders at that point) and gains a comment + saying so. These are the added import, export call, and comment lines. + """ + stripped = line.strip() + return ( + "community_labels=labels" in line + or stripped == "from graphify.export import to_json" + or "curated community_name (#2490)" in line + or "shrink-guard passes on node count" in line + or "surface the guard message - do not force past it" in line + or "No community_labels here" in line + or "re-exports graph.json with the curated names (#2490)" in line + ) + + # Every line that may differ between a rendered monolith and its pristine v8 # baseline. Each predicate documents one sanctioned change-class; a blank line is # allowed because the multi-line fix blocks insert spacing. Anything else failing @@ -974,6 +999,7 @@ _SANCTIONED_MONOLITH_DIFFS = ( _is_obsidian_usage_comment_line, _is_uv_from_interpreter_fix_line, _is_semantic_cache_scope_fix_line, + _is_community_label_export_fix_line, )