From cb96bdaa0c367bec8d5c5aee5d7c9ebb727e9780 Mon Sep 17 00:00:00 2001 From: safishamsi Date: Wed, 15 Jul 2026 19:51:53 +0100 Subject: [PATCH] fix: preserve semantic layer, stamp hyperedges, PHP namespaces, ignore diagnostic (#1925 #1920 #1923 #1922) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - #1925: a missing manifest.json no longer degrades `extract --code-only` into a full scan that discards the committed semantic layer. An existing graph.json is a sufficient incremental baseline (detect_incremental treats an absent manifest as "all new / none deleted"), so out-of-scope doc/paper/ image nodes are preserved while genuinely deleted sources still evict. - #1920: _stamped_manifest_files now counts hyperedge output, so a doc whose only chunk output is a hyperedge is stamped instead of re-extracted forever. - #1923: new namespace/use-aware PHP resolver (mirrors the Java resolver, runs before the unique-name rewire) so App\Models\Page and an imported Filament\Pages\Page stay distinct — no more false inherits/imports edge. - #1922: detect() records ignored files/dirs in a new `ignored` diagnostic field (the nested-ignore scoping bug itself shipped in 0.9.16 / #1873). Regression tests added for each; full suite 3325 passed, 3 skipped. Co-Authored-By: Claude Opus 4.8 (1M context) --- CHANGELOG.md | 4 + graphify/cli.py | 27 ++- graphify/detect.py | 21 ++- graphify/extract.py | 17 ++ graphify/extractors/resolution.py | 264 ++++++++++++++++++++++++++++++ tests/test_detect.py | 27 +++ tests/test_extract_cli.py | 135 +++++++++++++++ tests/test_php_type_resolution.py | 158 ++++++++++++++++++ 8 files changed, 645 insertions(+), 8 deletions(-) create mode 100644 tests/test_php_type_resolution.py diff --git a/CHANGELOG.md b/CHANGELOG.md index bd6306e..84d865a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,10 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## 0.9.17 (unreleased) +- Fix: a missing `manifest.json` no longer degrades `graphify extract --code-only` into a full scan that discards the committed semantic layer (#1925). On a fresh clone (or when the manifest is deliberately untracked because its mtimes churn), the incremental gate required both `manifest.json` and `graph.json`; with only the graph present it fell to a full scan, and under `--code-only` that dropped every doc/paper/image node — silently replacing a curated graph with an AST-only skeleton. An existing `graph.json` is now a sufficient incremental baseline: `detect_incremental` already treats an absent manifest as "everything new / nothing deleted", so `build_merge` + `_stale_graph_sources` preserve files that are merely out of this run's scope while still evicting genuinely deleted sources. +- Fix: hyperedge-only documents are now stamped in the manifest instead of being re-extracted on every run (#1920). `_stamped_manifest_files` (#1897) decided a semantic doc "produced output" by inspecting only `nodes` and `edges`, never `hyperedges`, so a chunk whose only output for a doc was a hyperedge (3+ nodes sharing a concept) left that doc unstamped and perpetually re-queued. Stamping now counts hyperedge output too, mirroring the per-`source_file` keying the semantic cache already uses. +- Fix: the PHP extractor now disambiguates same-named classes across namespaces (#1923). A bare class reference (`extends Page`) collapsed onto the only internal class named `Page`, so `App\Models\Page` and an imported `Filament\Pages\Page` fused into one node, manufacturing a false `inherits`/`imports` edge and a bogus cross-community bridge. A new namespace/`use`-aware resolution pass (mirroring the Java resolver, running before the unique-name rewire) re-points supertype/import references to the real definition, or parks provably-external ones on a fully-qualified stub the bare-name rewire cannot collapse. Plain, non-namespaced PHP is unchanged. +- Fix: `detect()` now records files and directories dropped by a `.gitignore`/`.graphifyignore` rule in a new `ignored` diagnostic field (#1922). The nested-ignore scoping bug itself was fixed in 0.9.16 (#1873); this closes the remaining gap where an ignored path left no trace in any diagnostic, so an over-broad rule looked like a clean scan. Entries are per-directory where a subtree is pruned, keeping the list bounded. - Fix: `_semantic_id_remap` is now idempotent, so incremental rebuilds stop churning (#1917). When a file's canonical stem contained its own legacy stem as a prefix (parent dir name equals the file stem, e.g. `.claude/CLAUDE.md`, `docs/docs.md`), an already-migrated semantic node id re-matched the legacy branch and gained another stem segment on every build (`claude_x` -> `claude_claude_x` -> ...). Because `_origin` is persisted, every `graphify update` re-fed nodes through the remap, so the ids grew unboundedly and the `same_topology`/`same_graph`/`no_change` short-circuits never fired — rewriting `graph.json` and re-running clustering on every zero-delta update. The remap now skips an id that already carries its canonical stem (mirroring the `graph_has_legacy_ids` check), while a genuine one-time legacy migration still applies. (An already-corrupted graph needs one `graphify extract --force` to reset the grown ids.) - Perf: `graphify query` now scores the graph once per query instead of T+1 times for a T-term query (#1889 / #1918, thanks @Sirhan1). The per-term-guarantee (#1445) previously re-scored the whole graph once per token; `_score_query` now computes the combined ranking and each token's singleton winner in a single traversal, feeding `_pick_seeds` via `best_seed_by_term`. Behavior is preserved (verified byte-identical against the old per-term scoring across a differential fuzz); ~1.3-1.4x faster and independent of query length. diff --git a/graphify/cli.py b/graphify/cli.py index f2ae1da..d817401 100644 --- a/graphify/cli.py +++ b/graphify/cli.py @@ -65,11 +65,18 @@ def _stamped_manifest_files( empty so detect_incremental re-queues them (#933). Both sides of the membership test are resolved against the scan ``root`` - before comparing (#1897): node/edge ``source_file`` values are + before comparing (#1897): node/edge/hyperedge ``source_file`` values are root-relative on a fresh extraction while ``files_by_type`` entries are absolute (from detect()), so a raw string comparison never matched and every freshly-extracted semantic doc was dropped from the manifest. Mirrors the #1890 path normalization in graphify.llm. + + Hyperedges are counted as output (#1920): a chunk whose only result for a + document is a hyperedge (3+ nodes sharing a concept) is valid output that + the semantic cache persists per-``source_file`` — omitting it here left the + doc unstamped, so detect_incremental re-queued it on every run. The stamping + condition mirrors the cache-write keying (a hyperedge carries its own + ``source_file``); do not derive it from member nodes. """ root = Path(root) @@ -83,7 +90,7 @@ def _stamped_manifest_files( return p sem_extracted: set[Path] = set() - for coll in ("nodes", "edges"): + for coll in ("nodes", "edges", "hyperedges"): for item in sem_result.get(coll, []): sf = item.get("source_file", "") if sf: @@ -2281,12 +2288,26 @@ def dispatch_command(cmd: str) -> None: ) manifest_path = graphify_out / "manifest.json" existing_graph_path = graphify_out / "graph.json" - incremental_mode = manifest_path.exists() and existing_graph_path.exists() if has_path else False + # #1925: a missing manifest.json must not degrade to a full scan that + # discards the existing graph's semantic layer. An existing graph.json + # is a sufficient incremental baseline: detect_incremental treats an + # absent manifest as "everything is new" (re-extract all, nothing + # deleted), and build_merge + _stale_graph_sources reconcile replaced + # and genuinely-deleted sources against the current corpus, so doc/ + # paper/image nodes survive a --code-only rebuild instead of being + # dropped with the rest of the committed graph. + incremental_mode = existing_graph_path.exists() if has_path else False # --force: full scan, not the manifest-gated incremental diff — a warm # unchanged tree would otherwise dispatch zero files (#1894). incremental_mode = incremental_mode and not force if force: print("[graphify extract] --force: full re-scan, semantic cache reads skipped") + elif incremental_mode and not manifest_path.exists(): + print( + "[graphify extract] manifest.json missing; using existing " + "graph.json as the incremental baseline (all files re-checked; " + "nodes for files outside this run's scope are preserved)" + ) if not has_path: code_files = [] diff --git a/graphify/detect.py b/graphify/detect.py index 1c1c27c..ce1aad8 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -1151,6 +1151,11 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: skipped_sensitive: list[str] = [] unclassified: list[str] = [] + # Files/dirs dropped by a .gitignore/.graphifyignore rule. Recorded so an + # over-broad ignore (or a legitimately-ignored subtree) is visible instead + # of silently vanishing from the graph (#1922). Directory-level entries keep + # this bounded — a pruned `data/` is one entry, not one per contained file. + ignored: list[str] = [] ignore_patterns = _load_graphifyignore(root) ignore_cache: dict[Path, bool] = {} # shared across all _is_ignored calls in this scan # CLI --exclude patterns are anchored at the scan root and appended last @@ -1222,11 +1227,15 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: # any `!` rule existed — e.g. a single `!docs/**` made the walk descend # bin/, obj/, wwwroot/, generated/, … : a pathological slowdown on large # repos for no correctness gain. - dirnames[:] = [ - d for d in dirnames - if not _is_noise_dir(d, dp) - and not _is_ignored(dp / d, root, ignore_patterns, _cache=ignore_cache) - ] + kept_dirs: list[str] = [] + for d in dirnames: + if _is_noise_dir(d, dp): + continue + if _is_ignored(dp / d, root, ignore_patterns, _cache=ignore_cache): + ignored.append(str(dp / d) + os.sep) + continue + kept_dirs.append(d) + dirnames[:] = kept_dirs if follow_symlinks: safe_dirs: list[str] = [] for d in dirnames: @@ -1256,6 +1265,7 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: if str(p).startswith(str(converted_dir)): continue if not in_memory and _is_ignored(p, root, ignore_patterns, _cache=ignore_cache): + ignored.append(str(p)) continue if not _resolves_under_root(p, root): skipped_sensitive.append(str(p) + " [symlink target outside scan root]") @@ -1338,6 +1348,7 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace: "skipped_sensitive": skipped_sensitive, "unclassified": sorted(unclassified), "walk_errors": walk_errors, + "ignored": sorted(ignored), "graphifyignore_patterns": len(ignore_patterns), "scan_root": str(root.resolve()), } diff --git a/graphify/extract.py b/graphify/extract.py index 12a4f10..2ff5411 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -113,6 +113,7 @@ from graphify.extractors.resolution import ( # noqa: E402,F401 _resolve_cross_file_java_imports, _resolve_export_target, _resolve_java_type_references, + _resolve_php_type_references, _resolve_js_import_path, _resolve_js_import_target, _resolve_js_module_path, @@ -4581,6 +4582,22 @@ def extract( _merge_swift_extensions(per_file, all_nodes, all_edges) _disambiguate_colliding_node_ids(all_nodes, all_edges, all_raw_calls, root) _canonicalize_csharp_namespace_nodes(all_nodes, all_edges) + # PHP namespace/use disambiguation must run BEFORE the unique-stub rewire: + # the false merge (#1923) happens inside the rewire when a bare-name stub + # matches a unique internal class from a different namespace. + _php_exts = {".php", ".phtml", ".php3", ".php4", ".php5", ".php7", ".phps"} + _php_sel = [ + (r, p) for r, p in zip(per_file, paths) + if p.suffix.lower() in _php_exts and not p.name.lower().endswith(".blade.php") + ] + if _php_sel: + try: + _resolve_php_type_references( + [r for r, _ in _php_sel], [p for _, p in _php_sel], all_nodes, all_edges + ) + except Exception as exc: + import logging + logging.getLogger(__name__).warning("PHP type-reference resolution failed, skipping: %s", exc) _rewire_unique_stub_nodes(all_nodes, all_edges) # Add cross-file class-level edges (Python only - uses Python parser internally) diff --git a/graphify/extractors/resolution.py b/graphify/extractors/resolution.py index 9736ccc..a88abf2 100644 --- a/graphify/extractors/resolution.py +++ b/graphify/extractors/resolution.py @@ -2213,6 +2213,270 @@ def _resolve_java_type_references( if node.get("id") not in repointed_from or node.get("id") in still_referenced ] + +_PHP_SUPERTYPE_RELATIONS = ("inherits", "implements", "mixes_in") +_PHP_REPOINT_RELATIONS = frozenset({"inherits", "implements", "mixes_in", "imports", "references"}) + + +def _php_fqn_from_raw(raw: str, ns: str, uses: dict[str, str]) -> str: + """Resolve a raw (possibly qualified) PHP class reference to an FQN. + + PHP name-resolution for class names: + \\A\\B -> absolute: A\\B + A\\B -> first segment through the `use` map (group-prefix semantics), + else relative to the current namespace + B -> `use` map, else current namespace (class names do NOT fall + back to the global namespace) + """ + raw = raw.strip() + if raw.startswith("\\"): + return raw[1:] + if "\\" in raw: + first, rest = raw.split("\\", 1) + mapped = uses.get(first.lower()) + if mapped: + return f"{mapped}\\{rest}" + return f"{ns}\\{raw}" if ns else raw + mapped = uses.get(raw.lower()) + if mapped: + return mapped + return f"{ns}\\{raw}" if ns else raw + + +def _resolve_php_type_references( + per_file: list[dict], + paths: list[Path], + all_nodes: list[dict], + all_edges: list[dict], +) -> None: + """Disambiguate PHP inherits/implements/mixes_in/imports/references targets + using each file's ``namespace`` declaration and ``use`` imports (#1923). + + Mirrors ``_resolve_java_type_references`` (a re-parse pass), but MUST run + BEFORE ``_rewire_unique_stub_nodes``: the false edge is manufactured by the + rewire itself — a bare ``Page`` stub collapses onto the only internal class + labeled ``Page`` even though the referencing file ``use``d a different + namespace (``Filament\\Pages\\Page`` vs ``App\\Models\\Page``). References + proven external by a ``use`` FQN or a qualified name are re-pointed to an + FQN-labeled sourceless stub, which the bare-label rewire cannot collapse. + References with no namespace facts are left untouched so the unique-label + rewire keeps handling plain (non-namespaced) PHP as before. + """ + try: + import tree_sitter_php as tsphp + from tree_sitter import Language, Parser + except ImportError: + return + + lang_fn = getattr(tsphp, "language_php", None) or getattr(tsphp, "language", None) + if lang_fn is None: + return + language = Language(lang_fn()) + parser = Parser(language) + + ns_by_file: dict[str, str] = {} + uses_by_file: dict[str, dict[str, str]] = {} # lower alias -> FQN + raw_by_file: dict[str, dict[tuple[str, str], str | None]] = {} # (relation, lower bare) -> raw | None(ambiguous) + + for path, result in zip(paths, per_file): + srcs = {n.get("source_file") for n in result.get("nodes", []) if n.get("source_file")} + if not srcs: + continue + try: + source = path.read_bytes() + tree = parser.parse(source) + except Exception: + continue + + namespaces: list[str] = [] + uses: dict[str, str] = {} + raws: dict[tuple[str, str], str | None] = {} + + def _record_raw(relation: str, raw: str) -> None: + bare = raw.rsplit("\\", 1)[-1].strip().lower() + if not bare: + return + key = (relation, bare) + if key in raws and raws[key] != raw: + raws[key] = None # e.g. `implements A\I, B\I` — never guess + else: + raws.setdefault(key, raw) + + def _record_use_clause(clause, prefix: str) -> None: + target = None + alias = None + saw_as = False + for c in clause.children: + if c.type in ("function", "const"): + return # not a class import + if c.type == "as": + saw_as = True + elif c.type in ("qualified_name", "name"): + if saw_as: + alias = _read_text(c, source) + elif target is None: + target = _read_text(c, source) + if not target: + return + fqn = (f"{prefix}\\{target}" if prefix else target).lstrip("\\") + key = (alias or fqn.rsplit("\\", 1)[-1]).strip().lower() + if key: + uses.setdefault(key, fqn) + + def walk(n) -> None: + t = n.type + if t == "namespace_definition": + for c in n.children: + if c.type == "namespace_name": + namespaces.append(_read_text(c, source)) + break + elif t == "namespace_use_declaration": + prefix = "" + group = None + for c in n.children: + if c.type == "namespace_name": + prefix = _read_text(c, source) # group-use prefix + elif c.type == "namespace_use_group": + group = c + elif c.type == "namespace_use_clause": + _record_use_clause(c, "") + if group is not None: + for c in group.children: + if c.type == "namespace_use_clause": + _record_use_clause(c, prefix) + return + elif t == "class_declaration": + for child in n.children: + if child.type == "base_clause": + for sub in child.children: + if sub.type in ("name", "qualified_name"): + _record_raw("inherits", _read_text(sub, source)) + elif child.type == "class_interface_clause": + for sub in child.children: + if sub.type in ("name", "qualified_name"): + _record_raw("implements", _read_text(sub, source)) + elif child.type == "declaration_list": + for member in child.children: + if member.type != "use_declaration": + continue + for sub in member.children: + if sub.type in ("name", "qualified_name"): + _record_raw("mixes_in", _read_text(sub, source)) + for child in n.children: + walk(child) + + walk(tree.root_node) + if len(set(namespaces)) > 1: + continue # multi-namespace file (PSR-1 violation): keep legacy behavior + ns = namespaces[0] if namespaces else "" + for s in srcs: + ns_by_file[s] = ns + uses_by_file[s] = uses + raw_by_file[s] = raws + + if not ns_by_file: + return + + # lower FQN -> definition node id (PHP class names are case-insensitive). + fqn_to_id: dict[str, str] = {} + for node in all_nodes: + label = node.get("label", "") + src = node.get("source_file", "") + nid = node.get("id", "") + if not (label and src and nid) or src not in ns_by_file: + continue + if label.endswith(")") or "." in label: # methods / file nodes + continue + ns = ns_by_file[src] + fqn = f"{ns}\\{label}" if ns else label + fqn_to_id.setdefault(fqn.lower(), nid) + + node_ids = {n.get("id") for n in all_nodes if n.get("id")} + stub_label: dict[str, str] = { + n["id"]: n.get("label", "") + for n in all_nodes + if n.get("id") and not n.get("source_file") and n.get("label") + } + + external_stub_ids: dict[str, str] = {} + new_nodes: list[dict] = [] + + def _external_stub(fqn: str) -> str: + key = fqn.lower() + nid = external_stub_ids.get(key) + if nid: + return nid + nid = _make_id(fqn) + if nid not in node_ids: + new_nodes.append({ + "id": nid, + "label": fqn, + "file_type": "code", + "source_file": "", + "source_location": "", + }) + node_ids.add(nid) + external_stub_ids[key] = nid + return nid + + repointed_from: set[str] = set() + for edge in all_edges: + relation = edge.get("relation") + if relation not in _PHP_REPOINT_RELATIONS: + continue + ref_file = edge.get("source_file", "") + if ref_file not in ns_by_file: + continue + tgt = edge.get("target") + label = stub_label.get(tgt) + if not label: + continue + bare = label.strip().lower() + ns = ns_by_file[ref_file] + uses = uses_by_file.get(ref_file, {}) + + raw = None + if relation in _PHP_SUPERTYPE_RELATIONS: + raw = raw_by_file.get(ref_file, {}).get((relation, bare)) + + explicit = False + if raw and "\\" in raw: + fqn = _php_fqn_from_raw(raw, ns, uses) + explicit = True + elif bare in uses: + fqn = uses[bare] + explicit = True + elif ns: + fqn = f"{ns}\\{label}" + else: + continue # no namespace facts: legacy unique-label rewire applies + + resolved = fqn_to_id.get(fqn.lower()) + if resolved and resolved != tgt: + edge["target"] = resolved + repointed_from.add(tgt) + elif explicit and resolved is None: + # Proven external: park the edge on an FQN-labeled stub the + # bare-name rewire cannot collapse (this is the #1923 fix). + edge["target"] = _external_stub(fqn) + repointed_from.add(tgt) + # non-explicit miss: leave the bare stub for the legacy rewire + + if new_nodes: + all_nodes.extend(new_nodes) + if not repointed_from: + return + + still_referenced: set[str] = set() + for edge in all_edges: + still_referenced.add(edge.get("source")) + still_referenced.add(edge.get("target")) + all_nodes[:] = [ + n for n in all_nodes + if n.get("id") not in repointed_from or n.get("id") in still_referenced + ] + + _pascal_unit_cache: dict[str, dict[str, str]] = {} _pascal_class_stem_cache: dict[str, dict[str, str]] = {} # root_key → {stem_lower: _file_stem} diff --git a/tests/test_detect.py b/tests/test_detect.py index a783eaf..498a9aa 100644 --- a/tests/test_detect.py +++ b/tests/test_detect.py @@ -1853,6 +1853,33 @@ def test_nested_gitignore_patterns_still_apply_inside_their_dir(tmp_path): assert result["total_files"] == 2 # main.py + sub/keep.py; sub/noise.log ignored +def test_nested_gitignore_does_not_govern_sibling_project(tmp_path): + """A nested .gitignore ('data/') in one project must not drop a sibling + project's data/ files, and the drop must be recorded in the `ignored` + diagnostic field rather than silently vanishing (#1922).""" + (tmp_path / "run.py").write_text("x = 1") + pa = tmp_path / "project_a" / "data" + pa.mkdir(parents=True) + (pa / "loader.py").write_text("def load(): pass") + pb = tmp_path / "project_b" + (pb / "data").mkdir(parents=True) + (pb / ".gitignore").write_text("data/\n") + (pb / "data" / "dump.csv").write_text("a,b\n1,2\n") + + result = detect(tmp_path) + + all_paths = [f for v in result["files"].values() for f in v] + assert any( + f.endswith(os.path.join("project_a", "data", "loader.py")) for f in all_paths + ), "sibling project_a/data/loader.py must survive project_b's nested ignore" + assert not any(f.endswith("dump.csv") for f in all_paths) + # The legitimately-ignored subtree is recorded, not silently dropped. + assert any( + e.rstrip(os.sep).endswith(os.path.join("project_b", "data")) + for e in result["ignored"] + ), f"ignored subtree should be recorded in detect()['ignored']: {result['ignored']}" + + # --------------------------------------------------------------------------- # #1908: manifest must not retain scan-excluded files as permanent # "deleted" entries. Full-scan saves prune excluded-but-alive rows; subset diff --git a/tests/test_extract_cli.py b/tests/test_extract_cli.py index 6ef1b2c..87a65e4 100644 --- a/tests/test_extract_cli.py +++ b/tests/test_extract_cli.py @@ -222,6 +222,74 @@ def test_stamped_manifest_files_normalizes_both_sides(tmp_path): assert out["document"] == [str(fresh_doc), str(cached_doc)] +def test_stamped_manifest_files_counts_hyperedge_only_docs(tmp_path): + """#1920: a doc whose only chunk output is a hyperedge (3+ nodes sharing a + concept) is valid output — the semantic cache persists it per source_file — + so it must be stamped. Before the fix the stamping loop only inspected + ``nodes``/``edges``, leaving such a doc unstamped and re-queued forever.""" + from graphify.cli import _stamped_manifest_files + + hyper_doc = tmp_path / "hyper.md"; hyper_doc.write_text("# hyper") + omitted_doc = tmp_path / "omitted.md"; omitted_doc.write_text("# omitted") + + files_by_type = {"document": [str(hyper_doc), str(omitted_doc)]} + sem_result = { + "nodes": [], + "edges": [], + "hyperedges": [ + {"id": "h1", "label": "L", "nodes": ["a", "b", "c"], + "relation": "participate_in", "source_file": "hyper.md"}, + ], + } + + out = _stamped_manifest_files(files_by_type, sem_result, tmp_path) + assert str(hyper_doc) in out["document"], ( + "a hyperedge-only doc must be stamped (#1920)" + ) + # A doc with no output at all still stays unstamped (#933). + assert str(omitted_doc) not in out["document"] + + +def test_manifest_stamps_hyperedge_only_docs(monkeypatch, tmp_path): + """#1920 end-to-end: a fresh extraction whose only output for a doc is a + hyperedge stamps that doc's semantic_hash, so it is not re-dispatched.""" + import json + + corpus = _make_corpus(tmp_path) # main.go + README.md + out_dir = tmp_path / "out" + monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-test-fake-key") + + def _hyperedge_only(paths, **kwargs): + on_chunk = kwargs.get("on_chunk_done") + if on_chunk: + on_chunk(0, 1, {"nodes": [], "edges": [], "hyperedges": []}) + return { + "nodes": [], + "edges": [], + "hyperedges": [{"id": "h1", "label": "Shared", "nodes": ["a", "b", "c"], + "relation": "participate_in", "source_file": "README.md"}], + "input_tokens": 10, + "output_tokens": 5, + } + + monkeypatch.setattr("graphify.llm.extract_corpus_parallel", _hyperedge_only) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + monkeypatch.setattr( + mainmod.sys, "argv", + ["graphify", "extract", str(corpus), "--backend", "claude", + "--no-cluster", "--out", str(out_dir)], + ) + try: + mainmod.main() + except SystemExit as exc: + assert exc.code in (None, 0), f"unexpected exit code {exc.code}" + + manifest = json.loads((out_dir / "graphify-out" / "manifest.json").read_text()) + assert manifest.get("README.md", {}).get("semantic_hash"), ( + f"hyperedge-only doc must be stamped (#1920): {sorted(manifest)}" + ) + + # --- #1894: --force and deep-mode dispatch over a warm cache ----------------- def _recording_extractor(calls): @@ -429,6 +497,73 @@ def test_extract_codeonly_succeeds_without_api_key(monkeypatch, tmp_path): assert len(json.loads(graph.read_text()).get("nodes", [])) > 0 +def test_missing_manifest_code_only_preserves_semantic_layer(monkeypatch, tmp_path): + """#1925: `graphify extract --code-only` with a MISSING manifest.json must + not degrade to a full scan that discards the committed semantic layer. An + existing graph.json is a sufficient incremental baseline, so doc/paper/image + nodes (excluded by --code-only, not deleted) are preserved; a genuinely + deleted source is still evicted (#1909 semantics retained).""" + import json + + corpus = tmp_path / "proj"; corpus.mkdir() + (corpus / "keep.py").write_text("def keep():\n return 1\n") + (corpus / "README.md").write_text("# Notes\nCurated docs.\n") + out_dir = tmp_path / "out" + graphify_out = out_dir / "graphify-out" + _clear_backend_keys(monkeypatch) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + + def _sem_doc_count(g): + return sum(1 for n in g["nodes"] if n.get("source_file") == "README.md") + + # 1) seed a code-only graph + _run_extract(monkeypatch, ["graphify", "extract", str(corpus), + "--code-only", "--out", str(out_dir)]) + graph_path = graphify_out / "graph.json" + graph = json.loads(graph_path.read_text()) + + # 2) inject a committed semantic layer for README.md (nodes + edge + hyperedge) + graph["nodes"].append({"id": "doc_readme_a", "label": "Concept A", + "source_file": "README.md", "file_type": "document"}) + graph["nodes"].append({"id": "doc_readme_b", "label": "Concept B", + "source_file": "README.md", "file_type": "document"}) + graph.setdefault("edges", []).append( + {"source": "doc_readme_a", "target": "doc_readme_b", + "relation": "relates_to", "source_file": "README.md"}) + graph.setdefault("hyperedges", []).append( + {"id": "h1", "label": "Shared", "nodes": ["doc_readme_a", "doc_readme_b"], + "relation": "participate_in", "source_file": "README.md"}) + graph_path.write_text(json.dumps(graph)) + (graphify_out / ".graphify_semantic_marker").write_text( + json.dumps({"output_tokens": 1})) + + # 3) manifest goes missing (fresh clone / deliberately untracked) + (graphify_out / "manifest.json").unlink() + + # 4) re-run the SAME code-only extract + _run_extract(monkeypatch, ["graphify", "extract", str(corpus), + "--code-only", "--out", str(out_dir)]) + after = json.loads(graph_path.read_text()) + assert _sem_doc_count(after) >= 2, ( + "committed semantic doc nodes must survive a missing-manifest " + f"--code-only rebuild (#1925); got {_sem_doc_count(after)}" + ) + assert any(h.get("id") == "h1" for h in after.get("hyperedges", [])), ( + "committed hyperedge must survive the rebuild" + ) + assert any("keep" in n["id"] for n in after["nodes"]), "code nodes intact" + + # 5) a genuine deletion still evicts the doc's semantic nodes + (corpus / "README.md").unlink() + (graphify_out / "manifest.json").unlink(missing_ok=True) + _run_extract(monkeypatch, ["graphify", "extract", str(corpus), + "--code-only", "--out", str(out_dir)]) + gone = json.loads(graph_path.read_text()) + assert _sem_doc_count(gone) == 0, ( + "a genuinely deleted doc must still be evicted (#1909 semantics preserved)" + ) + + def test_extract_out_keeps_project_root_clean(monkeypatch, tmp_path): """`extract --out DIR` routes every artifact to DIR/graphify-out/ and the scanned project must not grow a graphify-out/ (or anything else) beside diff --git a/tests/test_php_type_resolution.py b/tests/test_php_type_resolution.py new file mode 100644 index 0000000..0dff626 --- /dev/null +++ b/tests/test_php_type_resolution.py @@ -0,0 +1,158 @@ +from __future__ import annotations + +from pathlib import Path + +from graphify.extract import extract + + +def _write(path: Path, text: str) -> Path: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text, encoding="utf-8") + return path + + +def _node_by_id(result: dict, nid: str) -> dict | None: + return next((n for n in result["nodes"] if n.get("id") == nid), None) + + +def _class_defs(result: dict, label: str) -> list[dict]: + return [ + n for n in result["nodes"] + if n.get("label") == label and n.get("source_file") + ] + + +def test_php_external_namespaced_base_does_not_collapse_onto_internal_class(tmp_path: Path): + # #1923: `App\Models\Page` (internal) and `Filament\Pages\Page` (external, + # via `use`) share the simple name `Page`. The bare-name rewire must NOT + # collapse the external supertype reference onto the only internal `Page`. + model = _write( + tmp_path / "app/Models/Page.php", + "