refactor(extract): migrate 18 bespoke language extractors to extractors/ package

Continues the graphify/extract.py -> graphify/extractors/ split (MIGRATION.md,
upstream #1212), which already moved blade/zig/elixir/razor. Moves the
independent bespoke extractors whose closures are fully private (no shared
_extract_generic core, no shared mutable caches), verbatim, following the
documented invariants:

  dart, rust, go, powershell (+psd1 manifest), fortran, sql,
  dm (dm/dmm/dmi/dmf), bash, apex, terraform, sln,
  pascal_forms (delphi .dfm + lazarus .lfm), json_config

Each language's private helper funcs and constants move with it; only the
`extract_<lang>` entry points (plus fortran's _cpp_preprocess, which has a
direct unit test) are re-exported from extract.py's facade block, so every
existing importer (__main__.py, watch.py, tests) is unchanged and object
identity is preserved. Registry (extractors/__init__.py) grows 4 -> 22 langs.

Verified: AST closure-privacy analysis (no symbol referenced from outside its
moved set except via the facade); byte-identity of every moved span;
extract._DISPATCH still resolves every extension; ruff clean; skillgen --check
OK. extract.py drops 17,054 -> 13,121 LOC. Full suite unchanged: 3036 passed,
29 skipped (excluding env-only openai tests).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
tpateeq
2026-07-09 01:11:37 +05:30
co-authored by Claude Opus 4.8
parent cc2d3c13a6
commit 9fc30e680d
16 changed files with 4077 additions and 3947 deletions
+13 -3946
View File
File diff suppressed because it is too large Load Diff
+15 -1
View File
@@ -12,7 +12,21 @@ written so an AI agent can execute it in a single session.
| zig | yes |
| elixir | yes |
| razor | yes |
| (40 more in extract.py) | no |
| dart | yes |
| rust | yes |
| go | yes |
| powershell (ps1 + psd1 manifest) | yes |
| fortran | yes |
| sql | yes |
| dm (dm/dmm/dmi/dmf) | yes |
| bash | yes |
| apex | yes |
| terraform | yes |
| sln | yes |
| pascal_forms (dfm + lfm) | yes |
| json_config | yes |
| (config-driven core: python, js, java, c, cpp, csharp, kotlin, scala, php, lua, swift, groovy, vue, svelte, astro, xaml, groovy) | no — shared _extract_generic core, move as one batch |
| (other bespoke: julia, verilog, markdown, objc, csproj, slnx, lazarus_package, pascal) | no |
Note: config-driven extractors (python, js, java, c, cpp, ruby, csharp,
kotlin, scala, php, lua, swift, groovy) depend on the shared
+31
View File
@@ -10,14 +10,45 @@ from __future__ import annotations
from pathlib import Path
from typing import Callable
from graphify.extractors.apex import extract_apex
from graphify.extractors.bash import extract_bash
from graphify.extractors.blade import extract_blade
from graphify.extractors.dart import extract_dart
from graphify.extractors.dm import extract_dm, extract_dmf, extract_dmi, extract_dmm
from graphify.extractors.elixir import extract_elixir
from graphify.extractors.fortran import extract_fortran
from graphify.extractors.go import extract_go
from graphify.extractors.json_config import extract_json
from graphify.extractors.pascal_forms import extract_delphi_form, extract_lazarus_form
from graphify.extractors.powershell import extract_powershell, extract_powershell_manifest
from graphify.extractors.razor import extract_razor
from graphify.extractors.rust import extract_rust
from graphify.extractors.sln import extract_sln
from graphify.extractors.sql import extract_sql
from graphify.extractors.terraform import extract_terraform
from graphify.extractors.zig import extract_zig
LANGUAGE_EXTRACTORS: dict[str, Callable[[Path], dict]] = {
"apex": extract_apex,
"bash": extract_bash,
"blade": extract_blade,
"dart": extract_dart,
"delphi_form": extract_delphi_form,
"dm": extract_dm,
"dmf": extract_dmf,
"dmi": extract_dmi,
"dmm": extract_dmm,
"elixir": extract_elixir,
"fortran": extract_fortran,
"go": extract_go,
"json": extract_json,
"lazarus_form": extract_lazarus_form,
"powershell": extract_powershell,
"powershell_manifest": extract_powershell_manifest,
"razor": extract_razor,
"rust": extract_rust,
"sln": extract_sln,
"sql": extract_sql,
"terraform": extract_terraform,
"zig": extract_zig,
}
+215
View File
@@ -0,0 +1,215 @@
"""Apex extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
from pathlib import Path
from graphify.extractors.base import _file_stem, _make_id
def extract_apex(path: Path) -> dict:
"""Extract classes, interfaces, enums, methods, and Salesforce constructs from
Apex .cls and .trigger files using regex (no tree-sitter grammar on PyPI)."""
import re as _re
try:
source = path.read_text(encoding="utf-8", errors="replace")
except OSError:
return {"nodes": [], "edges": []}
str_path = str(path)
stem = _file_stem(path)
file_nid = _make_id(str_path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
def add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({
"id": nid,
"label": label,
"file_type": "code",
"source_file": str_path,
"source_location": f"L{line}",
})
def add_edge(src: str, tgt: str, relation: str, line: int,
confidence: str = "EXTRACTED") -> None:
edges.append({
"source": src,
"target": tgt,
"relation": relation,
"confidence": confidence,
"source_file": str_path,
"source_location": f"L{line}",
"weight": 1.0,
})
add_node(file_nid, path.name, 1)
lines = source.splitlines()
_ACCESS = r"(?:public|private|protected|global|webService)?"
_SHARING = r"(?:\s+(?:with|without|inherited)\s+sharing)?"
_MOD = r"(?:\s+(?:abstract|virtual|override|static|final|transient|testMethod))?"
_ANNOTATION = r"(?:\s*@\w+(?:\s*\([^)]*\))?\s*)*"
cls_re = _re.compile(
rf"^{_ANNOTATION}\s*{_ACCESS}{_SHARING}{_MOD}\s*class\s+(\w+)"
rf"(?:\s+extends\s+(\w+))?(?:\s+implements\s+([\w,\s]+))?\s*\{{?",
_re.IGNORECASE,
)
iface_re = _re.compile(
rf"^{_ANNOTATION}\s*{_ACCESS}{_SHARING}{_MOD}\s*interface\s+(\w+)"
rf"(?:\s+extends\s+([\w,\s]+))?\s*\{{?",
_re.IGNORECASE,
)
enum_re = _re.compile(
rf"^{_ANNOTATION}\s*{_ACCESS}{_SHARING}{_MOD}\s*enum\s+(\w+)\s*\{{?",
_re.IGNORECASE,
)
trigger_re = _re.compile(
r"^\s*trigger\s+(\w+)\s+on\s+(\w+)\s*\(",
_re.IGNORECASE,
)
method_re = _re.compile(
rf"^{_ANNOTATION}\s*{_ACCESS}{_MOD}\s*(?:static\s+)?[\w<>\[\]]+\s+(\w+)\s*\([^)]*\)\s*(?:throws\s+\w+\s*)?\{{?",
_re.IGNORECASE,
)
annotation_re = _re.compile(r"@(\w+)", _re.IGNORECASE)
soql_re = _re.compile(r"\[\s*SELECT\b[^\]]+FROM\s+(\w+)", _re.IGNORECASE)
dml_re = _re.compile(r"\b(insert|update|delete|upsert|merge|undelete)\s+\w", _re.IGNORECASE)
_CONTROL_FLOW = frozenset({
"if", "else", "for", "while", "do", "switch", "try", "catch",
"finally", "return", "throw", "new", "void", "null",
"true", "false", "this", "super", "class", "interface", "enum",
"trigger", "on",
})
current_class_nid: str | None = None
pending_annotations: list[str] = []
for lineno, line_text in enumerate(lines, start=1):
stripped = line_text.strip()
if stripped.startswith("@"):
for m in annotation_re.finditer(stripped):
pending_annotations.append(m.group(1).lower())
continue
tm = trigger_re.match(stripped)
if tm:
trig_name, sobject = tm.group(1), tm.group(2)
trig_nid = _make_id(stem, trig_name)
add_node(trig_nid, trig_name, lineno)
add_edge(file_nid, trig_nid, "contains", lineno)
sob_nid = _make_id(sobject)
if sob_nid not in seen_ids:
add_node(sob_nid, sobject, lineno)
add_edge(trig_nid, sob_nid, "uses", lineno, confidence="INFERRED")
current_class_nid = trig_nid
pending_annotations = []
continue
cm = cls_re.match(stripped)
if cm:
class_name = cm.group(1)
if class_name.lower() in _CONTROL_FLOW:
pending_annotations = []
continue
class_nid = _make_id(stem, class_name)
add_node(class_nid, class_name, lineno)
add_edge(file_nid, class_nid, "contains", lineno)
if cm.group(2):
base = cm.group(2).strip()
base_nid = _make_id(stem, base)
if base_nid not in seen_ids:
base_nid = _make_id(base)
if base_nid not in seen_ids:
add_node(base_nid, base, lineno)
add_edge(class_nid, base_nid, "extends", lineno, confidence="INFERRED")
if cm.group(3):
for iface in cm.group(3).split(","):
iface = iface.strip()
if iface:
iface_nid = _make_id(stem, iface)
if iface_nid not in seen_ids:
iface_nid = _make_id(iface)
if iface_nid not in seen_ids:
add_node(iface_nid, iface, lineno)
add_edge(class_nid, iface_nid, "implements", lineno, confidence="INFERRED")
current_class_nid = class_nid
pending_annotations = []
continue
im = iface_re.match(stripped)
if im:
iface_name = im.group(1)
if iface_name.lower() in _CONTROL_FLOW:
pending_annotations = []
continue
iface_nid = _make_id(stem, iface_name)
add_node(iface_nid, iface_name, lineno)
add_edge(file_nid if current_class_nid is None else current_class_nid,
iface_nid, "contains", lineno)
if im.group(2):
for parent in im.group(2).split(","):
parent = parent.strip()
if parent:
parent_nid = _make_id(stem, parent)
if parent_nid not in seen_ids:
parent_nid = _make_id(parent)
if parent_nid not in seen_ids:
add_node(parent_nid, parent, lineno)
add_edge(iface_nid, parent_nid, "extends", lineno, confidence="INFERRED")
pending_annotations = []
continue
em = enum_re.match(stripped)
if em:
enum_name = em.group(1)
if enum_name.lower() in _CONTROL_FLOW:
pending_annotations = []
continue
enum_nid = _make_id(stem, enum_name)
add_node(enum_nid, enum_name, lineno)
add_edge(file_nid if current_class_nid is None else current_class_nid,
enum_nid, "contains", lineno)
pending_annotations = []
continue
if current_class_nid is not None:
mm = method_re.match(stripped)
if mm:
method_name = mm.group(1)
if method_name.lower() not in _CONTROL_FLOW:
method_nid = _make_id(current_class_nid, method_name)
method_label = f".{method_name}()"
add_node(method_nid, method_label, lineno)
add_edge(current_class_nid, method_nid, "method", lineno)
if "auraenabled" in pending_annotations or "invocablemethod" in pending_annotations:
add_edge(file_nid, method_nid, "contains", lineno, confidence="INFERRED")
pending_annotations = []
continue
pending_annotations = []
for sm in soql_re.finditer(line_text):
sobject = sm.group(1)
sob_nid = _make_id(sobject)
if sob_nid not in seen_ids:
add_node(sob_nid, sobject, lineno)
src = current_class_nid or file_nid
add_edge(src, sob_nid, "uses", lineno, confidence="INFERRED")
for dm in dml_re.finditer(line_text):
dml_op = dm.group(1).lower()
dml_nid = _make_id(f"dml_{dml_op}")
if dml_nid not in seen_ids:
add_node(dml_nid, dml_op, lineno)
src = current_class_nid or file_nid
add_edge(src, dml_nid, "uses", lineno, confidence="INFERRED")
return {"nodes": nodes, "edges": edges}
+230
View File
@@ -0,0 +1,230 @@
"""Bash extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
from pathlib import Path
from typing import Any
from graphify.extractors.base import _file_stem, _make_id, _read_text
def extract_bash(path: Path) -> dict:
"""Extract functions, source imports, and cross-function calls from a .sh file."""
try:
import tree_sitter_bash as tsbash
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree-sitter-bash not installed"}
try:
language = Language(tsbash.language())
parser = Parser(language)
source = path.read_bytes()
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
function_bodies: list[tuple[str, Any]] = []
defined_functions: set[str] = set()
from graphify.security import sanitize_metadata # module-level cached import
def add_node(nid: str, label: str, line: int, kind: str = "code") -> None:
if nid and nid not in seen_ids:
seen_ids.add(nid)
nodes.append({"id": nid, "label": label, "file_type": "code",
"source_file": str_path, "source_location": f"L{line}",
"metadata": sanitize_metadata({"language": "bash", "kind": kind})}) # noqa: E501
def add_edge(src: str, tgt: str, relation: str, line: int,
confidence: str = "EXTRACTED", weight: float = 1.0,
context: str | None = None) -> None:
if not src or not tgt or src == tgt:
return
edge = {"source": src, "target": tgt, "relation": relation,
"confidence": confidence, "source_file": str_path,
"source_location": f"L{line}", "weight": weight}
if context:
edge["context"] = context
edges.append(edge)
file_nid = _make_id(str(path))
# file_nid is fully path-derived and never produced by _make_id(stem, func_name),
# so appending "__entry" guarantees a distinct ID from any function node.
entry_nid = file_nid + "__entry"
add_node(file_nid, path.name, 1, kind="file")
add_node(entry_nid, f"{path.name} script", 1, kind="bash_entrypoint")
add_edge(file_nid, entry_nid, "contains", 1)
_BASH_SOURCE_COMMANDS = frozenset({"source", "."})
# Parent node types that mean a contained command is part of a substitution
# or expansion, not a real function call. Token-level filtering misses
# these because `$(build)` exposes `build` as a child command whose name
# token has no metacharacters — only the parent does.
_BASH_EXPANSION_PARENTS = frozenset({
"command_substitution",
"process_substitution",
})
def text(node) -> str:
return source[node.start_byte:node.end_byte].decode("utf-8", errors="replace")
def is_inside_expansion(node) -> bool:
parent = node.parent
while parent is not None:
if parent.type in _BASH_EXPANSION_PARENTS:
return True
parent = parent.parent
return False
def literal(node) -> str | None:
# Token-level filter: rejects names containing shell metacharacters.
# Combined with `is_inside_expansion` for parent-context rejection.
raw = text(node).strip()
if not raw:
return None
if raw[0:1] in {"'", '"'} and raw[-1:] == raw[0]:
raw = raw[1:-1]
if any(token in raw for token in ("$", "`", "$(", "<(", ">", "|", ";", "&")):
return None
return raw
def _bash_func_name(node) -> str | None:
"""Get the name from a function_definition node."""
# bash grammar: function_definition has a word child (the name)
for child in node.children:
if child.type == "word":
return literal(child)
return None
def walk_calls(body_node, func_nid: str, seen_calls: set) -> None:
if body_node is None:
return
for child in body_node.children:
if child.type == "function_definition":
# Skip nested function definitions — their bodies are walked
# separately, so we don't attribute their calls to the
# enclosing scope.
continue
if child.type == "command" and not is_inside_expansion(child):
cmd_name_node = child.child_by_field_name("name")
if cmd_name_node is None and child.children:
cmd_name_node = child.children[0]
if cmd_name_node:
name = literal(cmd_name_node)
# Defined-functions wins. Skip-lists for external commands
# would create false negatives when a user defines a
# function shadowing an external (`install`, `find`, etc.).
if name and name in defined_functions:
tgt = _make_id(stem, name)
key = (func_nid, tgt)
if tgt and key not in seen_calls:
seen_calls.add(key)
add_edge(func_nid, tgt, "calls",
child.start_point[0] + 1,
confidence="EXTRACTED", context="call")
walk_calls(child, func_nid, seen_calls)
def walk(node, parent_nid: str) -> None:
t = node.type
if t == "function_definition":
name = _bash_func_name(node)
if name:
fn_nid = _make_id(stem, name)
line = node.start_point[0] + 1
add_node(fn_nid, f"{name}()", line, kind="bash_function")
add_edge(parent_nid, fn_nid, "defines", line)
defined_functions.add(name)
# find the compound_statement body
body = None
for child in node.children:
if child.type == "compound_statement":
body = child
break
function_bodies.append((fn_nid, body))
# Recurse into the body so nested function definitions are discovered
# and added to function_bodies for the second-pass walk_calls.
if body is not None:
walk(body, fn_nid)
return
if t == "command":
if is_inside_expansion(node):
return
cmd_name_node = node.child_by_field_name("name")
if cmd_name_node is None and node.children:
cmd_name_node = node.children[0]
if cmd_name_node:
cmd = literal(cmd_name_node)
if cmd in _BASH_SOURCE_COMMANDS and cmd not in defined_functions:
# find the path argument (first word after command name)
args = [c for c in node.children
if c.type in ("word", "string", "concatenation")
and c != cmd_name_node]
if args:
raw = _read_text(args[0], source).strip().strip("'\"")
line = node.start_point[0] + 1
if raw.startswith((".", "/")):
resolved = (path.parent / raw).resolve()
# Only emit the edge if the target actually exists on
# disk — prevents graph pollution from crafted paths
# like `source ../../etc/passwd` that traverse outside
# the project tree (B-1).
if resolved.exists():
tgt_nid = _make_id(str(resolved))
add_edge(file_nid, tgt_nid, "imports_from", line,
context="import")
else:
tgt_nid = _make_id(raw)
if tgt_nid:
add_edge(file_nid, tgt_nid, "imports", line,
context="import")
return
if t == "declaration_command":
# export/declare/readonly VAR=value at program level
if node.parent and node.parent.type == "program":
for child in node.children:
if child.type == "variable_assignment":
var_node = child.child_by_field_name("name")
if var_node:
var = _read_text(var_node, source).strip()
if var:
var_nid = _make_id(stem, var)
line = child.start_point[0] + 1
add_node(var_nid, var, line)
add_edge(file_nid, var_nid, "defines", line)
return
for child in node.children:
walk(child, parent_nid)
# Pre-pass: collect all defined function names so the source-command handler
# in walk() can detect user-defined functions that shadow 'source' / '.'
# regardless of definition order in the file.
def _prescan_functions(node) -> None:
if node.type == "function_definition":
name = _bash_func_name(node)
if name:
defined_functions.add(name)
for child in node.children:
_prescan_functions(child)
else:
for child in node.children:
_prescan_functions(child)
_prescan_functions(root)
walk(root, file_nid)
# Second pass: cross-function calls
top_seen: set = set()
walk_calls(root, entry_nid, top_seen) # top-level calls attributed to the entrypoint
for fn_nid, body in function_bodies:
walk_calls(body, fn_nid, set())
return {"nodes": nodes, "edges": edges}
+528
View File
@@ -0,0 +1,528 @@
"""Dart extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
import re
from pathlib import Path
from graphify.extractors.base import _file_stem, _make_id
def extract_dart(path: Path) -> dict:
"""Extract classes, mixins, functions, imports, generic calls, and annotations from a .dart file using regex."""
try:
src = path.read_text(encoding="utf-8", errors="replace")
except OSError:
return {"error": f"cannot read {path}"}
# Remove inline and multi-line comments while leaving string literals untouched to prevent stripping URLs/paths inside strings
comment_string_pattern = re.compile(
r'"""(?:\\.|[\s\S])*?"""'
r"|'''(?:\\.|[\s\S])*?'''"
r'|"(?:\\.|[^"\\])*"'
r"|'(?:\\.|[^'\\])*'"
r"|/\*[\s\S]*?\*/"
r"|//[^\n]*"
)
def _comment_replace(match: re.Match) -> str:
token = match.group(0)
if token.startswith("/"):
return ""
return token
src_clean = comment_string_pattern.sub(_comment_replace, src)
stem = _file_stem(path)
file_nid = _make_id(str(path))
# Check if this is a part-of file and redirect to parent
part_of_match = re.search(r"^\s*part\s+of\s+['\"]([^'\"]+)['\"]", src_clean, re.MULTILINE)
is_part = False
if part_of_match:
parent_ref = part_of_match.group(1)
if parent_ref.endswith(".dart"):
try:
parent_path = (path.parent / parent_ref).resolve()
if parent_path.exists():
stem = _file_stem(parent_path)
file_nid = _make_id(str(parent_path))
is_part = True
except Exception:
pass
nodes = []
if not is_part:
nodes.append({"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str(path), "source_location": None})
edges = []
defined: set[str] = set()
def add_node(nid: str, label: str, ftype: str = "code", source_file: str | None = str(path)) -> None:
if nid not in defined:
nodes.append({"id": nid, "label": label, "file_type": ftype,
"source_file": source_file, "source_location": None})
defined.add(nid)
def add_edge(src_id: str, tgt_id: str, relation: str, weight: float = 1.0, context: str | None = None) -> None:
edge = {"source": src_id, "target": tgt_id, "relation": relation,
"confidence": "EXTRACTED", "confidence_score": 1.0,
"source_file": str(path), "source_location": None, "weight": weight}
if context:
edge["context"] = context
edges.append(edge)
def _split_types(text: str) -> list[str]:
parts = []
current = []
depth = 0
for char in text:
if char == "<":
depth += 1
current.append(char)
elif char == ">":
depth -= 1
current.append(char)
elif char == "," and depth == 0:
parts.append("".join(current).strip())
current = []
else:
current.append(char)
if current:
parts.append("".join(current).strip())
return [p for p in parts if p]
def _find_matching_brace(text: str, start_pos: int) -> int:
brace_count = 0
in_double_quote = False
in_single_quote = False
escape = False
first_brace = text.find("{", start_pos)
if first_brace == -1:
return len(text)
brace_count = 1
i = first_brace + 1
n = len(text)
while i < n:
char = text[i]
if escape:
escape = False
i += 1
continue
if char == "\\":
escape = True
i += 1
continue
if text[i:i+3] == '"""' and not in_single_quote:
i += 3
end = text.find('"""', i)
i = end + 3 if end != -1 else n
continue
if text[i:i+3] == "'''" and not in_double_quote:
i += 3
end = text.find("'''", i)
i = end + 3 if end != -1 else n
continue
if char == '"' and not in_single_quote:
in_double_quote = not in_double_quote
elif char == "'" and not in_double_quote:
in_single_quote = not in_single_quote
elif not in_double_quote and not in_single_quote:
if char == "{":
brace_count += 1
elif char == "}":
brace_count -= 1
if brace_count == 0:
return i + 1
i += 1
return len(text)
# 1. Classes, mixins, and enums declarations (with inheritance, mixins, interfaces, and generics)
# Supports multiple combined modifiers (e.g., abstract base class, mixin class) without capturing "class" as a name
class_pattern = r"^\s*(?:(?:abstract|sealed|base|interface|final|mixin)\s+)*(?:class|mixin|enum|extension\s+type)\s+(\w+)"
for m in re.finditer(class_pattern, src_clean, re.MULTILINE):
class_name = m.group(1)
class_nid = _make_id(stem, class_name)
add_node(class_nid, class_name)
add_edge(file_nid, class_nid, "defines")
# Manually parse extends/on, with, and implements in header to handle nested generics brackets balanced
start_idx = m.end()
rest = src_clean[start_idx : start_idx + 500]
# Skip class generic parameters
if rest.lstrip().startswith("<"):
offset = rest.find("<")
depth = 1
i = offset + 1
while i < len(rest) and depth > 0:
if rest[i] == "<": depth += 1
elif rest[i] == ">": depth -= 1
i += 1
rest = rest[i:]
# Skip primary constructor (e.g. extension type MyExt(int id))
if rest.lstrip().startswith("("):
offset = rest.find("(")
depth = 1
i = offset + 1
while i < len(rest) and depth > 0:
if rest[i] == "(": depth += 1
elif rest[i] == ")": depth -= 1
i += 1
rest = rest[i:]
header_end = rest.find("{")
if header_end == -1:
header_end = rest.find(";")
if header_end == -1:
header_end = len(rest)
header = rest[:header_end]
base_class = None
generics = None
mixins_list = []
interfaces_list = []
# Parse extends or on
extends_m = re.search(r"^\s*(?:extends|on)\s+([a-zA-Z0-9_.]+)", header)
if extends_m:
base_class = extends_m.group(1)
rest_header = header[extends_m.end():]
if rest_header.strip().startswith("<"):
start_idx = rest_header.find("<")
depth = 1
i = start_idx + 1
while i < len(rest_header) and depth > 0:
if rest_header[i] == "<":
depth += 1
elif rest_header[i] == ">":
depth -= 1
if depth == 0:
generics = rest_header[start_idx + 1 : i]
break
i += 1
if generics is not None:
header = rest_header[i + 1:]
else:
header = rest_header
else:
header = rest_header
# Parse with
with_m = re.search(r"^\s*with\s+", header)
if with_m:
rest_header = header[with_m.end():]
impl_idx = rest_header.find("implements")
if impl_idx != -1:
mixins_str = rest_header[:impl_idx]
header = rest_header[impl_idx:]
else:
mixins_str = rest_header
header = ""
mixins_list = _split_types(mixins_str)
# Parse implements
impl_m = re.search(r"^\s*implements\s+", header)
if impl_m:
interfaces_list = _split_types(header[impl_m.end():])
# Map extends inheritance relation
if base_class:
base_nid = _make_id(base_class)
add_node(base_nid, base_class, source_file=None)
add_edge(class_nid, base_nid, "inherits")
# Map generic type arguments (e.g. MyBloc extends Bloc<MyEvent, MyState>)
if generics:
for gen in _split_types(generics):
gen_clean = gen.split("<")[0].strip()
if gen_clean not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "void"}:
gen_nid = _make_id(gen_clean)
add_node(gen_nid, gen_clean, source_file=None)
add_edge(class_nid, gen_nid, "references")
# Map mixins
for mixin in mixins_list:
mixin_clean = mixin.split("<")[0].strip()
mixin_nid = _make_id(mixin_clean)
add_node(mixin_nid, mixin_clean, source_file=None)
add_edge(class_nid, mixin_nid, "mixes_in")
# Map interfaces
for interface in interfaces_list:
interface_clean = interface.split("<")[0].strip()
interface_nid = _make_id(interface_clean)
add_node(interface_nid, interface_clean, source_file=None)
add_edge(class_nid, interface_nid, "implements")
# Extract class body for precise framework dependencies and event handling
start_idx = m.start()
brace_pos = src_clean.find("{", start_idx)
semi_pos = src_clean.find(";", start_idx)
has_body = brace_pos != -1
if has_body and semi_pos != -1 and semi_pos < brace_pos:
has_body = False
if has_body:
end_pos = _find_matching_brace(src_clean, start_idx)
class_body = src_clean[brace_pos:end_pos]
# Bloc event registration: on<MyEvent>()
for em in re.finditer(r"\bon<(\w+)>\s*\(", class_body):
event_name = em.group(1)
event_nid = _make_id(event_name)
add_node(event_nid, event_name, source_file=None)
add_edge(class_nid, event_nid, "calls", context="bloc_event")
# Bloc state emissions: emit(MyState) or yield MyState
for sm in re.finditer(r"\b(?:emit|yield)\s*\(?\s*(?:const\s+)?([A-Z]\w*)\b", class_body):
state_name = sm.group(1)
if state_name not in {"String", "List", "Map", "Set", "Future", "Stream", "Object"}:
state_nid = _make_id(state_name)
add_node(state_nid, state_name, source_file=None)
add_edge(class_nid, state_nid, "calls", context="emit_state")
# Bloc event additions: widget.add(MyEvent()) or bloc.add(MyEvent())
for am in re.finditer(r"\b(?:\w*[Bb]loc\w*|context\.read<\w+>\(\))\.add\(\s*(?:const\s+)?([A-Z]\w*)\b", class_body):
event_name = am.group(1)
if event_name not in {"String", "List", "Map", "Set", "Future", "Stream", "Object"}:
event_nid = _make_id(event_name)
add_node(event_nid, event_name, source_file=None)
add_edge(class_nid, event_nid, "calls", context="bloc_add_event")
# Riverpod provider references: ref.watch(provider)
for rm in re.finditer(r"\bref\.(?:watch|read|listen)\s*\(\s*(\w+)\b", class_body):
provider_name = rm.group(1)
provider_nid = _make_id(provider_name)
add_node(provider_nid, provider_name, source_file=None)
add_edge(class_nid, provider_nid, "references", context="riverpod_reference")
# Widget to Bloc references: BlocBuilder<MyBloc, ...>
for bm in re.finditer(r"\bBloc(?:Builder|Listener|Consumer|Provider|Selector)\s*<\s*([a-zA-Z0-9_]+)\b", class_body):
bloc_name = bm.group(1)
if bloc_name not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "void"}:
bloc_nid = _make_id(bloc_name)
add_node(bloc_nid, bloc_name, source_file=None)
add_edge(class_nid, bloc_nid, "references", context="bloc_widget_binding")
# context.read<MyBloc>() or BlocProvider.of<MyBloc>(context)
for lm in re.finditer(r"\b(?:read|watch|select|of)\s*<([a-zA-Z0-9_]+)>", class_body):
bloc_name = lm.group(1)
if bloc_name not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "void"}:
bloc_nid = _make_id(bloc_name)
add_node(bloc_nid, bloc_name, source_file=None)
add_edge(class_nid, bloc_nid, "references", context="bloc_lookup")
# 2. Annotations mapping (class, mixin, enum, or function level annotations)
# Support: @riverpod, @Riverpod(...), @injectable, @singleton, @RoutePage(), @HiveType(typeId: 0), @RestApi()
# Matches `@annotation` and links it to the next class/mixin/enum/function declaration in the file
annotation_pattern = r"@(\w+)(?:\([^)]*\))?"
for am in re.finditer(annotation_pattern, src_clean):
annotation_name = am.group(1)
if annotation_name in {"override", "deprecated", "required", "protected", "mustCallSuper"}:
continue
annotation_pos = am.end()
intervening_text = src_clean[annotation_pos : annotation_pos + 300]
class_m = re.search(r"^\s*(?:(?:abstract|sealed|base|interface|final|mixin)\s+)*(?:class|mixin|enum|extension\s+type)\s+(\w+)", intervening_text, re.MULTILINE)
func_m = re.search(r"^\s*(?:factory\s+|static\s+|async\s+|external\s+|abstract\s+)?(?:\([^)]+\)|[a-zA-Z0-9_<>,.?]+)(?:\s+[a-zA-Z0-9_<>,.?]+){0,3}\s+(\w+)\s*\(", intervening_text, re.MULTILINE)
target_nid = None
target_name = None
target_type = None
if class_m and func_m:
if class_m.start() < func_m.start():
target_name = class_m.group(1)
target_type = "class"
target_nid = _make_id(stem, target_name)
else:
target_name = func_m.group(1)
target_type = "function"
target_nid = _make_id(stem, target_name)
elif class_m:
target_name = class_m.group(1)
target_type = "class"
target_nid = _make_id(stem, target_name)
elif func_m:
target_name = func_m.group(1)
target_type = "function"
target_nid = _make_id(stem, target_name)
if target_nid and target_name:
actual_intervening = intervening_text[:min(class_m.start() if class_m else 300, func_m.start() if func_m else 300)]
if ";" not in actual_intervening and "}" not in actual_intervening and "{" not in actual_intervening:
annotation_nid = _make_id("annotation", annotation_name.lower())
add_node(annotation_nid, f"@{annotation_name}", ftype="concept", source_file=None)
add_edge(target_nid, annotation_nid, "configures")
# Riverpod specific provider generation mapping (supports camelCase class and functional providers)
if annotation_name.lower() == "riverpod":
if target_type == "class":
provider_name = target_name[0].lower() + target_name[1:] + "Provider" if len(target_name) > 1 else target_name.lower() + "Provider"
else:
provider_name = target_name + "Provider"
provider_nid = _make_id(provider_name)
add_node(provider_nid, provider_name, ftype="concept", source_file=str(path))
add_edge(target_nid, provider_nid, "defines", context="riverpod_provider")
# 2.5 Typedefs (Type Aliases)
typedef_pattern = r"^\s*typedef\s+(\w+)\s*(?:<[^>]+>)?\s*=\s*([a-zA-Z0-9_<>,.?\s]+);"
for m in re.finditer(typedef_pattern, src_clean, re.MULTILINE):
typedef_name = m.group(1)
target_type = m.group(2).split("<")[0].split(".")[-1].strip()
if target_type not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "List", "Map", "Set", "void", "Function"}:
typedef_nid = _make_id(stem, typedef_name)
add_node(typedef_nid, typedef_name)
add_edge(file_nid, typedef_nid, "defines")
target_nid = _make_id(target_type)
add_node(target_nid, target_type, source_file=None)
add_edge(typedef_nid, target_nid, "references", context="typedef")
# 3. Extensions (extension MyExt on MyClass)
ext_pattern = r"^\s{0,4}extension\s+(\w+)?(?:<[^>]+>)?\s+on\s+(\w+)"
for m in re.finditer(ext_pattern, src_clean, re.MULTILINE):
ext_name = m.group(1) or f"{stem}_anonymous_extension"
target_class = m.group(2)
ext_nid = _make_id(stem, ext_name)
label = m.group(1) or f"Extension on {target_class}"
add_node(ext_nid, label)
add_edge(file_nid, ext_nid, "defines")
target_nid = _make_id(target_class)
add_node(target_nid, target_class, source_file=None)
add_edge(ext_nid, target_nid, "extends")
# 4. Top-level and class-level variable declarations (generic variables, records, late, and destructuring)
# Restrict indentation to 0-2 spaces to avoid matching local variables inside functions or switch expressions
var_pattern = r"^\s{0,2}(?:late\s+)?(?:(?:final|const|var)\s+)?(?:\([^)]+\)\s+|([a-zA-Z0-9_<>,.?]+(?:\s+[a-zA-Z0-9_<>,.?]+){0,3})\s+)?(?:(\w+)|(?:\w+\s*)?\(([^)]+)\))\s*(?:=|$|;)"
for m in re.finditer(var_pattern, src_clean, re.MULTILINE):
var_type = m.group(1)
single_name = m.group(2)
destructured_names = m.group(3)
if not re.match(r"^\s*(?:late|final|const|var)\b", m.group(0)) and not var_type:
continue
if single_name:
if single_name not in {"if", "for", "while", "switch", "catch", "return"}:
var_nid = _make_id(stem, single_name)
add_node(var_nid, single_name)
add_edge(file_nid, var_nid, "defines")
if var_type and var_type not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "List", "Map", "Set", "void"}:
clean_type = var_type.split("<")[0].split(".")[-1].strip()
type_nid = _make_id(clean_type)
add_node(type_nid, clean_type, source_file=None)
add_edge(file_nid, type_nid, "references", context="variable_type")
elif destructured_names:
for name in [n.strip() for n in destructured_names.split(",") if n.strip()]:
if ":" in name:
name = name.split(":")[-1].strip()
if re.match(r"^[a-zA-Z_]\w*$", name) and not re.match(r"^[A-Z]", name):
if name not in {"if", "for", "while", "switch", "catch", "return"}:
var_nid = _make_id(stem, name)
add_node(var_nid, name)
add_edge(file_nid, var_nid, "defines")
# 5. Top-level and member functions/methods (supports typed/generic/record return types and Riverpod/Bloc references)
# Restrict indentation to 0-2 spaces to avoid matching nested local functions or methods inside multiline switch statements
method_pattern = r"^\s{0,2}(?:factory\s+|static\s+|async\s+|external\s+|abstract\s+)?(?:\([^)]+\)|[a-zA-Z0-9_<>,.?]+)(?:\s+[a-zA-Z0-9_<>,.?]+){0,3}\s+(\w+(?:\.\w+)?)\s*\("
for m in re.finditer(method_pattern, src_clean, re.MULTILINE):
raw_name = m.group(1)
name = raw_name.split(".")[-1]
if name in {"if", "for", "while", "switch", "catch", "return", "void", "dynamic", "final", "const", "get", "set"}:
continue
if re.match(r"^[A-Z]", name):
continue
nid = _make_id(stem, name)
add_node(nid, name)
add_edge(file_nid, nid, "defines")
# Get function body using matching brace to extract Riverpod reference patterns
start_idx = m.start()
brace_pos = src_clean.find("{", start_idx)
semi_pos = src_clean.find(";", start_idx)
arrow_pos = src_clean.find("=>", start_idx)
has_body = brace_pos != -1
if has_body and semi_pos != -1 and semi_pos < brace_pos:
has_body = False
if has_body and arrow_pos != -1 and arrow_pos < brace_pos:
has_body = False
if has_body:
end_pos = _find_matching_brace(src_clean, start_idx)
func_body = src_clean[brace_pos:end_pos]
# Extract Riverpod provider references: ref.watch(provider)
for rm in re.finditer(r"\bref\.(?:watch|read|listen)\s*\(\s*(\w+)\b", func_body):
provider_name = rm.group(1)
provider_nid = _make_id(provider_name)
add_node(provider_nid, provider_name, source_file=None)
add_edge(nid, provider_nid, "references", context="riverpod_reference")
# Extract Bloc event additions: widget.add(MyEvent()) or bloc.add(MyEvent())
for am in re.finditer(r"\b(?:\w*[Bb]loc\w*|context\.read<\w+>\(\))\.add\(\s*(?:const\s+)?([A-Z]\w*)\b", func_body):
event_name = am.group(1)
if event_name not in {"String", "List", "Map", "Set", "Future", "Stream", "Object"}:
event_nid = _make_id(event_name)
add_node(event_nid, event_name, source_file=None)
add_edge(nid, event_nid, "calls", context="bloc_add_event")
# context.read<MyBloc>() or BlocProvider.of<MyBloc>(context)
for lm in re.finditer(r"\b(?:read|watch|select|of)\s*<([a-zA-Z0-9_]+)>", func_body):
bloc_name = lm.group(1)
if bloc_name not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "void"}:
bloc_nid = _make_id(bloc_name)
add_node(bloc_nid, bloc_name, source_file=None)
add_edge(nid, bloc_nid, "references", context="bloc_lookup")
# Universal Navigation Patters (GoRouter, AutoRoute, Navigator)
for nm in re.finditer(r"\b(?:go|push|goNamed|pushNamed|replace|replaceNamed)\s*\(\s*(?:context\s*,\s*)?['\"]([a-zA-Z0-9_/?=&%-]+)['\"]", func_body):
route_path = nm.group(1)
route_nid = _make_id("route", route_path.replace("/", "_").replace("?", "_").replace("=", "_").replace("&", "_"))
add_node(route_nid, f"Route {route_path}", ftype="concept", source_file=None)
add_edge(nid, route_nid, "navigates", context="route_path")
for cm in re.finditer(r"\b(?:go|push|goNamed|pushNamed|replace|replaceNamed)\s*\(\s*(?:context\s*,\s*)?([A-Z][a-zA-Z0-9_]*\.[a-zA-Z0-9_]+)", func_body):
route_const = cm.group(1)
route_nid = _make_id("route", route_const.replace(".", "_"))
add_node(route_nid, route_const, ftype="concept", source_file=None)
add_edge(nid, route_nid, "navigates", context="route_const")
for om in re.finditer(r"\b(?:push|replace)\s*\(\s*(?:context\s*,\s*)?.*?\b([A-Z]\w*(?:Route|Screen|Page))\b", func_body):
route_class = om.group(1)
route_nid = _make_id(route_class)
add_node(route_nid, route_class, source_file=None)
add_edge(nid, route_nid, "navigates", context="route_object")
# 6. Imports and Exports
for m in re.finditer(r"""^\s*import\s+['"]([^'"]+)['"]""", src_clean, re.MULTILINE):
pkg = m.group(1)
tgt_nid = _make_id(pkg)
add_node(tgt_nid, pkg, source_file=None)
add_edge(file_nid, tgt_nid, "imports")
for m in re.finditer(r"""^\s*export\s+['"]([^'"]+)['"]""", src_clean, re.MULTILINE):
pkg = m.group(1)
tgt_nid = _make_id(pkg)
add_node(tgt_nid, pkg, source_file=None)
add_edge(file_nid, tgt_nid, "exports")
# 7. Generic Invocations / Type Lookups (Universal Dependency Lookup)
# Matches any method call with type parameters: methodName<Type>() or object.methodName<Type>()
# Automatically extracts GetIt, Injectable, Riverpod, Provider, BlocProvider, and InheritedWidget type lookups!
generic_call_pattern = r"\b\w+<([a-zA-Z0-9_.]+(?:<[a-zA-Z0-9_.,\s<>]+>)?)\s*>\s*\("
type_blacklist = {"String", "int", "double", "bool", "num", "dynamic", "Object", "List", "Map", "Set", "Future", "Stream", "void"}
for m in re.finditer(generic_call_pattern, src_clean):
type_name = m.group(1).split(".")[-1].strip()
clean_name = type_name.split("<")[0].strip()
if clean_name not in type_blacklist:
target_nid = _make_id(clean_name)
add_node(target_nid, clean_name, source_file=None)
add_edge(file_nid, target_nid, "references", context="type_lookup")
return {"nodes": nodes, "edges": edges}
+494
View File
@@ -0,0 +1,494 @@
"""Dm extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
import re
from pathlib import Path
from typing import Any
from graphify.extractors.base import _file_stem, _make_id, _read_text
def extract_dm(path: Path) -> dict:
"""Extract types, procs, includes, and calls from a .dm/.dme file."""
try:
import tree_sitter_dm as tsdm
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree-sitter-dm not installed"}
try:
language = Language(tsdm.language())
parser = Parser(language)
source = path.read_bytes()
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
function_bodies: list[tuple[str, Any, "str | None"]] = []
def add_node(nid: str, label: str, line: int) -> None:
if nid and nid not in seen_ids:
seen_ids.add(nid)
nodes.append({"id": nid, "label": label, "file_type": "code",
"source_file": str_path, "source_location": f"L{line}"})
def add_edge(src: str, tgt: str, relation: str, line: int,
confidence: str = "EXTRACTED", weight: float = 1.0,
context: str | None = None) -> None:
if not src or not tgt or src == tgt:
return
edge: dict = {"source": src, "target": tgt, "relation": relation,
"confidence": confidence, "source_file": str_path,
"source_location": f"L{line}", "weight": weight}
if context:
edge["context"] = context
edges.append(edge)
file_nid = _make_id(str(path))
add_node(file_nid, path.name, 1)
def _type_path_text(node) -> str:
return _read_text(node, source).strip()
def _ensure_type(path_text: str, line: int) -> str:
nid = _make_id(stem, path_text)
add_node(nid, path_text, line)
return nid
def _find_child(node, type_name: str):
for c in node.children:
if c.type == type_name:
return c
return None
def _read_include_path(file_node) -> str:
if file_node is None:
return ""
if file_node.type == "string_literal":
parts = []
for c in file_node.children:
if c.type == "string_content":
parts.append(_read_text(c, source))
return "".join(parts)
return _read_text(file_node, source).strip("'\"")
def walk(node, parent_type_path: "str | None" = None,
parent_type_nid: "str | None" = None) -> None:
t = node.type
line = node.start_point[0] + 1
if t == "preproc_include":
file_node = node.child_by_field_name("file")
raw = _read_include_path(file_node)
if raw:
norm = raw.replace("\\", "/").lstrip("./")
resolved = (path.parent / norm).resolve()
edge: dict = {
"source": file_nid,
"target": _make_id(str(resolved)) if resolved.exists() else _make_id(norm),
"relation": "imports_from" if resolved.exists() else "imports",
"context": "import",
"confidence": "EXTRACTED",
"source_file": str_path,
"source_location": f"L{line}",
"weight": 1.0,
}
if not resolved.exists():
edge["external"] = True
edges.append(edge)
return
if t == "type_definition":
tp_node = _find_child(node, "type_path")
if tp_node is None:
return
type_path_str = _type_path_text(tp_node)
type_nid = _ensure_type(type_path_str, line)
add_edge(file_nid, type_nid, "contains", line)
body = _find_child(node, "type_body")
if body is not None:
for c in body.children:
walk(c, parent_type_path=type_path_str, parent_type_nid=type_nid)
return
if t in ("type_body_intended", "type_body_braced"):
for c in node.children:
walk(c, parent_type_path, parent_type_nid)
return
if t in ("type_proc_definition", "type_proc_override"):
if parent_type_nid is None or parent_type_path is None:
return
name_node = node.child_by_field_name("name")
if name_node is None:
return
proc_name = _read_text(name_node, source)
proc_nid = _make_id(stem, parent_type_path, proc_name)
add_node(proc_nid, f"{parent_type_path}/{proc_name}()", line)
add_edge(parent_type_nid, proc_nid, "method", line)
block = _find_child(node, "block")
if block is not None:
function_bodies.append((proc_nid, block, parent_type_path))
return
if t in ("proc_definition", "proc_override"):
tp_node = _find_child(node, "type_path")
owner_path: "str | None" = None
owner_nid: "str | None" = None
if tp_node is not None:
owner_path = _type_path_text(tp_node)
owner_nid = _ensure_type(owner_path, line)
add_edge(file_nid, owner_nid, "contains", line)
name_node = node.child_by_field_name("name")
if name_node is None:
return
proc_name = _read_text(name_node, source)
if owner_path and owner_nid:
proc_nid = _make_id(stem, owner_path, proc_name)
add_node(proc_nid, f"{owner_path}/{proc_name}()", line)
add_edge(owner_nid, proc_nid, "method", line)
else:
proc_nid = _make_id(stem, proc_name)
add_node(proc_nid, f"{proc_name}()", line)
add_edge(file_nid, proc_nid, "contains", line)
block = _find_child(node, "block")
if block is not None:
function_bodies.append((proc_nid, block, owner_path))
return
if t in ("operator_override", "type_operator_override"):
return
for child in node.children:
walk(child, parent_type_path, parent_type_nid)
walk(root)
label_to_nids: dict[str, list[str]] = {}
path_to_nids: dict[str, list[str]] = {}
for n in nodes:
label = n["label"].strip("()")
last = label.rsplit("/", 1)[-1] if "/" in label else label
if last:
label_to_nids.setdefault(last.lower(), []).append(n["id"])
if label.startswith("/"):
path_to_nids.setdefault(label.lower(), []).append(n["id"])
seen_call_pairs: set[tuple[str, str]] = set()
raw_calls: list[dict] = []
def _emit_call(caller_nid: str, callee: str, line: int, is_member: bool) -> None:
candidates = label_to_nids.get(callee.lower(), [])
tgt_nid = candidates[0] if len(candidates) == 1 else None
if tgt_nid and tgt_nid != caller_nid:
pair = (caller_nid, tgt_nid)
if pair in seen_call_pairs:
return
seen_call_pairs.add(pair)
edges.append({
"source": caller_nid, "target": tgt_nid, "relation": "calls",
"context": "call", "confidence": "EXTRACTED",
"source_file": str_path, "source_location": f"L{line}", "weight": 1.0,
})
else:
raw_calls.append({
"caller_nid": caller_nid, "callee": callee,
"is_member_call": is_member, "source_file": str_path,
"source_location": f"L{line}",
})
def walk_calls(body_node, caller_nid: str) -> None:
if body_node is None:
return
t = body_node.type
if t in ("proc_definition", "proc_override", "type_proc_definition",
"type_proc_override", "type_definition"):
return
if t == "call_expression":
name_node = body_node.child_by_field_name("name")
if name_node is not None:
callee = _read_text(name_node, source)
if callee and callee != "..":
_emit_call(caller_nid, callee, body_node.start_point[0] + 1,
is_member=False)
elif t == "field_proc_expression":
proc_field = body_node.child_by_field_name("proc")
if proc_field is not None:
callee = _read_text(proc_field, source)
if callee:
_emit_call(caller_nid, callee, body_node.start_point[0] + 1,
is_member=True)
elif t == "new_expression":
tp_node = _find_child(body_node, "type_path")
if tp_node is not None:
target_text = _type_path_text(tp_node)
candidates = path_to_nids.get(target_text.lower(), [])
tgt_nid = candidates[0] if len(candidates) == 1 else None
if tgt_nid and tgt_nid != caller_nid:
pair = (caller_nid, tgt_nid)
if pair not in seen_call_pairs:
seen_call_pairs.add(pair)
edges.append({
"source": caller_nid, "target": tgt_nid,
"relation": "instantiates", "context": "call",
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{body_node.start_point[0] + 1}",
"weight": 1.0,
})
for child in body_node.children:
walk_calls(child, caller_nid)
for proc_nid, block, _owner_path in function_bodies:
walk_calls(block, proc_nid)
return {"nodes": nodes, "edges": edges, "raw_calls": raw_calls}
def _read_dmi_description(data: bytes) -> str:
"""Pull the BYOND metadata text out of a .dmi PNG, or empty string on failure."""
import struct
import zlib as _zlib
if not data.startswith(b"\x89PNG\r\n\x1a\n"):
return ""
i = 8
while i + 8 <= len(data):
length = struct.unpack(">I", data[i:i + 4])[0]
chunk_type = data[i + 4:i + 8]
payload = data[i + 8:i + 8 + length]
if chunk_type in (b"tEXt", b"zTXt"):
try:
null = payload.index(b"\x00")
except ValueError:
return ""
keyword = payload[:null]
if keyword == b"Description":
if chunk_type == b"zTXt":
return _zlib.decompressobj().decompress(payload[null + 2:], max_length=1024 * 1024).decode("utf-8", errors="replace")
return payload[null + 1:].decode("utf-8", errors="replace")
i += 8 + length + 4
return ""
def extract_dmi(path: Path) -> dict:
"""Extract icon state names from a .dmi (BYOND PNG icon sheet)."""
try:
data = path.read_bytes()
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
str_path = str(path)
stem = _file_stem(path)
file_nid = _make_id(str(path))
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str_path, "source_location": "L1"}]
edges: list[dict] = []
seen: set[str] = {file_nid}
description = _read_dmi_description(data)
if not description:
return {"nodes": nodes, "edges": edges}
line_no = 0
for raw_line in description.splitlines():
line_no += 1
stripped = raw_line.strip()
if not stripped.startswith("state ="):
continue
value = stripped.split("=", 1)[1].strip()
if value.startswith('"') and value.endswith('"') and len(value) >= 2:
state_name = value[1:-1]
else:
state_name = value
if not state_name:
continue
nid = _make_id(stem, "state", state_name)
if nid in seen:
continue
seen.add(nid)
nodes.append({"id": nid, "label": f'"{state_name}"', "file_type": "code",
"source_file": str_path, "source_location": f"L{line_no}"})
edges.append({"source": file_nid, "target": nid, "relation": "contains",
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{line_no}", "weight": 1.0})
return {"nodes": nodes, "edges": edges}
_DMM_GRID_RE = re.compile(r"^\(\s*\d+\s*,\s*\d+\s*,\s*\d+\s*\)\s*=", re.MULTILINE)
def _split_dmm_tile(body: str) -> list[str]:
out: list[str] = []
buf: list[str] = []
depth = 0
in_string = False
escape = False
for ch in body:
if escape:
buf.append(ch)
escape = False
continue
if in_string:
buf.append(ch)
if ch == "\\":
escape = True
elif ch == '"':
in_string = False
continue
if ch == '"':
in_string = True
buf.append(ch)
elif ch in "({[":
depth += 1
buf.append(ch)
elif ch in ")}]":
depth -= 1
buf.append(ch)
elif ch == "," and depth == 0:
out.append("".join(buf).strip())
buf = []
else:
buf.append(ch)
tail = "".join(buf).strip()
if tail:
out.append(tail)
return out
def _dmm_type_path(entry: str) -> str:
brace = entry.find("{")
if brace != -1:
entry = entry[:brace]
return entry.strip()
def extract_dmm(path: Path) -> dict:
"""Extract type-path references from a .dmm map file's tile dictionary."""
try:
if path.stat().st_size > 50 * 1024 * 1024:
return {"nodes": [], "edges": [], "error": "file too large (>50 MB)"}
text = path.read_text(encoding="utf-8", errors="replace")
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
str_path = str(path)
file_nid = _make_id(str(path))
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str_path, "source_location": "L1"}]
edges: list[dict] = []
grid_match = _DMM_GRID_RE.search(text)
dict_text = text[:grid_match.start()] if grid_match else text
seen_targets: set[str] = set()
buf: list[str] = []
open_line = 0
depth = 0
in_string = False
escape = False
for line_idx, line in enumerate(dict_text.splitlines(), start=1):
for ch in line:
if escape:
escape = False
elif in_string:
if ch == "\\":
escape = True
elif ch == '"':
in_string = False
elif ch == '"':
in_string = True
elif ch == "(":
if depth == 0:
open_line = line_idx
depth += 1
elif ch == ")":
depth -= 1
buf.append(ch)
buf.append("\n")
if depth == 0 and buf:
chunk = "".join(buf)
buf = []
lp = chunk.find("(")
rp = chunk.rfind(")")
if lp == -1 or rp == -1 or rp <= lp:
continue
inner = chunk[lp + 1:rp]
for entry in _split_dmm_tile(inner):
tpath = _dmm_type_path(entry)
if not tpath.startswith("/"):
continue
tgt = _make_id(tpath)
if tgt in seen_targets:
continue
seen_targets.add(tgt)
edges.append({"source": file_nid, "target": tgt, "relation": "uses",
"context": "map", "confidence": "EXTRACTED",
"source_file": str_path,
"source_location": f"L{open_line}", "weight": 1.0})
return {"nodes": nodes, "edges": edges}
_DMF_WINDOW_RE = re.compile(r'^\s*window\s+"([^"]+)"\s*$')
_DMF_ELEM_RE = re.compile(r'^\s*elem\s+"([^"]+)"\s*$')
_DMF_TYPE_RE = re.compile(r'^\s*type\s*=\s*(\S+)\s*$')
def extract_dmf(path: Path) -> dict:
"""Extract windows and controls from a .dmf interface file."""
try:
text = path.read_text(encoding="utf-8", errors="replace")
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
str_path = str(path)
stem = _file_stem(path)
file_nid = _make_id(str(path))
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str_path, "source_location": "L1"}]
edges: list[dict] = []
seen: set[str] = {file_nid}
current_window_nid: str | None = None
current_elem_nid: str | None = None
current_elem_name: str | None = None
for line_idx, line in enumerate(text.splitlines(), start=1):
m = _DMF_WINDOW_RE.match(line)
if m:
name = m.group(1)
nid = _make_id(stem, "window", name)
if nid not in seen:
seen.add(nid)
nodes.append({"id": nid, "label": f'window "{name}"', "file_type": "code",
"source_file": str_path, "source_location": f"L{line_idx}"})
edges.append({"source": file_nid, "target": nid, "relation": "contains",
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{line_idx}", "weight": 1.0})
current_window_nid = nid
current_elem_nid = None
current_elem_name = None
continue
m = _DMF_ELEM_RE.match(line)
if m and current_window_nid is not None:
name = m.group(1)
nid = _make_id(stem, "elem", current_window_nid, name)
if nid not in seen:
seen.add(nid)
nodes.append({"id": nid, "label": f'elem "{name}"', "file_type": "code",
"source_file": str_path, "source_location": f"L{line_idx}"})
edges.append({"source": current_window_nid, "target": nid,
"relation": "contains", "confidence": "EXTRACTED",
"source_file": str_path, "source_location": f"L{line_idx}",
"weight": 1.0})
current_elem_nid = nid
current_elem_name = name
continue
m = _DMF_TYPE_RE.match(line)
if m and current_elem_nid is not None and current_elem_name is not None:
ctype = m.group(1)
for n in nodes:
if n["id"] == current_elem_nid and " [" not in n["label"]:
n["label"] = f'elem "{current_elem_name}" [{ctype}]'
break
return {"nodes": nodes, "edges": edges}
+309
View File
@@ -0,0 +1,309 @@
"""Fortran extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
from pathlib import Path
from graphify.extractors.base import _file_stem, _make_id, _read_text
_FORTRAN_CPP_EXTS = {".F", ".F90", ".F95", ".F03", ".F08"}
def _cpp_preprocess(path: Path) -> bytes:
"""Run cpp -w -P on a capital-F Fortran file and return preprocessed bytes.
Falls back to raw file bytes if cpp is not available. Capital-F extensions
conventionally require C preprocessor expansion (#ifdef MPI, #define REAL8, etc.)
before parsing.
Security (F-007): we pass `-nostdinc` and `-I /dev/null` so a malicious
source file containing `#include "/home/victim/.ssh/id_rsa"` (or any other
include directive) cannot inline arbitrary host files into the output that
we then ship to an LLM. Without these flags `cpp` happily resolves any
relative or absolute include path it can read, which is a corpus-side
file-exfiltration vector.
"""
import shutil
import subprocess
if not shutil.which("cpp"):
return path.read_bytes()
try:
# Pass an absolute path so a corpus file named like "-I/etc/x.F90" cannot
# be parsed by cpp as an option (cpp does not accept a "--" end-of-options
# terminator). An absolute path always begins with "/".
result = subprocess.run(
["cpp", "-w", "-P", "-nostdinc", "-I", "/dev/null", str(path.resolve())],
capture_output=True,
timeout=30,
)
if result.returncode == 0 and result.stdout:
return result.stdout
except Exception:
pass
return path.read_bytes()
def extract_fortran(path: Path) -> dict:
"""Extract programs, modules, subroutines, functions, use statements, and calls from Fortran files.
Capital-F extensions (.F, .F90, etc.) are run through the C preprocessor before
parsing so #ifdef/#define macros are resolved.
"""
try:
import tree_sitter_fortran as tsfortran
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree-sitter-fortran not installed"}
try:
language = Language(tsfortran.language())
parser = Parser(language)
source = _cpp_preprocess(path) if path.suffix in _FORTRAN_CPP_EXTS else path.read_bytes()
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
scope_bodies: list[tuple[str, object]] = []
def add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({
"id": nid,
"label": label,
"file_type": "code",
"source_file": str_path,
"source_location": f"L{line}",
})
def add_edge(src: str, tgt: str, relation: str, line: int,
confidence: str = "EXTRACTED", weight: float = 1.0,
context: str | None = None) -> None:
edge = {
"source": src,
"target": tgt,
"relation": relation,
"confidence": confidence,
"source_file": str_path,
"source_location": f"L{line}",
"weight": weight,
}
if context:
edge["context"] = context
edges.append(edge)
file_nid = _make_id(str(path))
add_node(file_nid, path.name, 1)
def _fortran_name(stmt_node) -> str | None:
"""Extract name from a *_statement node. Fortran is case-insensitive; lowercase."""
for child in stmt_node.children:
if child.type in ("name", "identifier"):
return _read_text(child, source).lower()
return None
def ensure_named_node(name: str, line: int) -> str:
nid = _make_id(stem, name)
if nid in seen_ids:
return nid
nid = _make_id(name)
if nid not in seen_ids:
# The name isn't defined in this file, so this is a cross-file reference
# (e.g. a `Thing` type annotation imported from another module). Emit a
# SOURCELESS stub — like the inheritance-base path below — so the
# corpus-level rewire can collapse it onto the real definition. A sourced
# stub here makes _disambiguate_colliding_node_ids bake the referencing
# file's path (with extension) into the id and blocks the rewire, which is
# the phantom-duplicate-node bug (#1402).
seen_ids.add(nid)
nodes.append({
"id": nid,
"label": name,
"file_type": "code",
"source_file": "",
"source_location": "",
"origin_file": str_path,
})
return nid
def emit_signature_refs(scope_node, fn_nid: str, is_function: bool) -> None:
"""Emit references[parameter_type] / references[return_type] edges for
a subroutine/function based on its variable_declaration siblings."""
stmt_type = "function_statement" if is_function else "subroutine_statement"
stmt = next((c for c in scope_node.children if c.type == stmt_type), None)
if stmt is None:
return
param_names: set[str] = set()
params_node = next((c for c in stmt.children if c.type == "parameters"), None)
if params_node is not None:
for c in params_node.children:
if c.type == "identifier":
param_names.add(_read_text(c, source).lower())
result_name: str | None = None
if is_function:
result_node = next((c for c in stmt.children if c.type == "function_result"), None)
if result_node is not None:
res_id = next((c for c in result_node.children if c.type == "identifier"), None)
if res_id is not None:
result_name = _read_text(res_id, source).lower()
else:
# implicit result variable: same name as the function
result_name = _fortran_name(stmt)
for child in scope_node.children:
if child.type != "variable_declaration":
continue
derived = next((c for c in child.children if c.type == "derived_type"), None)
if derived is None:
continue
type_name_node = next((c for c in derived.children if c.type == "type_name"), None)
if type_name_node is None:
continue
type_name = _read_text(type_name_node, source).lower()
for var in child.children:
if var.type != "identifier":
continue
var_name = _read_text(var, source).lower()
var_line = var.start_point[0] + 1
if var_name in param_names:
tgt = ensure_named_node(type_name, var_line)
if tgt != fn_nid:
add_edge(fn_nid, tgt, "references", var_line, context="parameter_type")
elif is_function and var_name == result_name:
tgt = ensure_named_node(type_name, var_line)
if tgt != fn_nid:
add_edge(fn_nid, tgt, "references", var_line, context="return_type")
def walk_calls(node, scope_nid: str) -> None:
if node is None:
return
t = node.type
if t in ("subroutine", "function", "module", "program", "internal_procedures"):
return
# call FOO(args) — tree-sitter-fortran uses subroutine_call
if t == "subroutine_call":
name_node = next((c for c in node.children if c.type == "identifier"), None)
if name_node:
callee = _read_text(name_node, source).lower()
target_nid = _make_id(stem, callee)
add_edge(scope_nid, target_nid, "calls", node.start_point[0] + 1,
confidence="EXTRACTED", context="call")
# x = compute(args) — function invocations are `call_expression`, which
# shares Fortran's `name(...)` syntax with array indexing. Only emit a
# call edge when the callee resolves to a procedure defined in this file
# (an array variable produces no matching node), so array accesses can't
# fabricate spurious `calls` edges.
elif t == "call_expression":
name_node = next((c for c in node.children if c.type == "identifier"), None)
if name_node:
callee = _read_text(name_node, source).lower()
target_nid = _make_id(stem, callee)
if target_nid in seen_ids and target_nid != scope_nid:
add_edge(scope_nid, target_nid, "calls", node.start_point[0] + 1,
confidence="EXTRACTED", context="call")
for child in node.children:
walk_calls(child, scope_nid)
def walk(node, scope_nid: str) -> None:
t = node.type
if t == "program":
stmt = next((c for c in node.children if c.type == "program_statement"), None)
name = _fortran_name(stmt) if stmt else None
if name:
nid = _make_id(stem, name)
line = node.start_point[0] + 1
add_node(nid, name, line)
add_edge(file_nid, nid, "defines", line)
scope_bodies.append((nid, node))
for child in node.children:
walk(child, nid)
return
if t == "module":
stmt = next((c for c in node.children if c.type == "module_statement"), None)
name = _fortran_name(stmt) if stmt else None
if name:
nid = _make_id(stem, name)
line = node.start_point[0] + 1
add_node(nid, name, line)
add_edge(file_nid, nid, "defines", line)
for child in node.children:
walk(child, nid)
return
# subroutines/functions inside a module live under internal_procedures
if t == "internal_procedures":
for child in node.children:
walk(child, scope_nid)
return
if t == "derived_type_definition":
stmt = next((c for c in node.children if c.type == "derived_type_statement"), None)
if stmt is not None:
name_node = next((c for c in stmt.children if c.type == "type_name"), None)
if name_node is not None:
type_name = _read_text(name_node, source).lower()
type_nid = _make_id(stem, type_name)
line = node.start_point[0] + 1
add_node(type_nid, type_name, line)
add_edge(scope_nid, type_nid, "defines", line)
return
if t == "subroutine":
stmt = next((c for c in node.children if c.type == "subroutine_statement"), None)
name = _fortran_name(stmt) if stmt else None
if name:
nid = _make_id(stem, name)
line = node.start_point[0] + 1
add_node(nid, f"{name}()", line)
add_edge(scope_nid, nid, "defines", line)
scope_bodies.append((nid, node))
emit_signature_refs(node, nid, is_function=False)
for child in node.children:
walk(child, nid)
return
if t == "function":
stmt = next((c for c in node.children if c.type == "function_statement"), None)
name = _fortran_name(stmt) if stmt else None
if name:
nid = _make_id(stem, name)
line = node.start_point[0] + 1
add_node(nid, f"{name}()", line)
add_edge(scope_nid, nid, "defines", line)
scope_bodies.append((nid, node))
emit_signature_refs(node, nid, is_function=True)
for child in node.children:
walk(child, nid)
return
if t == "use_statement":
line = node.start_point[0] + 1
# tree-sitter-fortran uses module_name node for the used module
name_node = next((c for c in node.children if c.type in ("module_name", "name", "identifier")), None)
if name_node:
mod_name = _read_text(name_node, source).lower()
imp_nid = _make_id(mod_name)
add_node(imp_nid, mod_name, line)
add_edge(scope_nid, imp_nid, "imports", line, context="use")
return
for child in node.children:
walk(child, scope_nid)
walk(root, file_nid)
_stmt_headers = {
"subroutine_statement", "function_statement",
"program_statement", "module_statement",
}
for scope_nid, body_node in scope_bodies:
for child in body_node.children:
if child.type not in _stmt_headers:
walk_calls(child, scope_nid)
return {"nodes": nodes, "edges": edges}
+396
View File
@@ -0,0 +1,396 @@
"""Go extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
from pathlib import Path
from graphify.extractors.base import _LANGUAGE_BUILTIN_GLOBALS, _file_stem, _make_id, _read_text
_GO_PREDECLARED_TYPES = frozenset({
"bool", "byte", "complex64", "complex128", "error", "float32", "float64",
"int", "int8", "int16", "int32", "int64", "rune", "string",
"uint", "uint8", "uint16", "uint32", "uint64", "uintptr", "any", "comparable",
})
def _go_collect_type_refs(node, source: bytes, generic: bool, out: list[tuple[str, str]]) -> None:
"""Walk a Go type expression; append (name, role) tuples."""
if node is None:
return
t = node.type
if t == "type_identifier":
text = _read_text(node, source)
if text and text not in _GO_PREDECLARED_TYPES:
out.append((text, "generic_arg" if generic else "type"))
return
if t == "qualified_type":
text = _read_text(node, source).rsplit(".", 1)[-1]
if text and text not in _GO_PREDECLARED_TYPES:
out.append((text, "generic_arg" if generic else "type"))
return
if t == "generic_type":
type_field = node.child_by_field_name("type")
if type_field is not None:
sub: list[tuple[str, str]] = []
_go_collect_type_refs(type_field, source, generic, sub)
out.extend(sub)
for c in node.children:
if c.type == "type_arguments":
for arg in c.children:
if arg.is_named:
_go_collect_type_refs(arg, source, True, out)
return
if t in ("pointer_type", "slice_type", "array_type", "map_type",
"channel_type", "parenthesized_type"):
for c in node.children:
if c.is_named:
_go_collect_type_refs(c, source, generic, out)
return
if node.is_named:
for c in node.children:
if c.is_named:
_go_collect_type_refs(c, source, generic, out)
def extract_go(path: Path) -> dict:
"""Extract functions, methods, type declarations, and imports from a .go file."""
try:
import tree_sitter_go as tsgo
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree-sitter-go not installed"}
try:
language = Language(tsgo.language())
parser = Parser(language)
source = path.read_bytes()
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = _file_stem(path)
# Use directory name as package scope so methods on the same type across
# multiple files in a package share one canonical type node.
pkg_scope = path.parent.name or stem
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
function_bodies: list[tuple[str, object]] = []
go_imported_pkgs: set[str] = set() # local names of imported packages
def add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({
"id": nid,
"label": label,
"file_type": "code",
"source_file": str_path,
"source_location": f"L{line}",
})
def add_edge(src: str, tgt: str, relation: str, line: int,
confidence: str = "EXTRACTED", weight: float = 1.0,
context: str | None = None) -> None:
edge = {
"source": src,
"target": tgt,
"relation": relation,
"confidence": confidence,
"source_file": str_path,
"source_location": f"L{line}",
"weight": weight,
}
if context:
edge["context"] = context
edges.append(edge)
file_nid = _make_id(str(path))
add_node(file_nid, path.name, 1)
def ensure_named_node(name: str, line: int) -> str:
nid = _make_id(pkg_scope, name)
if nid in seen_ids:
return nid
nid = _make_id(name)
if nid not in seen_ids:
# The name isn't declared in this file, so this is a cross-file reference
# (e.g. a type defined in another file of the package). Emit a SOURCELESS
# stub — like the inheritance-base path in the other extractors — so the
# corpus-level rewire can collapse it onto the real definition. A sourced
# stub here makes _disambiguate_colliding_node_ids bake the referencing
# file's path (with extension) into the id and blocks the rewire, which is
# the phantom-duplicate-node bug (#1402).
seen_ids.add(nid)
nodes.append({
"id": nid,
"label": name,
"file_type": "code",
"source_file": "",
"source_location": "",
"origin_file": str_path,
})
return nid
def emit_go_method_refs(func_node, func_nid: str, line: int) -> None:
params = func_node.child_by_field_name("parameters")
if params is not None:
for p in params.children:
if p.type != "parameter_declaration":
continue
type_node = p.child_by_field_name("type")
refs: list[tuple[str, str]] = []
_go_collect_type_refs(type_node, source, False, refs)
for ref_name, role in refs:
ctx = "generic_arg" if role == "generic_arg" else "parameter_type"
tgt = ensure_named_node(ref_name, line)
if tgt != func_nid:
add_edge(func_nid, tgt, "references", line, context=ctx)
result = func_node.child_by_field_name("result")
if result is not None:
if result.type == "parameter_list":
for p in result.children:
if p.type != "parameter_declaration":
continue
type_node = p.child_by_field_name("type")
if type_node is None:
for c in p.children:
if c.is_named:
type_node = c
break
refs = []
_go_collect_type_refs(type_node, source, False, refs)
for ref_name, role in refs:
ctx = "generic_arg" if role == "generic_arg" else "return_type"
tgt = ensure_named_node(ref_name, line)
if tgt != func_nid:
add_edge(func_nid, tgt, "references", line, context=ctx)
else:
refs = []
_go_collect_type_refs(result, source, False, refs)
for ref_name, role in refs:
ctx = "generic_arg" if role == "generic_arg" else "return_type"
tgt = ensure_named_node(ref_name, line)
if tgt != func_nid:
add_edge(func_nid, tgt, "references", line, context=ctx)
def walk(node) -> None:
t = node.type
if t == "function_declaration":
name_node = node.child_by_field_name("name")
if name_node:
func_name = _read_text(name_node, source)
line = node.start_point[0] + 1
func_nid = _make_id(stem, func_name)
add_node(func_nid, f"{func_name}()", line)
add_edge(file_nid, func_nid, "contains", line)
emit_go_method_refs(node, func_nid, line)
body = node.child_by_field_name("body")
if body:
function_bodies.append((func_nid, body))
return
if t == "method_declaration":
receiver = node.child_by_field_name("receiver")
receiver_type: str | None = None
if receiver:
for param in receiver.children:
if param.type == "parameter_declaration":
type_node = param.child_by_field_name("type")
if type_node:
receiver_type = _read_text(type_node, source).lstrip("*").strip()
break
name_node = node.child_by_field_name("name")
if not name_node:
return
method_name = _read_text(name_node, source)
line = node.start_point[0] + 1
if receiver_type:
parent_nid = _make_id(pkg_scope, receiver_type)
add_node(parent_nid, receiver_type, line)
method_nid = _make_id(parent_nid, method_name)
add_node(method_nid, f".{method_name}()", line)
add_edge(parent_nid, method_nid, "method", line)
else:
method_nid = _make_id(stem, method_name)
add_node(method_nid, f"{method_name}()", line)
add_edge(file_nid, method_nid, "contains", line)
emit_go_method_refs(node, method_nid, line)
body = node.child_by_field_name("body")
if body:
function_bodies.append((method_nid, body))
return
if t == "type_declaration":
for child in node.children:
if child.type != "type_spec":
continue
name_node = child.child_by_field_name("name")
if not name_node:
continue
type_name = _read_text(name_node, source)
line = child.start_point[0] + 1
type_nid = _make_id(pkg_scope, type_name)
add_node(type_nid, type_name, line)
add_edge(file_nid, type_nid, "contains", line)
# Type body: struct fields (with embeds) or interface embedding.
type_body = None
for tc in child.children:
if tc.type in ("struct_type", "interface_type"):
type_body = tc
break
if type_body is None:
continue
if type_body.type == "struct_type":
for fdl in type_body.children:
if fdl.type != "field_declaration_list":
continue
for field in fdl.children:
if field.type != "field_declaration":
continue
has_name = any(
fc.type == "field_identifier" for fc in field.children
)
type_node = field.child_by_field_name("type")
if type_node is None:
for fc in field.children:
if fc.is_named and fc.type != "field_identifier":
type_node = fc
break
refs: list[tuple[str, str]] = []
_go_collect_type_refs(type_node, source, False, refs)
for ref_name, role in refs:
tgt = ensure_named_node(ref_name, field.start_point[0] + 1)
if tgt == type_nid:
continue
if not has_name and role == "type":
add_edge(type_nid, tgt, "embeds",
field.start_point[0] + 1)
else:
ctx = "generic_arg" if role == "generic_arg" else "field"
add_edge(type_nid, tgt, "references",
field.start_point[0] + 1, context=ctx)
elif type_body.type == "interface_type":
for elem in type_body.children:
if elem.type != "type_elem":
continue
refs = []
for sub in elem.children:
if sub.is_named:
_go_collect_type_refs(sub, source, False, refs)
for ref_name, role in refs:
tgt = ensure_named_node(ref_name, elem.start_point[0] + 1)
if tgt == type_nid:
continue
if role == "type":
add_edge(type_nid, tgt, "embeds",
elem.start_point[0] + 1)
else:
add_edge(type_nid, tgt, "references",
elem.start_point[0] + 1, context="generic_arg")
return
if t == "import_declaration":
for child in node.children:
if child.type == "import_spec_list":
for spec in child.children:
if spec.type == "import_spec":
path_node = spec.child_by_field_name("path")
if path_node:
raw = _read_text(path_node, source).strip('"')
# Prefix with go_pkg_ so stdlib names (e.g. "context")
# don't collide with local files of the same basename.
tgt_nid = _make_id("go", "pkg", raw)
add_edge(file_nid, tgt_nid, "imports_from", spec.start_point[0] + 1, context="import")
# Track local name (alias or last path segment)
alias = spec.child_by_field_name("name")
local_name = _read_text(alias, source) if alias else raw.split("/")[-1]
if local_name and local_name != "_" and local_name != ".":
go_imported_pkgs.add(local_name)
elif child.type == "import_spec":
path_node = child.child_by_field_name("path")
if path_node:
raw = _read_text(path_node, source).strip('"')
tgt_nid = _make_id("go", "pkg", raw)
add_edge(file_nid, tgt_nid, "imports_from", child.start_point[0] + 1, context="import")
alias = child.child_by_field_name("name")
local_name = _read_text(alias, source) if alias else raw.split("/")[-1]
if local_name and local_name != "_" and local_name != ".":
go_imported_pkgs.add(local_name)
return
for child in node.children:
walk(child)
walk(root)
label_to_nid: dict[str, str] = {}
for n in nodes:
raw = n["label"]
normalised = raw.strip("()").lstrip(".")
label_to_nid[normalised] = n["id"]
seen_call_pairs: set[tuple[str, str]] = set()
raw_calls: list[dict] = []
def walk_calls(node, caller_nid: str) -> None:
if node.type in ("function_declaration", "method_declaration"):
return
if node.type == "call_expression":
func_node = node.child_by_field_name("function")
callee_name: str | None = None
is_member_call: bool = False
if func_node:
if func_node.type == "identifier":
callee_name = _read_text(func_node, source)
elif func_node.type == "selector_expression":
field = func_node.child_by_field_name("field")
operand = func_node.child_by_field_name("operand")
receiver_name = _read_text(operand, source) if operand else ""
# Package-qualified call (e.g. fmt.Println) → allow cross-file resolution.
# Receiver method call (e.g. s.logger.Log) → skip, no import evidence.
is_member_call = receiver_name not in go_imported_pkgs
if field:
callee_name = _read_text(field, source)
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
tgt_nid = label_to_nid.get(callee_name)
if tgt_nid and tgt_nid != caller_nid:
pair = (caller_nid, tgt_nid)
if pair not in seen_call_pairs:
seen_call_pairs.add(pair)
line = node.start_point[0] + 1
edges.append({
"source": caller_nid,
"target": tgt_nid,
"relation": "calls",
"context": "call",
"confidence": "EXTRACTED",
"source_file": str_path,
"source_location": f"L{line}",
"weight": 1.0,
})
elif callee_name:
raw_calls.append({
"caller_nid": caller_nid,
"callee": callee_name,
"is_member_call": is_member_call,
"source_file": str_path,
"source_location": f"L{node.start_point[0] + 1}",
})
for child in node.children:
walk_calls(child, caller_nid)
for caller_nid, body_node in function_bodies:
walk_calls(body_node, caller_nid)
valid_ids = seen_ids
clean_edges = []
for edge in edges:
src, tgt = edge["source"], edge["target"]
if src in valid_ids and (tgt in valid_ids or edge["relation"] in ("imports", "imports_from")):
clean_edges.append(edge)
return {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls}
+206
View File
@@ -0,0 +1,206 @@
"""Json_config extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
from pathlib import Path
from graphify.extractors.base import _file_stem, _make_id, _read_text
_CONFIG_JSON_NAMES = frozenset({
"package.json", "tsconfig.json", "jsconfig.json", "composer.json",
"deno.json", "deno.jsonc", "bower.json", "manifest.json",
"app.json", "now.json", "vercel.json", "angular.json", "nest-cli.json",
"biome.json", "biome.jsonc", "renovate.json", ".babelrc", ".babelrc.json",
".eslintrc.json", ".prettierrc.json", ".prettierrc", "babel.config.json",
})
_CONFIG_JSON_KEYS = frozenset({
"dependencies", "devDependencies", "peerDependencies",
"optionalDependencies", "bundleDependencies", "bundledDependencies",
"extends", "$ref", "$schema", "compilerOptions",
})
def _is_config_json(path: Path, obj_node, source: bytes) -> bool:
"""True if a .json file is a recognized config/manifest worth AST-extracting.
Matches by filename first (cheap), then falls back to a top-level key probe
so arbitrarily-named config files (e.g. ``api.tsconfig.json``,
``foo.eslintrc.json``) are still picked up. Returns False for data JSON so it
is skipped by the structural pass (#1224)."""
name = path.name.casefold()
if name in _CONFIG_JSON_NAMES:
return True
# Common compound config names: *.eslintrc.json, *.prettierrc.json, etc.
if name.endswith((".eslintrc.json", ".prettierrc.json", ".babelrc.json",
"tsconfig.json", "jsconfig.json")):
return True
# Top-level key probe: scan the root object's immediate keys (no deep walk).
for top_key in obj_node.children:
if top_key.type != "pair":
continue
key_node = top_key.child_by_field_name("key")
if key_node is None:
continue
kc = key_node.child_by_field_name("string_content")
text = _read_text(kc, source) if kc else _read_text(key_node, source).strip('"\'')
if text in _CONFIG_JSON_KEYS:
return True
return False
def extract_json(path: Path) -> dict:
"""Extract structure and dependency edges from a *config/manifest* .json file.
Data-shaped JSON (eval fixtures, datasets, GeoJSON, API response dumps) is
deliberately skipped — AST-walking it produced hundreds of orphan key-nodes
and duplicate communities that swamped real structure (#1224). Recognition
is by filename (package.json, tsconfig.json, …) or a top-level key probe
(dependencies / extends / $ref / $schema / compilerOptions)."""
_JSON_MAX_BYTES = 1_048_576 # 1 MiB — skip large fixture dumps / GeoJSON blobs
try:
import tree_sitter_json as tsjson
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree-sitter-json not installed"}
try:
# Bounded read instead of stat()+read() to eliminate TOCTOU (J-1):
# read one byte beyond the limit so we can detect oversized files even
# if the file grows between stat and read.
with path.open("rb") as _f:
source = _f.read(_JSON_MAX_BYTES + 1)
if len(source) > _JSON_MAX_BYTES:
return {"nodes": [], "edges": [], "error": "json file too large to index"}
language = Language(tsjson.language())
parser = Parser(language)
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
# Keys whose string values become imports (package.json dep blocks)
_DEP_KEYS = frozenset({
"dependencies", "devDependencies", "peerDependencies",
"optionalDependencies", "bundleDependencies", "bundledDependencies",
})
def add_node(nid: str, label: str, line: int) -> None:
if nid and nid not in seen_ids:
seen_ids.add(nid)
nodes.append({"id": nid, "label": label, "file_type": "code",
"source_file": str_path, "source_location": f"L{line}"})
def add_edge(src: str, tgt: str, relation: str, line: int,
context: str | None = None) -> None:
if not src or not tgt or src == tgt:
return
edge = {"source": src, "target": tgt, "relation": relation,
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{line}", "weight": 1.0}
if context:
edge["context"] = context
edges.append(edge)
file_nid = _make_id(str(path))
add_node(file_nid, path.name, 1)
def _key_text(pair_node) -> str | None:
"""Extract the string content of a pair's key."""
key_node = pair_node.child_by_field_name("key")
if key_node is None:
return None
if key_node.type == "string":
content = key_node.child_by_field_name("string_content")
if content:
return _read_text(content, source)
# fallback: strip surrounding quotes
raw = _read_text(key_node, source)
return raw.strip('"\'')
return _read_text(key_node, source)
def _val_node(pair_node):
return pair_node.child_by_field_name("value")
def walk_object(obj_node, parent_nid: str, parent_key: str | None,
depth: int, pair_count: list) -> None:
if depth > 6:
return
for child in obj_node.children:
if child.type != "pair":
continue
if pair_count[0] >= 500: # check per-pair so the cap is honoured exactly (J-3)
return
pair_count[0] += 1
key = _key_text(child)
if not key:
continue
key_nid = _make_id(stem, *(([parent_key] if parent_key else []) + [key]))
if not key_nid:
continue
line = child.start_point[0] + 1
add_node(key_nid, key, line)
add_edge(parent_nid, key_nid, "contains", line)
val = _val_node(child)
if val is None:
continue
if val.type == "object":
walk_object(val, key_nid, key, depth + 1, pair_count)
elif val.type == "array":
# For "extends" arrays (tsconfig, eslint): each string element.
# Prefix with "ref_" so external refs don't collide with real
# code/file node IDs that share the same collapsed _make_id (J-4).
for item in val.children:
if item.type == "string":
content = item.child_by_field_name("string_content")
ref = _read_text(content, source) if content else _read_text(item, source).strip('"\'')
if ref:
ref_nid = _make_id("ref", ref)
if ref_nid:
add_edge(key_nid, ref_nid, "extends", line, context="import")
elif val.type == "string":
content = val.child_by_field_name("string_content")
val_text = _read_text(content, source) if content else _read_text(val, source).strip('"\'')
if key == "extends" and val_text:
# Namespace external refs to avoid ID collision with file nodes (J-4)
ref_nid = _make_id("ref", val_text)
if ref_nid:
add_edge(file_nid, ref_nid, "extends", line, context="import")
elif key == "$ref" and val_text:
# Namespace $ref values to prevent edge hijacking into code nodes (J-4)
ref_nid = _make_id("ref", val_text)
if ref_nid:
add_edge(parent_nid, ref_nid, "references", line)
elif parent_key in _DEP_KEYS and val_text:
dep_nid = _make_id(key)
if dep_nid:
add_edge(key_nid, dep_nid, "imports", line, context="import")
# Entry: find root document → object
doc = root
if doc.type == "document" and doc.child_count > 0:
doc = doc.children[0]
if doc.type == "object":
# Only AST-extract recognized config/manifest JSON. Data JSON (fixtures,
# datasets, GeoJSON, API dumps) is skipped so it doesn't explode into
# orphan key-nodes (#1224); it's left to the LLM semantic pass.
if not _is_config_json(path, doc, source):
return {"nodes": [], "edges": [], "skipped": "data json (not a config/manifest)"}
walk_object(doc, file_nid, None, 0, [0])
else:
# Top-level array or scalar => data JSON, never a config/manifest.
return {"nodes": [], "edges": [], "skipped": "data json (non-object root)"}
return {"nodes": nodes, "edges": edges}
+196
View File
@@ -0,0 +1,196 @@
"""Pascal_forms extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
from pathlib import Path
from typing import Any
from graphify.extractors.base import _file_stem, _make_id
def extract_lazarus_form(path: Path) -> dict:
"""Extract component hierarchy from Lazarus .lfm form files.
.lfm is a text-based declarative format for UI component trees, structured as:
object ComponentName: TClassName
PropertyName = Value
OnEvent = HandlerName
object ChildName: TChildClass
...
end
end
Produces nodes for:
- The form file itself
- Each component class encountered (TForm1, TButton, TPanel, ...)
- Event handler names referenced by OnXxx properties
Produces edges for:
- file --contains--> root form class
- parent component --contains--> child component class
- component --references--> event handler (context: "event")
"""
try:
text = path.read_text(encoding="utf-8", errors="replace")
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
import re
str_path = str(path)
stem = _file_stem(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
seen_edge_pairs: set[tuple[str, str, str]] = set()
def add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({
"id": nid, "label": label, "file_type": "code",
"source_file": str_path, "source_location": f"L{line}",
})
def add_edge(
src: str, tgt: str, relation: str, line: int,
context: str | None = None,
) -> None:
key = (src, tgt, relation)
if key in seen_edge_pairs:
return
seen_edge_pairs.add(key)
edge: dict[str, Any] = {
"source": src, "target": tgt, "relation": relation,
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{line}", "weight": 1.0,
}
if context:
edge["context"] = context
edges.append(edge)
file_nid = _make_id(str(path))
add_node(file_nid, path.name, 1)
obj_re = re.compile(r"^\s*object\s+\w+\s*:\s*(\w+)", re.IGNORECASE)
event_re = re.compile(r"^\s*On\w+\s*=\s*(\w+)", re.IGNORECASE)
end_re = re.compile(r"^\s*end\s*$", re.IGNORECASE)
# Stack of node IDs representing the nesting of object...end blocks
stack: list[str] = [file_nid]
for lineno, line in enumerate(text.splitlines(), 1):
m = obj_re.match(line)
if m:
class_name = m.group(1)
nid = _make_id(stem, class_name)
add_node(nid, class_name, lineno)
add_edge(stack[-1], nid, "contains", lineno)
stack.append(nid)
continue
m = event_re.match(line)
if m and len(stack) > 1:
handler = m.group(1)
handler_nid = _make_id(stem, handler)
add_node(handler_nid, f"{handler}()", lineno)
add_edge(stack[-1], handler_nid, "references", lineno, context="event")
continue
if end_re.match(line) and len(stack) > 1:
stack.pop()
return {"nodes": nodes, "edges": edges, "input_tokens": 0, "output_tokens": 0}
def extract_delphi_form(path: Path) -> dict:
"""Extract component hierarchy from Delphi .dfm form files.
.dfm files come in two formats:
- Text (same `object Name: TClassName ... end` syntax as .lfm)
- Binary (starts with a TPF0/FF0A magic header — unreadable as text)
Binary .dfm files are skipped gracefully: an empty result is returned
so the rest of the pipeline is unaffected. Convert binary forms to
text in the Delphi IDE via File → Save As (Text DFM) if you want them
indexed.
Text .dfm files are parsed identically to .lfm: component containment
(`contains`) and event handler references (`references`, context "event").
"""
try:
raw = path.read_bytes()
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
# Detect binary DFM: Delphi binary resource streams start with FF 0A
if raw[:2] == b"\xff\x0a":
return {
"nodes": [], "edges": [],
"error": f"binary DFM (convert to text in Delphi IDE to index): {path.name}",
}
# Text DFM — delegate to the shared form parser (same syntax as .lfm)
try:
text = raw.decode("utf-8", errors="replace")
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
import re
str_path = str(path)
stem = _file_stem(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
seen_edge_pairs: set[tuple[str, str, str]] = set()
def add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({
"id": nid, "label": label, "file_type": "code",
"source_file": str_path, "source_location": f"L{line}",
})
def add_edge(
src: str, tgt: str, relation: str, line: int,
context: str | None = None,
) -> None:
key = (src, tgt, relation)
if key in seen_edge_pairs:
return
seen_edge_pairs.add(key)
edge: dict[str, Any] = {
"source": src, "target": tgt, "relation": relation,
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{line}", "weight": 1.0,
}
if context:
edge["context"] = context
edges.append(edge)
file_nid = _make_id(str(path))
add_node(file_nid, path.name, 1)
obj_re = re.compile(r"^\s*object\s+\w+\s*:\s*(\w+)", re.IGNORECASE)
event_re = re.compile(r"^\s*On\w+\s*=\s*(\w+)", re.IGNORECASE)
end_re = re.compile(r"^\s*end\s*$", re.IGNORECASE)
stack: list[str] = [file_nid]
for lineno, line in enumerate(text.splitlines(), 1):
m = obj_re.match(line)
if m:
class_name = m.group(1)
nid = _make_id(stem, class_name)
add_node(nid, class_name, lineno)
add_edge(stack[-1], nid, "contains", lineno)
stack.append(nid)
continue
m = event_re.match(line)
if m and len(stack) > 1:
handler = m.group(1)
handler_nid = _make_id(stem, handler)
add_node(handler_nid, f"{handler}()", lineno)
add_edge(stack[-1], handler_nid, "references", lineno, context="event")
continue
if end_re.match(line) and len(stack) > 1:
stack.pop()
return {"nodes": nodes, "edges": edges, "input_tokens": 0, "output_tokens": 0}
+496
View File
@@ -0,0 +1,496 @@
"""Powershell extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
import re
from pathlib import Path
from typing import Any
from graphify.extractors.base import _file_stem, _make_id, _read_text
def extract_powershell(path: Path) -> dict:
"""Extract functions, classes, methods, and using statements from a .ps1 file."""
try:
import tree_sitter_powershell as tsps
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree_sitter_powershell not installed"}
try:
language = Language(tsps.language())
parser = Parser(language)
source = path.read_bytes()
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
function_bodies: list[tuple[str, Any]] = []
def add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({"id": nid, "label": label, "file_type": "code",
"source_file": str_path, "source_location": f"L{line}"})
def add_edge(src: str, tgt: str, relation: str, line: int,
confidence: str = "EXTRACTED", weight: float = 1.0,
context: str | None = None) -> None:
edge = {"source": src, "target": tgt, "relation": relation,
"confidence": confidence, "source_file": str_path,
"source_location": f"L{line}", "weight": weight}
if context:
edge["context"] = context
edges.append(edge)
file_nid = _make_id(str(path))
add_node(file_nid, path.name, 1)
_PS_SKIP = frozenset({
"using", "return", "if", "else", "elseif", "foreach", "for",
"while", "do", "switch", "try", "catch", "finally", "throw",
"break", "continue", "exit", "param", "begin", "process", "end",
# Import commands — handled as import edges, not function calls
"import-module",
})
def _find_script_block_body(node):
for child in node.children:
if child.type == "script_block":
for sc in child.children:
if sc.type == "script_block_body":
return sc
return child
return None
def ensure_named_node(name: str, line: int) -> str:
nid = _make_id(stem, name)
if nid in seen_ids:
return nid
nid = _make_id(name)
if nid not in seen_ids:
# The name isn't defined in this file, so this is a cross-file reference
# (e.g. a `Thing` type annotation imported from another module). Emit a
# SOURCELESS stub — like the inheritance-base path below — so the
# corpus-level rewire can collapse it onto the real definition. A sourced
# stub here makes _disambiguate_colliding_node_ids bake the referencing
# file's path (with extension) into the id and blocks the rewire, which is
# the phantom-duplicate-node bug (#1402).
seen_ids.add(nid)
nodes.append({
"id": nid,
"label": name,
"file_type": "code",
"source_file": "",
"source_location": "",
"origin_file": str_path,
})
return nid
def _ps_type_name(type_literal_node) -> str | None:
"""Drill into a type_literal node and return the inner type_identifier text."""
if type_literal_node is None:
return None
for spec in type_literal_node.children:
if spec.type != "type_spec":
continue
for tname in spec.children:
if tname.type != "type_name":
continue
for tid in tname.children:
if tid.type == "type_identifier":
return _read_text(tid, source)
return None
def walk(node, parent_class_nid: str | None = None) -> None:
t = node.type
if t == "function_statement":
name_node = next((c for c in node.children if c.type == "function_name"), None)
if name_node:
func_name = _read_text(name_node, source)
line = node.start_point[0] + 1
func_nid = _make_id(stem, func_name)
add_node(func_nid, f"{func_name}()", line)
add_edge(file_nid, func_nid, "contains", line)
body = _find_script_block_body(node)
if body:
function_bodies.append((func_nid, body))
# Also walk the body during the main pass so that
# Import-Module / dot-source inside functions emit
# file-level imports_from edges (#1331).
walk(body, parent_class_nid)
return
if t == "class_statement":
name_node = next((c for c in node.children if c.type == "simple_name"), None)
if name_node:
class_name = _read_text(name_node, source)
line = node.start_point[0] + 1
class_nid = _make_id(stem, class_name)
add_node(class_nid, class_name, line)
add_edge(file_nid, class_nid, "contains", line)
# Base type(s) after ':'. PowerShell has no syntactic base vs
# interface split, so (matching the C# convention) treat the
# first base as the superclass (inherits) and the rest as
# interfaces (implements). Bases are the simple_name children
# after the ':' token.
colon_seen = False
base_index = 0
for child in node.children:
if child.type == ":":
colon_seen = True
elif colon_seen and child.type == "simple_name":
base_nid = ensure_named_node(_read_text(child, source), line)
if base_nid != class_nid:
rel = "inherits" if base_index == 0 else "implements"
add_edge(class_nid, base_nid, rel, line)
base_index += 1
for child in node.children:
walk(child, parent_class_nid=class_nid)
return
if t == "class_property_definition" and parent_class_nid:
type_literal = next((c for c in node.children if c.type == "type_literal"), None)
type_name = _ps_type_name(type_literal)
if type_name:
line = node.start_point[0] + 1
target_nid = ensure_named_node(type_name, line)
if target_nid != parent_class_nid:
add_edge(parent_class_nid, target_nid, "references",
line, context="field")
return
if t == "class_method_definition":
name_node = next((c for c in node.children if c.type == "simple_name"), None)
if name_node:
method_name = _read_text(name_node, source)
line = node.start_point[0] + 1
if parent_class_nid:
method_nid = _make_id(parent_class_nid, method_name)
add_node(method_nid, f".{method_name}()", line)
add_edge(parent_class_nid, method_nid, "method", line)
else:
method_nid = _make_id(stem, method_name)
add_node(method_nid, f"{method_name}()", line)
add_edge(file_nid, method_nid, "contains", line)
# Return type: type_literal sibling of simple_name
return_type_literal = next(
(c for c in node.children if c.type == "type_literal"), None)
return_type_name = _ps_type_name(return_type_literal)
if return_type_name:
target_nid = ensure_named_node(return_type_name, line)
if target_nid != method_nid:
add_edge(method_nid, target_nid, "references",
line, context="return_type")
# Parameter types: class_method_parameter_list
param_list = next(
(c for c in node.children if c.type == "class_method_parameter_list"), None)
if param_list is not None:
for p in param_list.children:
if p.type != "class_method_parameter":
continue
ptype_literal = next(
(c for c in p.children if c.type == "type_literal"), None)
ptype_name = _ps_type_name(ptype_literal)
if not ptype_name:
continue
p_line = p.start_point[0] + 1
target_nid = ensure_named_node(ptype_name, p_line)
if target_nid != method_nid:
add_edge(method_nid, target_nid, "references",
p_line, context="parameter_type")
body = _find_script_block_body(node)
if body:
function_bodies.append((method_nid, body))
return
if t == "command":
# Dot-sourcing: `. ./Shared.psm1`
# Uses command_invokation_operator '.' + command_name_expr (not command_name)
invoke_op = next(
(c for c in node.children if c.type == "command_invokation_operator"), None
)
if invoke_op is not None and _read_text(invoke_op, source).strip() == ".":
name_expr = next(
(c for c in node.children if c.type == "command_name_expr"), None
)
if name_expr is not None:
name_node = next(
(c for c in name_expr.children if c.type == "command_name"), None
)
if name_node:
raw_path = _read_text(name_node, source)
# Strip relative path prefix (./ or .\ or just the dot)
module_stem = re.sub(r'^[./\\]+', '', raw_path)
# Drop extension to get bare module name
module_stem = re.sub(r'\.[^.]+$', '', module_stem).replace('\\', '/')
module_name = module_stem.split('/')[-1]
if module_name:
add_edge(file_nid, _make_id(module_name), "imports_from",
node.start_point[0] + 1)
return
cmd_name_node = next((c for c in node.children if c.type == "command_name"), None)
if cmd_name_node:
cmd_text = _read_text(cmd_name_node, source).lower()
if cmd_text == "using":
tokens = []
for child in node.children:
if child.type == "command_elements":
for el in child.children:
if el.type == "generic_token":
tokens.append(_read_text(el, source))
module_tokens = [t for t in tokens
if t.lower() not in ("namespace", "module", "assembly")]
if module_tokens:
module_name = module_tokens[-1].split(".")[-1]
add_edge(file_nid, _make_id(module_name), "imports_from",
node.start_point[0] + 1)
elif cmd_text == "import-module":
# Collect generic_token args; skip command_parameter flags like -Name
# The module name is the first generic_token (or the one after -Name)
module_name: str | None = None
expect_name = False
for child in node.children:
if child.type != "command_elements":
continue
for el in child.children:
if el.type == "command_parameter":
param_text = _read_text(el, source).lstrip("-").lower()
expect_name = param_text in ("name", "n")
elif el.type == "generic_token":
token = _read_text(el, source)
if module_name is None or expect_name:
module_name = token
expect_name = False
if module_name:
# Strip extension; keep only the stem for the node ID
bare = re.sub(r'\.[^.]+$', '', module_name).split('/')[-1].split('\\')[-1]
if bare:
add_edge(file_nid, _make_id(bare), "imports_from",
node.start_point[0] + 1)
return
for child in node.children:
walk(child, parent_class_nid)
walk(root)
label_to_nid = {n["label"].strip("()").lstrip(".").lower(): n["id"] for n in nodes}
seen_call_pairs: set[tuple[str, str]] = set()
raw_calls: list[dict] = []
def walk_calls(node, caller_nid: str) -> None:
if node.type in ("function_statement", "class_statement"):
return
if node.type == "command":
cmd_name_node = next((c for c in node.children if c.type == "command_name"), None)
if cmd_name_node:
cmd_text = _read_text(cmd_name_node, source)
if cmd_text.lower() not in _PS_SKIP:
tgt_nid = label_to_nid.get(cmd_text.lower())
if tgt_nid and tgt_nid != caller_nid:
pair = (caller_nid, tgt_nid)
if pair not in seen_call_pairs:
seen_call_pairs.add(pair)
add_edge(caller_nid, tgt_nid, "calls",
node.start_point[0] + 1,
confidence="EXTRACTED", weight=1.0)
elif cmd_text:
raw_calls.append({
"caller_nid": caller_nid,
"callee": cmd_text,
"is_member_call": False,
"source_file": str_path,
"source_location": f"L{node.start_point[0] + 1}",
})
for child in node.children:
walk_calls(child, caller_nid)
for caller_nid, body_node in function_bodies:
walk_calls(body_node, caller_nid)
clean_edges = [e for e in edges if e["source"] in seen_ids and
(e["target"] in seen_ids or e["relation"] in ("imports_from", "imports"))]
return {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls}
_PSD1_IMPORT_KEYS = frozenset({"RootModule", "NestedModules", "RequiredModules"})
def _psd1_collect_string_literals(node, source: bytes) -> list[str]:
"""Recursively collect all string_literal text values under *node*."""
results: list[str] = []
def _walk(n) -> None:
if n.type == "string_literal":
raw = source[n.start_byte:n.end_byte].decode(errors="replace")
# Strip surrounding quote chars (' or ")
results.append(raw.strip("'\""))
return
for child in n.children:
_walk(child)
_walk(node)
return results
def _psd1_module_name(raw: str) -> str:
"""Derive a bare module name from a raw string value.
e.g. 'MyModule.psm1' → 'MyModule', './sub/Util.psm1' → 'Util', 'PSReadLine' → 'PSReadLine'
"""
# Strip path prefix and extension
name = raw.replace("\\", "/").split("/")[-1]
name = re.sub(r"\.[^.]+$", "", name) # remove last extension
return name.strip()
def extract_powershell_manifest(path: Path) -> dict:
"""Extract module dependency edges from a PowerShell .psd1 manifest file.
.psd1 files are PowerShell data hashtables, not scripts. tree-sitter-powershell
parses them correctly (they are syntactically valid PS). We walk the AST looking
for RootModule, NestedModules, and RequiredModules keys and emit imports_from
edges for every referenced module.
RequiredModules supports two forms:
- Simple string: 'PSReadLine'
- Module specification: @{ ModuleName = 'Pester'; ModuleVersion = '5.0' }
For the hashtable form we only follow the ModuleName key.
"""
try:
import tree_sitter_powershell as tsps
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree_sitter_powershell not installed"}
try:
language = Language(tsps.language())
parser = Parser(language)
source = path.read_bytes()
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
def add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({"id": nid, "label": label, "file_type": "code",
"source_file": str_path, "source_location": f"L{line}"})
def add_import_edge(src: str, module_raw: str, line: int) -> None:
name = _psd1_module_name(module_raw)
if not name:
return
tgt_nid = _make_id(name)
edges.append({
"source": src,
"target": tgt_nid,
"relation": "imports_from",
"confidence": "EXTRACTED",
"source_file": str_path,
"source_location": f"L{line}",
"weight": 1.0,
"context": "import",
})
file_nid = _make_id(str(path))
add_node(file_nid, path.name, 1)
def walk_manifest(node) -> None:
"""Walk the AST and emit edges for import-relevant hash_entry nodes."""
if node.type != "hash_entry":
for child in node.children:
walk_manifest(child)
return
# Identify the key
key_node = next((c for c in node.children if c.type == "key_expression"), None)
if key_node is None:
return
key_text = source[key_node.start_byte:key_node.end_byte].decode(errors="replace").strip()
if key_text not in _PSD1_IMPORT_KEYS:
# Still recurse in case there are nested hashes (e.g. ModuleVersion entries
# contain sub-hashes, but we only care about top-level keys for imports)
return
line = node.start_point[0] + 1
value_node = next((c for c in node.children if c.type == "pipeline"), None)
if value_node is None:
return
if key_text == "RootModule":
# Value is a single string
strings = _psd1_collect_string_literals(value_node, source)
for s in strings:
add_import_edge(file_nid, s, line)
elif key_text == "NestedModules":
# Value is a string or @('a', 'b', ...) array — collect all string literals
strings = _psd1_collect_string_literals(value_node, source)
for s in strings:
add_import_edge(file_nid, s, line)
elif key_text == "RequiredModules":
# Two forms:
# 1) 'SimpleModule' — direct string literals in the array
# 2) @{ ModuleName = 'Foo'; ModuleVersion = '2.0' } — use ModuleName only
#
# Strategy: walk the value for hash_entry nodes whose key is 'ModuleName';
# collect their string values. For the remaining string_literal nodes that
# are NOT inside a hash_entry subtree, treat them as simple module names.
module_name_strings: list[str] = []
inside_hash_entries: set[int] = set() # byte offsets of handled strings
def find_modulename_entries(n) -> None:
if n.type == "hash_entry":
sub_key = next((c for c in n.children if c.type == "key_expression"), None)
if sub_key is not None:
sk_text = source[sub_key.start_byte:sub_key.end_byte].decode(errors="replace").strip()
# Collect strings inside *all* sub-keys so we can exclude them
for c in n.children:
if c.type == "pipeline":
for s_node in _collect_string_nodes(c):
inside_hash_entries.add(s_node.start_byte)
if sk_text == "ModuleName":
for c in n.children:
if c.type == "pipeline":
for s in _psd1_collect_string_literals(c, source):
module_name_strings.append(s)
return # don't recurse further into this hash_entry
for child in n.children:
find_modulename_entries(child)
def _collect_string_nodes(n):
"""Return all string_literal nodes in subtree."""
if n.type == "string_literal":
yield n
return
for child in n.children:
yield from _collect_string_nodes(child)
find_modulename_entries(value_node)
# Now gather direct string literals not inside hash entries
direct_strings: list[str] = []
for s_node in _collect_string_nodes(value_node):
if s_node.start_byte not in inside_hash_entries:
raw = source[s_node.start_byte:s_node.end_byte].decode(errors="replace")
direct_strings.append(raw.strip("'\""))
for s in direct_strings + module_name_strings:
add_import_edge(file_nid, s, line)
walk_manifest(root)
return {"nodes": nodes, "edges": edges, "raw_calls": []}
+410
View File
@@ -0,0 +1,410 @@
"""Rust extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
from pathlib import Path
from graphify.extractors.base import _LANGUAGE_BUILTIN_GLOBALS, _file_stem, _make_id, _read_text
def _rust_collect_type_refs(node, source: bytes, generic: bool, out: list[tuple[str, str]]) -> None:
"""Walk a Rust type expression; append (name, role) tuples."""
if node is None:
return
t = node.type
if t == "primitive_type":
return
if t == "type_identifier":
text = _read_text(node, source)
if text:
out.append((text, "generic_arg" if generic else "type"))
return
if t == "scoped_type_identifier":
text = _read_text(node, source).rsplit("::", 1)[-1]
if text:
out.append((text, "generic_arg" if generic else "type"))
return
if t == "generic_type":
name_node = node.child_by_field_name("type")
if name_node is None:
for c in node.children:
if c.type in ("type_identifier", "scoped_type_identifier"):
name_node = c
break
if name_node is not None:
text = _read_text(name_node, source).rsplit("::", 1)[-1]
if text:
out.append((text, "generic_arg" if generic else "type"))
for c in node.children:
if c.type == "type_arguments":
for arg in c.children:
if arg.is_named:
_rust_collect_type_refs(arg, source, True, out)
return
if t in ("reference_type", "pointer_type", "array_type", "tuple_type", "slice_type"):
for c in node.children:
if c.is_named:
_rust_collect_type_refs(c, source, generic, out)
return
if node.is_named:
for c in node.children:
if c.is_named:
_rust_collect_type_refs(c, source, generic, out)
_RUST_TRAIT_METHOD_BLOCKLIST: frozenset[str] = frozenset({
"new", "default", "parse", "from_str", "now", "clone", "into", "from",
"to_string", "to_owned", "len", "is_empty", "iter", "next", "build",
"start", "run", "init", "app", "get", "set", "push", "pop", "insert",
"remove", "contains", "collect", "map", "filter", "unwrap", "expect",
"ok", "err", "some", "none", "send", "recv", "lock", "read", "write",
})
def extract_rust(path: Path) -> dict:
"""Extract functions, structs, enums, traits, impl methods, and use declarations from a .rs file."""
try:
import tree_sitter_rust as tsrust
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree-sitter-rust not installed"}
try:
language = Language(tsrust.language())
parser = Parser(language)
source = path.read_bytes()
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = _file_stem(path)
str_path = str(path)
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
function_bodies: list[tuple[str, object]] = []
def add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({
"id": nid,
"label": label,
"file_type": "code",
"source_file": str_path,
"source_location": f"L{line}",
})
def add_edge(src: str, tgt: str, relation: str, line: int,
confidence: str = "EXTRACTED", weight: float = 1.0,
context: str | None = None) -> None:
edge = {
"source": src,
"target": tgt,
"relation": relation,
"confidence": confidence,
"source_file": str_path,
"source_location": f"L{line}",
"weight": weight,
}
if context:
edge["context"] = context
edges.append(edge)
file_nid = _make_id(str(path))
add_node(file_nid, path.name, 1)
def ensure_named_node(name: str, line: int) -> str:
nid = _make_id(stem, name)
if nid in seen_ids:
return nid
nid = _make_id(name)
if nid not in seen_ids:
# The name isn't defined in this file, so this is a cross-file reference
# (e.g. a `Thing` type annotation imported from another module). Emit a
# SOURCELESS stub — like the inheritance-base path below — so the
# corpus-level rewire can collapse it onto the real definition. A sourced
# stub here makes _disambiguate_colliding_node_ids bake the referencing
# file's path (with extension) into the id and blocks the rewire, which is
# the phantom-duplicate-node bug (#1402).
seen_ids.add(nid)
nodes.append({
"id": nid,
"label": name,
"file_type": "code",
"source_file": "",
"source_location": "",
"origin_file": str_path,
})
return nid
def emit_param_return_refs(func_node, func_nid: str, line: int) -> None:
params = func_node.child_by_field_name("parameters")
if params is not None:
for p in params.children:
if p.type != "parameter":
continue
type_node = p.child_by_field_name("type")
refs: list[tuple[str, str]] = []
_rust_collect_type_refs(type_node, source, False, refs)
for ref_name, role in refs:
ctx = "generic_arg" if role == "generic_arg" else "parameter_type"
tgt = ensure_named_node(ref_name, line)
if tgt != func_nid:
add_edge(func_nid, tgt, "references", line, context=ctx)
return_type = func_node.child_by_field_name("return_type")
if return_type is not None:
refs = []
_rust_collect_type_refs(return_type, source, False, refs)
for ref_name, role in refs:
ctx = "generic_arg" if role == "generic_arg" else "return_type"
tgt = ensure_named_node(ref_name, line)
if tgt != func_nid:
add_edge(func_nid, tgt, "references", line, context=ctx)
def walk(node, parent_impl_nid: str | None = None) -> None:
t = node.type
if t == "function_item":
name_node = node.child_by_field_name("name")
if name_node:
func_name = _read_text(name_node, source)
line = node.start_point[0] + 1
if parent_impl_nid:
func_nid = _make_id(parent_impl_nid, func_name)
add_node(func_nid, f".{func_name}()", line)
add_edge(parent_impl_nid, func_nid, "method", line)
else:
func_nid = _make_id(stem, func_name)
add_node(func_nid, f"{func_name}()", line)
add_edge(file_nid, func_nid, "contains", line)
emit_param_return_refs(node, func_nid, line)
body = node.child_by_field_name("body")
if body:
function_bodies.append((func_nid, body))
return
if t in ("struct_item", "enum_item", "trait_item"):
name_node = node.child_by_field_name("name")
if name_node:
item_name = _read_text(name_node, source)
line = node.start_point[0] + 1
item_nid = _make_id(stem, item_name)
add_node(item_nid, item_name, line)
add_edge(file_nid, item_nid, "contains", line)
if t == "trait_item":
for c in node.children:
if c.type != "trait_bounds":
continue
for sub in c.children:
if not sub.is_named:
continue
refs: list[tuple[str, str]] = []
_rust_collect_type_refs(sub, source, False, refs)
for idx, (ref_name, _role) in enumerate(refs):
tgt = ensure_named_node(ref_name, line)
if tgt == item_nid:
continue
rel = "inherits" if idx == 0 else "references"
if rel == "inherits":
add_edge(item_nid, tgt, "inherits", line)
else:
add_edge(item_nid, tgt, "references", line,
context="generic_arg")
if t == "struct_item":
for c in node.children:
if c.type != "field_declaration_list":
continue
for field in c.children:
if field.type != "field_declaration":
continue
type_node = field.child_by_field_name("type")
if type_node is None:
for fc in field.children:
if fc.type in ("type_identifier", "generic_type",
"scoped_type_identifier",
"reference_type", "primitive_type"):
type_node = fc
break
refs = []
_rust_collect_type_refs(type_node, source, False, refs)
for ref_name, role in refs:
ctx = "generic_arg" if role == "generic_arg" else "field"
tgt = ensure_named_node(ref_name, field.start_point[0] + 1)
if tgt != item_nid:
add_edge(item_nid, tgt, "references",
field.start_point[0] + 1, context=ctx)
# Tuple structs (`struct Wrapper(pub Logger, Config);`) nest their
# positional field types directly under ordered_field_declaration_list
# with no field_declaration wrapper -- the same shape handled for tuple
# enum variants below. Without this branch these field type references
# are silently dropped.
for c in node.children:
if c.type != "ordered_field_declaration_list":
continue
fline = c.start_point[0] + 1
for tc in c.children:
if tc.type not in ("type_identifier", "generic_type",
"scoped_type_identifier", "reference_type",
"primitive_type", "tuple_type", "array_type"):
continue
refs = []
_rust_collect_type_refs(tc, source, False, refs)
for ref_name, role in refs:
ctx = "generic_arg" if role == "generic_arg" else "field"
tgt = ensure_named_node(ref_name, fline)
if tgt != item_nid:
add_edge(item_nid, tgt, "references", fline, context=ctx)
if t == "enum_item":
# Variant payload types nest under enum_variant_list ->
# enum_variant -> ordered_field_declaration_list (tuple variant,
# `Click(Logger)`) | field_declaration_list (struct variant,
# `Resize { size: Dim }`). Neither was traversed, so every
# enum-variant type reference was silently dropped.
_TYPE_NODES = ("type_identifier", "generic_type",
"scoped_type_identifier", "reference_type",
"primitive_type", "tuple_type", "array_type")
def _emit_enum_type(type_node, at_line):
if type_node is None:
return
refs2: list[tuple[str, str]] = []
_rust_collect_type_refs(type_node, source, False, refs2)
for ref_name, role in refs2:
ctx = "generic_arg" if role == "generic_arg" else "field"
tgt = ensure_named_node(ref_name, at_line)
if tgt != item_nid:
add_edge(item_nid, tgt, "references", at_line, context=ctx)
for c in node.children:
if c.type != "enum_variant_list":
continue
for variant in c.children:
if variant.type != "enum_variant":
continue
vline = variant.start_point[0] + 1
for vc in variant.children:
if vc.type == "ordered_field_declaration_list":
for tc in vc.children:
if tc.type in _TYPE_NODES:
_emit_enum_type(tc, vline)
elif vc.type == "field_declaration_list":
for field in vc.children:
if field.type != "field_declaration":
continue
type_node = field.child_by_field_name("type")
_emit_enum_type(type_node, field.start_point[0] + 1)
return
if t == "impl_item":
type_node = node.child_by_field_name("type")
trait_node = node.child_by_field_name("trait")
impl_nid: str | None = None
if type_node:
type_name = _read_text(type_node, source).strip()
impl_nid = _make_id(stem, type_name)
add_node(impl_nid, type_name, node.start_point[0] + 1)
if trait_node is not None and impl_nid is not None:
refs: list[tuple[str, str]] = []
_rust_collect_type_refs(trait_node, source, False, refs)
for idx, (ref_name, _role) in enumerate(refs):
tgt = ensure_named_node(ref_name, node.start_point[0] + 1)
if tgt == impl_nid:
continue
if idx == 0:
add_edge(impl_nid, tgt, "implements", node.start_point[0] + 1)
else:
add_edge(impl_nid, tgt, "references", node.start_point[0] + 1,
context="generic_arg")
body = node.child_by_field_name("body")
if body:
for child in body.children:
walk(child, parent_impl_nid=impl_nid)
return
if t == "use_declaration":
arg = node.child_by_field_name("argument")
if arg:
raw = _read_text(arg, source)
clean = raw.split("{")[0].rstrip(":").rstrip("*").rstrip(":")
module_name = clean.split("::")[-1].strip()
if module_name:
tgt_nid = _make_id(module_name)
add_edge(file_nid, tgt_nid, "imports_from", node.start_point[0] + 1, context="import")
return
for child in node.children:
walk(child, parent_impl_nid=None)
walk(root)
label_to_nid: dict[str, str] = {}
for n in nodes:
raw = n["label"]
normalised = raw.strip("()").lstrip(".")
label_to_nid[normalised] = n["id"]
seen_call_pairs: set[tuple[str, str]] = set()
raw_calls: list[dict] = []
def walk_calls(node, caller_nid: str) -> None:
if node.type == "function_item":
return
if node.type == "call_expression":
func_node = node.child_by_field_name("function")
callee_name: str | None = None
is_member_call: bool = False
is_scoped_call: bool = False
if func_node:
if func_node.type == "identifier":
callee_name = _read_text(func_node, source)
elif func_node.type == "field_expression":
is_member_call = True
field = func_node.child_by_field_name("field")
if field:
callee_name = _read_text(field, source)
elif func_node.type == "scoped_identifier":
# Type::method() — still allow in-file EXTRACTED match, but
# skip cross-file resolution: bare last-segment lookup ignores
# crate boundaries and produces spurious INFERRED edges (#908).
is_scoped_call = True
name = func_node.child_by_field_name("name")
if name:
callee_name = _read_text(name, source)
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
tgt_nid = label_to_nid.get(callee_name)
if tgt_nid and tgt_nid != caller_nid:
pair = (caller_nid, tgt_nid)
if pair not in seen_call_pairs:
seen_call_pairs.add(pair)
line = node.start_point[0] + 1
edges.append({
"source": caller_nid,
"target": tgt_nid,
"relation": "calls",
"context": "call",
"confidence": "EXTRACTED",
"source_file": str_path,
"source_location": f"L{line}",
"weight": 1.0,
})
elif not is_scoped_call and callee_name.lower() not in _RUST_TRAIT_METHOD_BLOCKLIST:
raw_calls.append({
"caller_nid": caller_nid,
"callee": callee_name,
"is_member_call": is_member_call,
"source_file": str_path,
"source_location": f"L{node.start_point[0] + 1}",
})
for child in node.children:
walk_calls(child, caller_nid)
for caller_nid, body_node in function_bodies:
walk_calls(body_node, caller_nid)
valid_ids = seen_ids
clean_edges = []
for edge in edges:
src, tgt = edge["source"], edge["target"]
if src in valid_ids and (tgt in valid_ids or edge["relation"] in ("imports", "imports_from")):
clean_edges.append(edge)
return {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls}
+81
View File
@@ -0,0 +1,81 @@
"""Sln extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
import re
from pathlib import Path
from graphify.extractors.base import _make_id
def extract_sln(path: Path) -> dict:
"""Extract projects and inter-project dependencies from a .sln file."""
try:
src = path.read_text(encoding="utf-8", errors="replace")
except OSError:
return {"nodes": [], "edges": [], "error": f"cannot read {path}"}
file_nid = _make_id(str(path))
str_path = str(path)
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str_path, "source_location": None}]
edges: list[dict] = []
seen_ids: set[str] = set()
seen_ids.add(file_nid)
_PROJECT_RE = re.compile(
r'Project\("[^"]*"\)\s*=\s*"([^"]+)"\s*,\s*"([^"]+)"\s*,\s*"([^"]*)"'
)
_DEP_RE = re.compile(r'\{([0-9a-fA-F-]+)\}\s*=\s*\{([0-9a-fA-F-]+)\}')
guid_to_nid: dict[str, str] = {}
for m in _PROJECT_RE.finditer(src):
proj_name = m.group(1)
proj_path = m.group(2).replace("\\", "/")
proj_guid = m.group(3).strip("{}")
try:
abs_proj = str((path.parent / proj_path).resolve())
except Exception:
abs_proj = proj_path
proj_nid = _make_id(abs_proj)
if proj_nid and proj_nid not in seen_ids:
seen_ids.add(proj_nid)
nodes.append({"id": proj_nid, "label": proj_name,
"file_type": "code", "source_file": abs_proj,
"source_location": None})
edges.append({"source": file_nid, "target": proj_nid,
"relation": "contains", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})
if proj_guid:
guid_to_nid[proj_guid.lower()] = proj_nid
in_dep_section = False
current_proj_guid: str | None = None
_PROJECT_LINE_RE = re.compile(r'Project\("[^"]*"\)\s*=\s*"[^"]+"\s*,\s*"[^"]+"\s*,\s*"\{([^}]+)\}"')
for line in src.splitlines():
proj_line_m = _PROJECT_LINE_RE.search(line)
if proj_line_m:
current_proj_guid = proj_line_m.group(1).lower()
continue
if line.strip() == "EndProject":
current_proj_guid = None
continue
if "ProjectSection(ProjectDependencies)" in line:
in_dep_section = True
continue
if in_dep_section and "EndProjectSection" in line:
in_dep_section = False
continue
if in_dep_section and current_proj_guid:
dep_m = _DEP_RE.search(line)
if dep_m:
to_guid = dep_m.group(1).lower()
from_nid = guid_to_nid.get(current_proj_guid)
to_nid = guid_to_nid.get(to_guid)
if from_nid and to_nid and from_nid != to_nid:
edges.append({"source": from_nid, "target": to_nid,
"relation": "imports", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})
return {"nodes": nodes, "edges": edges}
+276
View File
@@ -0,0 +1,276 @@
"""Sql extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
import re
from pathlib import Path
from graphify.extractors.base import _file_stem, _make_id
def extract_sql(path: Path, content: str | bytes | None = None) -> dict:
"""Extract tables, views, functions, and relationships from .sql files via tree-sitter."""
try:
import tree_sitter_sql as tssql
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree_sitter_sql not installed. Run: pip install tree-sitter-sql"}
try:
language = Language(tssql.language())
parser = Parser(language)
source = (
content.encode("utf-8") if isinstance(content, str)
else content if content is not None
else path.read_bytes()
)
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
stem = _file_stem(path)
str_path = str(path)
file_nid = _make_id(str_path)
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str_path, "source_location": None}]
edges: list[dict] = []
seen_ids: set[str] = {file_nid}
table_nids: dict[str, str] = {} # name → nid for reference resolution
def _read(n) -> str:
return source[n.start_byte:n.end_byte].decode("utf-8", errors="replace")
def _obj_name(n) -> str | None:
for c in n.children:
if c.type == "object_reference":
return _read(c)
return None
def _add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({"id": nid, "label": label, "file_type": "code",
"source_file": str_path, "source_location": f"L{line}"})
edges.append({"source": file_nid, "target": nid, "relation": "contains",
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{line}", "weight": 1.0})
def _add_edge(src: str, tgt: str, relation: str, line: int) -> None:
edges.append({"source": src, "target": tgt, "relation": relation,
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{line}", "weight": 1.0})
def walk(node) -> None:
t = node.type
line = node.start_point[0] + 1
if t == "create_table":
name = _obj_name(node)
if name:
nid = _make_id(stem, name)
_add_node(nid, name, line)
table_nids[name.lower()] = nid
# Foreign key REFERENCES
for col in node.children:
if col.type == "column_definitions":
has_error = any(cd.type == "ERROR" for cd in col.children)
seen_refs: set[str] = set()
for cd in col.children:
if cd.type == "column_definition":
# Inline column-level REFERENCES
ref_name: str | None = None
found_ref = False
for cc in cd.children:
if cc.type == "keyword_references":
found_ref = True
elif found_ref and cc.type == "object_reference":
ref_name = _read(cc)
break
if ref_name:
ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name)
_add_edge(nid, ref_nid, "references", line)
seen_refs.add(ref_name.lower())
elif cd.type == "constraints":
# Table-level FOREIGN KEY ... REFERENCES ... constraints
for constraint in cd.children:
if constraint.type != "constraint":
continue
ref_name = None
found_ref = False
for cc in constraint.children:
if cc.type == "keyword_references":
found_ref = True
elif found_ref and cc.type == "object_reference":
ref_name = _read(cc)
break
if ref_name:
ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name)
_add_edge(nid, ref_nid, "references", line)
seen_refs.add(ref_name.lower())
if has_error:
# Dialect-specific syntax (e.g. Firebird COMPUTED BY) causes ERROR
# nodes that make the parser drop the trailing constraints block.
# Regex-scan the raw column_definitions text as fallback.
col_text = _read(col)
for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", col_text, re.IGNORECASE):
ref_name = rm.group(1)
if ref_name.lower() not in seen_refs:
ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name)
_add_edge(nid, ref_nid, "references", line)
seen_refs.add(ref_name.lower())
elif t == "create_view":
name = _obj_name(node)
if name:
nid = _make_id(stem, name)
_add_node(nid, name, line)
table_nids[name.lower()] = nid
# FROM/JOIN table references inside view body
_walk_from_refs(node, nid, line)
elif t == "create_function":
name = _obj_name(node)
if name:
nid = _make_id(stem, name)
_add_node(nid, f"{name}()", line)
_walk_from_refs(node, nid, line)
elif t == "create_procedure":
name = _obj_name(node)
if name:
nid = _make_id(stem, name)
_add_node(nid, f"{name}()", line)
_walk_from_refs(node, nid, line)
elif t == "alter_table":
name = _obj_name(node)
if name:
src_nid = table_nids.get(name.lower())
if not src_nid:
src_nid = _make_id(stem, name)
_add_node(src_nid, name, line)
table_nids[name.lower()] = src_nid
for child in node.children:
if child.type == "add_constraint":
for cc in child.children:
if cc.type != "constraint":
continue
found_ref = False
ref_name: str | None = None
for ccc in cc.children:
if ccc.type == "keyword_references":
found_ref = True
elif found_ref and ccc.type == "object_reference":
ref_name = _read(ccc)
break
if ref_name:
ref_nid = table_nids.get(ref_name.lower())
if not ref_nid:
ref_nid = _make_id(stem, ref_name)
_add_edge(src_nid, ref_nid, "references", line)
elif t == "create_trigger":
trig_name: str | None = None
tbl_name: str | None = None
after_trigger = False
after_for = False
for c in node.children:
if c.type == "keyword_trigger":
after_trigger = True
elif after_trigger and not trig_name and c.type == "object_reference":
trig_name = _read(c)
elif c.type == "keyword_for":
after_for = True
elif after_for and not tbl_name and c.type == "object_reference":
tbl_name = _read(c)
if trig_name:
trig_nid = _make_id(stem, trig_name)
_add_node(trig_nid, trig_name, line)
if tbl_name:
tbl_nid = table_nids.get(tbl_name.lower()) or _make_id(stem, tbl_name)
_add_edge(trig_nid, tbl_nid, "triggers", line)
elif t == "fb_proc_or_trigger":
text = _read(node)
m = re.match(
r"CREATE\s+(?:OR\s+(?:REPLACE|ALTER)\s+)?"
r"(PROCEDURE|TRIGGER|FUNCTION)\s+([\w$]+)",
text, re.IGNORECASE,
)
if m:
obj_type = m.group(1).upper()
obj_name = m.group(2)
obj_nid = _make_id(stem, obj_name)
label = obj_name if obj_type == "TRIGGER" else f"{obj_name}()"
_add_node(obj_nid, label, line)
if obj_type == "TRIGGER":
fm = re.search(r"\bFOR\s+([\w$]+)", text, re.IGNORECASE)
if fm:
tbl = fm.group(1)
tbl_nid = table_nids.get(tbl.lower()) or _make_id(stem, tbl)
_add_edge(obj_nid, tbl_nid, "triggers", line)
_NON_TABLES = {
"select", "where", "set", "dual", "null", "true", "false",
"first", "skip", "rows", "next", "only", "lateral",
}
seen_tbls: set[str] = set()
for rm in re.finditer(r"\b(?:FROM|JOIN|INTO)\s+([\w$]+)", text, re.IGNORECASE):
tbl = rm.group(1)
if tbl.lower() not in _NON_TABLES and tbl.lower() not in seen_tbls:
seen_tbls.add(tbl.lower())
tbl_nid = table_nids.get(tbl.lower()) or _make_id(stem, tbl)
_add_edge(obj_nid, tbl_nid, "reads_from", line)
for rm in re.finditer(r"\bUPDATE\s+([\w$]+)", text, re.IGNORECASE):
tbl = rm.group(1)
if tbl.lower() not in _NON_TABLES and tbl.lower() not in seen_tbls:
seen_tbls.add(tbl.lower())
tbl_nid = table_nids.get(tbl.lower()) or _make_id(stem, tbl)
_add_edge(obj_nid, tbl_nid, "reads_from", line)
for child in node.children:
walk(child)
def _walk_from_refs(node, caller_nid: str, line: int) -> None:
"""Recursively find FROM/JOIN table references inside a node."""
if node.type in ("from", "join"):
for c in node.children:
if c.type == "relation":
for cc in c.children:
if cc.type == "object_reference":
tbl = _read(cc)
tbl_nid = _make_id(stem, tbl)
_add_edge(caller_nid, tbl_nid, "reads_from",
c.start_point[0] + 1)
for child in node.children:
_walk_from_refs(child, caller_nid, line)
for stmt in root.children:
if stmt.type == "statement":
for child in stmt.children:
walk(child)
elif stmt.type in ("fb_proc_or_trigger", "set_term", "declare_external_function"):
walk(stmt)
# Global regex fallback: catch any REFERENCES missed due to ERROR nodes in the parse tree
# (e.g. Firebird COMPUTED BY columns push constraints out of the tree entirely).
# Snapshot after tree walk so we don't re-emit edges already captured above.
emitted = {(e["source"], e["target"]) for e in edges if e["relation"] == "references"}
src_text = source.decode("utf-8", errors="replace")
for m in re.finditer(r"CREATE\s+TABLE\s+([\w$]+)\s*\(", src_text, re.IGNORECASE):
tbl_name = m.group(1)
tbl_nid = table_nids.get(tbl_name.lower())
if tbl_nid is None:
continue
tbl_line = src_text[: m.start()].count("\n") + 1
tail = src_text[m.start():]
end = re.search(r"(?:^|\n)(?:CREATE|SET\s+TERM|ALTER)\s", tail[1:], re.IGNORECASE)
block = tail[: end.start() + 1] if end else tail
for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", block, re.IGNORECASE):
ref_name = rm.group(1)
ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name)
if (tbl_nid, ref_nid) not in emitted:
_add_edge(tbl_nid, ref_nid, "references", tbl_line)
emitted.add((tbl_nid, ref_nid))
return {"nodes": nodes, "edges": edges}
+181
View File
@@ -0,0 +1,181 @@
"""Terraform extractor. Moved verbatim from graphify/extract.py."""
from __future__ import annotations
from pathlib import Path
from graphify.extractors.base import _make_id
_TF_META_HEADS = frozenset({"count", "each", "self", "path", "terraform"})
def extract_terraform(path: Path) -> dict:
"""Extract Terraform/HCL blocks and the references between them via tree-sitter.
Nodes: resources, data sources, modules, variables, outputs, providers, and
locals. Edges: `contains` (file -> block), `references` (block -> the blocks
it interpolates, e.g. `aws_instance.web` -> `var.region`), and `depends_on`
(explicit dependency edges).
Node IDs are scoped by the parent directory, not the file stem, because
Terraform resources are module(directory)-scoped: a resource defined in
main.tf is referenced from other .tf files in the same directory. Directory
scoping lets those cross-file references resolve when per-file extractions
are merged (stem scoping would split a definition from its references).
"""
try:
import tree_sitter_hcl as tshcl
from tree_sitter import Language, Parser
except ImportError:
return {"nodes": [], "edges": [], "error": "tree_sitter_hcl not installed. Run: pip install tree-sitter-hcl"}
try:
language = Language(tshcl.language())
parser = Parser(language)
source = path.read_bytes()
tree = parser.parse(source)
root = tree.root_node
except Exception as e:
return {"nodes": [], "edges": [], "error": str(e)}
str_path = str(path)
file_nid = _make_id(str_path)
scope = path.parent.name or "tf"
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str_path, "source_location": None}]
edges: list[dict] = []
seen_ids: set[str] = {file_nid}
seen_edges: set[tuple[str, str, str]] = set()
def _read(n) -> str:
return source[n.start_byte:n.end_byte].decode("utf-8", errors="replace")
def _label_text(n) -> str:
return _read(n).strip().strip('"')
def _add_node(address: str, label: str, line: int) -> str:
nid = _make_id(scope, address)
if nid not in seen_ids:
seen_ids.add(nid)
nodes.append({"id": nid, "label": label, "file_type": "code",
"source_file": str_path, "source_location": f"L{line}"})
edges.append({"source": file_nid, "target": nid, "relation": "contains",
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{line}", "weight": 1.0})
return nid
def _add_edge(src: str, address: str, relation: str, line: int) -> None:
tgt = _make_id(scope, address)
if src == tgt:
return
key = (src, tgt, relation)
if key in seen_edges:
return
seen_edges.add(key)
edges.append({"source": src, "target": tgt, "relation": relation,
"confidence": "EXTRACTED", "source_file": str_path,
"source_location": f"L{line}", "weight": 1.0})
def _block_parts(block) -> tuple:
btype = None
labels: list[str] = []
for c in block.children:
if c.type in ("block_start", "body", "block_end"):
break
if c.type == "identifier" and btype is None:
btype = _read(c)
elif c.type in ("string_lit", "identifier"):
labels.append(_label_text(c))
return btype, labels
def _ref_address(expr):
head = _read(expr)
parent = expr.parent
attrs: list[str] = []
if parent is not None:
seen_self = False
for c in parent.children:
if c.id == expr.id:
seen_self = True
continue
if seen_self and c.type == "get_attr":
name = None
for gc in c.children:
if gc.type == "identifier":
name = _read(gc)
break
if name is None:
break
attrs.append(name)
elif seen_self and c.type not in ("get_attr",):
break
if head in _TF_META_HEADS or not head:
return None
if head == "var":
return f"var.{attrs[0]}" if attrs else None
if head == "local":
return f"local.{attrs[0]}" if attrs else None
if head == "module":
return f"module.{attrs[0]}" if attrs else None
if head == "data":
return f"data.{attrs[0]}.{attrs[1]}" if len(attrs) >= 2 else None
return f"{head}.{attrs[0]}" if attrs else None
def _collect_refs(node, owner_nid: str, relation: str) -> None:
rel = relation
if node.type == "attribute":
key_node = node.child_by_field_name("key") or (
node.children[0] if node.children else None
)
if key_node is not None and _read(key_node) == "depends_on":
rel = "depends_on"
if node.type == "variable_expr":
addr = _ref_address(node)
if addr:
_add_edge(owner_nid, addr, rel, node.start_point[0] + 1)
for c in node.children:
if c.is_named:
_collect_refs(c, owner_nid, rel)
def _body_of(block):
for c in block.children:
if c.type == "body":
return c
return None
body = next((c for c in root.children if c.type == "body"), root)
for block in body.children:
if block.type != "block":
continue
btype, labels = _block_parts(block)
line = block.start_point[0] + 1
blk_body = _body_of(block)
if btype == "resource" and len(labels) >= 2:
owner = _add_node(f"{labels[0]}.{labels[1]}", f"{labels[0]}.{labels[1]}", line)
elif btype == "data" and len(labels) >= 2:
owner = _add_node(f"data.{labels[0]}.{labels[1]}", f"data.{labels[0]}.{labels[1]}", line)
elif btype == "module" and labels:
owner = _add_node(f"module.{labels[0]}", f"module.{labels[0]}", line)
elif btype == "variable" and labels:
owner = _add_node(f"var.{labels[0]}", f"var.{labels[0]}", line)
elif btype == "output" and labels:
owner = _add_node(f"output.{labels[0]}", f"output.{labels[0]}", line)
elif btype == "provider" and labels:
owner = _add_node(f"provider.{labels[0]}", f"provider.{labels[0]}", line)
elif btype == "locals" and blk_body is not None:
for attr in blk_body.children:
if attr.type != "attribute":
continue
key_node = attr.children[0] if attr.children else None
if key_node is None:
continue
key = _read(key_node)
lnid = _add_node(f"local.{key}", f"local.{key}", attr.start_point[0] + 1)
_collect_refs(attr, lnid, "references")
continue
else:
continue
if blk_body is not None:
_collect_refs(blk_body, owner, "references")
return {"nodes": nodes, "edges": edges}