refactor(extract): migrate 18 bespoke language extractors to extractors/ package
Continues the graphify/extract.py -> graphify/extractors/ split (MIGRATION.md, upstream #1212), which already moved blade/zig/elixir/razor. Moves the independent bespoke extractors whose closures are fully private (no shared _extract_generic core, no shared mutable caches), verbatim, following the documented invariants: dart, rust, go, powershell (+psd1 manifest), fortran, sql, dm (dm/dmm/dmi/dmf), bash, apex, terraform, sln, pascal_forms (delphi .dfm + lazarus .lfm), json_config Each language's private helper funcs and constants move with it; only the `extract_<lang>` entry points (plus fortran's _cpp_preprocess, which has a direct unit test) are re-exported from extract.py's facade block, so every existing importer (__main__.py, watch.py, tests) is unchanged and object identity is preserved. Registry (extractors/__init__.py) grows 4 -> 22 langs. Verified: AST closure-privacy analysis (no symbol referenced from outside its moved set except via the facade); byte-identity of every moved span; extract._DISPATCH still resolves every extension; ruff clean; skillgen --check OK. extract.py drops 17,054 -> 13,121 LOC. Full suite unchanged: 3036 passed, 29 skipped (excluding env-only openai tests). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
cc2d3c13a6
commit
9fc30e680d
+13
-3946
File diff suppressed because it is too large
Load Diff
@@ -12,7 +12,21 @@ written so an AI agent can execute it in a single session.
|
||||
| zig | yes |
|
||||
| elixir | yes |
|
||||
| razor | yes |
|
||||
| (40 more in extract.py) | no |
|
||||
| dart | yes |
|
||||
| rust | yes |
|
||||
| go | yes |
|
||||
| powershell (ps1 + psd1 manifest) | yes |
|
||||
| fortran | yes |
|
||||
| sql | yes |
|
||||
| dm (dm/dmm/dmi/dmf) | yes |
|
||||
| bash | yes |
|
||||
| apex | yes |
|
||||
| terraform | yes |
|
||||
| sln | yes |
|
||||
| pascal_forms (dfm + lfm) | yes |
|
||||
| json_config | yes |
|
||||
| (config-driven core: python, js, java, c, cpp, csharp, kotlin, scala, php, lua, swift, groovy, vue, svelte, astro, xaml, groovy) | no — shared _extract_generic core, move as one batch |
|
||||
| (other bespoke: julia, verilog, markdown, objc, csproj, slnx, lazarus_package, pascal) | no |
|
||||
|
||||
Note: config-driven extractors (python, js, java, c, cpp, ruby, csharp,
|
||||
kotlin, scala, php, lua, swift, groovy) depend on the shared
|
||||
|
||||
@@ -10,14 +10,45 @@ from __future__ import annotations
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
|
||||
from graphify.extractors.apex import extract_apex
|
||||
from graphify.extractors.bash import extract_bash
|
||||
from graphify.extractors.blade import extract_blade
|
||||
from graphify.extractors.dart import extract_dart
|
||||
from graphify.extractors.dm import extract_dm, extract_dmf, extract_dmi, extract_dmm
|
||||
from graphify.extractors.elixir import extract_elixir
|
||||
from graphify.extractors.fortran import extract_fortran
|
||||
from graphify.extractors.go import extract_go
|
||||
from graphify.extractors.json_config import extract_json
|
||||
from graphify.extractors.pascal_forms import extract_delphi_form, extract_lazarus_form
|
||||
from graphify.extractors.powershell import extract_powershell, extract_powershell_manifest
|
||||
from graphify.extractors.razor import extract_razor
|
||||
from graphify.extractors.rust import extract_rust
|
||||
from graphify.extractors.sln import extract_sln
|
||||
from graphify.extractors.sql import extract_sql
|
||||
from graphify.extractors.terraform import extract_terraform
|
||||
from graphify.extractors.zig import extract_zig
|
||||
|
||||
LANGUAGE_EXTRACTORS: dict[str, Callable[[Path], dict]] = {
|
||||
"apex": extract_apex,
|
||||
"bash": extract_bash,
|
||||
"blade": extract_blade,
|
||||
"dart": extract_dart,
|
||||
"delphi_form": extract_delphi_form,
|
||||
"dm": extract_dm,
|
||||
"dmf": extract_dmf,
|
||||
"dmi": extract_dmi,
|
||||
"dmm": extract_dmm,
|
||||
"elixir": extract_elixir,
|
||||
"fortran": extract_fortran,
|
||||
"go": extract_go,
|
||||
"json": extract_json,
|
||||
"lazarus_form": extract_lazarus_form,
|
||||
"powershell": extract_powershell,
|
||||
"powershell_manifest": extract_powershell_manifest,
|
||||
"razor": extract_razor,
|
||||
"rust": extract_rust,
|
||||
"sln": extract_sln,
|
||||
"sql": extract_sql,
|
||||
"terraform": extract_terraform,
|
||||
"zig": extract_zig,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,215 @@
|
||||
"""Apex extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
from pathlib import Path
|
||||
from graphify.extractors.base import _file_stem, _make_id
|
||||
|
||||
|
||||
def extract_apex(path: Path) -> dict:
|
||||
"""Extract classes, interfaces, enums, methods, and Salesforce constructs from
|
||||
Apex .cls and .trigger files using regex (no tree-sitter grammar on PyPI)."""
|
||||
import re as _re
|
||||
try:
|
||||
source = path.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return {"nodes": [], "edges": []}
|
||||
|
||||
str_path = str(path)
|
||||
stem = _file_stem(path)
|
||||
file_nid = _make_id(str_path)
|
||||
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid,
|
||||
"label": label,
|
||||
"file_type": "code",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
})
|
||||
|
||||
def add_edge(src: str, tgt: str, relation: str, line: int,
|
||||
confidence: str = "EXTRACTED") -> None:
|
||||
edges.append({
|
||||
"source": src,
|
||||
"target": tgt,
|
||||
"relation": relation,
|
||||
"confidence": confidence,
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
"weight": 1.0,
|
||||
})
|
||||
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
lines = source.splitlines()
|
||||
|
||||
_ACCESS = r"(?:public|private|protected|global|webService)?"
|
||||
_SHARING = r"(?:\s+(?:with|without|inherited)\s+sharing)?"
|
||||
_MOD = r"(?:\s+(?:abstract|virtual|override|static|final|transient|testMethod))?"
|
||||
_ANNOTATION = r"(?:\s*@\w+(?:\s*\([^)]*\))?\s*)*"
|
||||
|
||||
cls_re = _re.compile(
|
||||
rf"^{_ANNOTATION}\s*{_ACCESS}{_SHARING}{_MOD}\s*class\s+(\w+)"
|
||||
rf"(?:\s+extends\s+(\w+))?(?:\s+implements\s+([\w,\s]+))?\s*\{{?",
|
||||
_re.IGNORECASE,
|
||||
)
|
||||
iface_re = _re.compile(
|
||||
rf"^{_ANNOTATION}\s*{_ACCESS}{_SHARING}{_MOD}\s*interface\s+(\w+)"
|
||||
rf"(?:\s+extends\s+([\w,\s]+))?\s*\{{?",
|
||||
_re.IGNORECASE,
|
||||
)
|
||||
enum_re = _re.compile(
|
||||
rf"^{_ANNOTATION}\s*{_ACCESS}{_SHARING}{_MOD}\s*enum\s+(\w+)\s*\{{?",
|
||||
_re.IGNORECASE,
|
||||
)
|
||||
trigger_re = _re.compile(
|
||||
r"^\s*trigger\s+(\w+)\s+on\s+(\w+)\s*\(",
|
||||
_re.IGNORECASE,
|
||||
)
|
||||
method_re = _re.compile(
|
||||
rf"^{_ANNOTATION}\s*{_ACCESS}{_MOD}\s*(?:static\s+)?[\w<>\[\]]+\s+(\w+)\s*\([^)]*\)\s*(?:throws\s+\w+\s*)?\{{?",
|
||||
_re.IGNORECASE,
|
||||
)
|
||||
annotation_re = _re.compile(r"@(\w+)", _re.IGNORECASE)
|
||||
soql_re = _re.compile(r"\[\s*SELECT\b[^\]]+FROM\s+(\w+)", _re.IGNORECASE)
|
||||
dml_re = _re.compile(r"\b(insert|update|delete|upsert|merge|undelete)\s+\w", _re.IGNORECASE)
|
||||
|
||||
_CONTROL_FLOW = frozenset({
|
||||
"if", "else", "for", "while", "do", "switch", "try", "catch",
|
||||
"finally", "return", "throw", "new", "void", "null",
|
||||
"true", "false", "this", "super", "class", "interface", "enum",
|
||||
"trigger", "on",
|
||||
})
|
||||
|
||||
current_class_nid: str | None = None
|
||||
pending_annotations: list[str] = []
|
||||
|
||||
for lineno, line_text in enumerate(lines, start=1):
|
||||
stripped = line_text.strip()
|
||||
|
||||
if stripped.startswith("@"):
|
||||
for m in annotation_re.finditer(stripped):
|
||||
pending_annotations.append(m.group(1).lower())
|
||||
continue
|
||||
|
||||
tm = trigger_re.match(stripped)
|
||||
if tm:
|
||||
trig_name, sobject = tm.group(1), tm.group(2)
|
||||
trig_nid = _make_id(stem, trig_name)
|
||||
add_node(trig_nid, trig_name, lineno)
|
||||
add_edge(file_nid, trig_nid, "contains", lineno)
|
||||
sob_nid = _make_id(sobject)
|
||||
if sob_nid not in seen_ids:
|
||||
add_node(sob_nid, sobject, lineno)
|
||||
add_edge(trig_nid, sob_nid, "uses", lineno, confidence="INFERRED")
|
||||
current_class_nid = trig_nid
|
||||
pending_annotations = []
|
||||
continue
|
||||
|
||||
cm = cls_re.match(stripped)
|
||||
if cm:
|
||||
class_name = cm.group(1)
|
||||
if class_name.lower() in _CONTROL_FLOW:
|
||||
pending_annotations = []
|
||||
continue
|
||||
class_nid = _make_id(stem, class_name)
|
||||
add_node(class_nid, class_name, lineno)
|
||||
add_edge(file_nid, class_nid, "contains", lineno)
|
||||
if cm.group(2):
|
||||
base = cm.group(2).strip()
|
||||
base_nid = _make_id(stem, base)
|
||||
if base_nid not in seen_ids:
|
||||
base_nid = _make_id(base)
|
||||
if base_nid not in seen_ids:
|
||||
add_node(base_nid, base, lineno)
|
||||
add_edge(class_nid, base_nid, "extends", lineno, confidence="INFERRED")
|
||||
if cm.group(3):
|
||||
for iface in cm.group(3).split(","):
|
||||
iface = iface.strip()
|
||||
if iface:
|
||||
iface_nid = _make_id(stem, iface)
|
||||
if iface_nid not in seen_ids:
|
||||
iface_nid = _make_id(iface)
|
||||
if iface_nid not in seen_ids:
|
||||
add_node(iface_nid, iface, lineno)
|
||||
add_edge(class_nid, iface_nid, "implements", lineno, confidence="INFERRED")
|
||||
current_class_nid = class_nid
|
||||
pending_annotations = []
|
||||
continue
|
||||
|
||||
im = iface_re.match(stripped)
|
||||
if im:
|
||||
iface_name = im.group(1)
|
||||
if iface_name.lower() in _CONTROL_FLOW:
|
||||
pending_annotations = []
|
||||
continue
|
||||
iface_nid = _make_id(stem, iface_name)
|
||||
add_node(iface_nid, iface_name, lineno)
|
||||
add_edge(file_nid if current_class_nid is None else current_class_nid,
|
||||
iface_nid, "contains", lineno)
|
||||
if im.group(2):
|
||||
for parent in im.group(2).split(","):
|
||||
parent = parent.strip()
|
||||
if parent:
|
||||
parent_nid = _make_id(stem, parent)
|
||||
if parent_nid not in seen_ids:
|
||||
parent_nid = _make_id(parent)
|
||||
if parent_nid not in seen_ids:
|
||||
add_node(parent_nid, parent, lineno)
|
||||
add_edge(iface_nid, parent_nid, "extends", lineno, confidence="INFERRED")
|
||||
pending_annotations = []
|
||||
continue
|
||||
|
||||
em = enum_re.match(stripped)
|
||||
if em:
|
||||
enum_name = em.group(1)
|
||||
if enum_name.lower() in _CONTROL_FLOW:
|
||||
pending_annotations = []
|
||||
continue
|
||||
enum_nid = _make_id(stem, enum_name)
|
||||
add_node(enum_nid, enum_name, lineno)
|
||||
add_edge(file_nid if current_class_nid is None else current_class_nid,
|
||||
enum_nid, "contains", lineno)
|
||||
pending_annotations = []
|
||||
continue
|
||||
|
||||
if current_class_nid is not None:
|
||||
mm = method_re.match(stripped)
|
||||
if mm:
|
||||
method_name = mm.group(1)
|
||||
if method_name.lower() not in _CONTROL_FLOW:
|
||||
method_nid = _make_id(current_class_nid, method_name)
|
||||
method_label = f".{method_name}()"
|
||||
add_node(method_nid, method_label, lineno)
|
||||
add_edge(current_class_nid, method_nid, "method", lineno)
|
||||
if "auraenabled" in pending_annotations or "invocablemethod" in pending_annotations:
|
||||
add_edge(file_nid, method_nid, "contains", lineno, confidence="INFERRED")
|
||||
pending_annotations = []
|
||||
continue
|
||||
|
||||
pending_annotations = []
|
||||
|
||||
for sm in soql_re.finditer(line_text):
|
||||
sobject = sm.group(1)
|
||||
sob_nid = _make_id(sobject)
|
||||
if sob_nid not in seen_ids:
|
||||
add_node(sob_nid, sobject, lineno)
|
||||
src = current_class_nid or file_nid
|
||||
add_edge(src, sob_nid, "uses", lineno, confidence="INFERRED")
|
||||
|
||||
for dm in dml_re.finditer(line_text):
|
||||
dml_op = dm.group(1).lower()
|
||||
dml_nid = _make_id(f"dml_{dml_op}")
|
||||
if dml_nid not in seen_ids:
|
||||
add_node(dml_nid, dml_op, lineno)
|
||||
src = current_class_nid or file_nid
|
||||
add_edge(src, dml_nid, "uses", lineno, confidence="INFERRED")
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
@@ -0,0 +1,230 @@
|
||||
"""Bash extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from graphify.extractors.base import _file_stem, _make_id, _read_text
|
||||
|
||||
|
||||
def extract_bash(path: Path) -> dict:
|
||||
"""Extract functions, source imports, and cross-function calls from a .sh file."""
|
||||
try:
|
||||
import tree_sitter_bash as tsbash
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree-sitter-bash not installed"}
|
||||
|
||||
try:
|
||||
language = Language(tsbash.language())
|
||||
parser = Parser(language)
|
||||
source = path.read_bytes()
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
stem = _file_stem(path)
|
||||
str_path = str(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
function_bodies: list[tuple[str, Any]] = []
|
||||
defined_functions: set[str] = set()
|
||||
|
||||
from graphify.security import sanitize_metadata # module-level cached import
|
||||
|
||||
def add_node(nid: str, label: str, line: int, kind: str = "code") -> None:
|
||||
if nid and nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({"id": nid, "label": label, "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line}",
|
||||
"metadata": sanitize_metadata({"language": "bash", "kind": kind})}) # noqa: E501
|
||||
|
||||
def add_edge(src: str, tgt: str, relation: str, line: int,
|
||||
confidence: str = "EXTRACTED", weight: float = 1.0,
|
||||
context: str | None = None) -> None:
|
||||
if not src or not tgt or src == tgt:
|
||||
return
|
||||
edge = {"source": src, "target": tgt, "relation": relation,
|
||||
"confidence": confidence, "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": weight}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
# file_nid is fully path-derived and never produced by _make_id(stem, func_name),
|
||||
# so appending "__entry" guarantees a distinct ID from any function node.
|
||||
entry_nid = file_nid + "__entry"
|
||||
add_node(file_nid, path.name, 1, kind="file")
|
||||
add_node(entry_nid, f"{path.name} script", 1, kind="bash_entrypoint")
|
||||
add_edge(file_nid, entry_nid, "contains", 1)
|
||||
|
||||
_BASH_SOURCE_COMMANDS = frozenset({"source", "."})
|
||||
# Parent node types that mean a contained command is part of a substitution
|
||||
# or expansion, not a real function call. Token-level filtering misses
|
||||
# these because `$(build)` exposes `build` as a child command whose name
|
||||
# token has no metacharacters — only the parent does.
|
||||
_BASH_EXPANSION_PARENTS = frozenset({
|
||||
"command_substitution",
|
||||
"process_substitution",
|
||||
})
|
||||
|
||||
def text(node) -> str:
|
||||
return source[node.start_byte:node.end_byte].decode("utf-8", errors="replace")
|
||||
|
||||
def is_inside_expansion(node) -> bool:
|
||||
parent = node.parent
|
||||
while parent is not None:
|
||||
if parent.type in _BASH_EXPANSION_PARENTS:
|
||||
return True
|
||||
parent = parent.parent
|
||||
return False
|
||||
|
||||
def literal(node) -> str | None:
|
||||
# Token-level filter: rejects names containing shell metacharacters.
|
||||
# Combined with `is_inside_expansion` for parent-context rejection.
|
||||
raw = text(node).strip()
|
||||
if not raw:
|
||||
return None
|
||||
if raw[0:1] in {"'", '"'} and raw[-1:] == raw[0]:
|
||||
raw = raw[1:-1]
|
||||
if any(token in raw for token in ("$", "`", "$(", "<(", ">", "|", ";", "&")):
|
||||
return None
|
||||
return raw
|
||||
|
||||
def _bash_func_name(node) -> str | None:
|
||||
"""Get the name from a function_definition node."""
|
||||
# bash grammar: function_definition has a word child (the name)
|
||||
for child in node.children:
|
||||
if child.type == "word":
|
||||
return literal(child)
|
||||
return None
|
||||
|
||||
def walk_calls(body_node, func_nid: str, seen_calls: set) -> None:
|
||||
if body_node is None:
|
||||
return
|
||||
for child in body_node.children:
|
||||
if child.type == "function_definition":
|
||||
# Skip nested function definitions — their bodies are walked
|
||||
# separately, so we don't attribute their calls to the
|
||||
# enclosing scope.
|
||||
continue
|
||||
if child.type == "command" and not is_inside_expansion(child):
|
||||
cmd_name_node = child.child_by_field_name("name")
|
||||
if cmd_name_node is None and child.children:
|
||||
cmd_name_node = child.children[0]
|
||||
if cmd_name_node:
|
||||
name = literal(cmd_name_node)
|
||||
# Defined-functions wins. Skip-lists for external commands
|
||||
# would create false negatives when a user defines a
|
||||
# function shadowing an external (`install`, `find`, etc.).
|
||||
if name and name in defined_functions:
|
||||
tgt = _make_id(stem, name)
|
||||
key = (func_nid, tgt)
|
||||
if tgt and key not in seen_calls:
|
||||
seen_calls.add(key)
|
||||
add_edge(func_nid, tgt, "calls",
|
||||
child.start_point[0] + 1,
|
||||
confidence="EXTRACTED", context="call")
|
||||
walk_calls(child, func_nid, seen_calls)
|
||||
|
||||
def walk(node, parent_nid: str) -> None:
|
||||
t = node.type
|
||||
if t == "function_definition":
|
||||
name = _bash_func_name(node)
|
||||
if name:
|
||||
fn_nid = _make_id(stem, name)
|
||||
line = node.start_point[0] + 1
|
||||
add_node(fn_nid, f"{name}()", line, kind="bash_function")
|
||||
add_edge(parent_nid, fn_nid, "defines", line)
|
||||
defined_functions.add(name)
|
||||
# find the compound_statement body
|
||||
body = None
|
||||
for child in node.children:
|
||||
if child.type == "compound_statement":
|
||||
body = child
|
||||
break
|
||||
function_bodies.append((fn_nid, body))
|
||||
# Recurse into the body so nested function definitions are discovered
|
||||
# and added to function_bodies for the second-pass walk_calls.
|
||||
if body is not None:
|
||||
walk(body, fn_nid)
|
||||
return
|
||||
|
||||
if t == "command":
|
||||
if is_inside_expansion(node):
|
||||
return
|
||||
cmd_name_node = node.child_by_field_name("name")
|
||||
if cmd_name_node is None and node.children:
|
||||
cmd_name_node = node.children[0]
|
||||
if cmd_name_node:
|
||||
cmd = literal(cmd_name_node)
|
||||
if cmd in _BASH_SOURCE_COMMANDS and cmd not in defined_functions:
|
||||
# find the path argument (first word after command name)
|
||||
args = [c for c in node.children
|
||||
if c.type in ("word", "string", "concatenation")
|
||||
and c != cmd_name_node]
|
||||
if args:
|
||||
raw = _read_text(args[0], source).strip().strip("'\"")
|
||||
line = node.start_point[0] + 1
|
||||
if raw.startswith((".", "/")):
|
||||
resolved = (path.parent / raw).resolve()
|
||||
# Only emit the edge if the target actually exists on
|
||||
# disk — prevents graph pollution from crafted paths
|
||||
# like `source ../../etc/passwd` that traverse outside
|
||||
# the project tree (B-1).
|
||||
if resolved.exists():
|
||||
tgt_nid = _make_id(str(resolved))
|
||||
add_edge(file_nid, tgt_nid, "imports_from", line,
|
||||
context="import")
|
||||
else:
|
||||
tgt_nid = _make_id(raw)
|
||||
if tgt_nid:
|
||||
add_edge(file_nid, tgt_nid, "imports", line,
|
||||
context="import")
|
||||
return
|
||||
|
||||
if t == "declaration_command":
|
||||
# export/declare/readonly VAR=value at program level
|
||||
if node.parent and node.parent.type == "program":
|
||||
for child in node.children:
|
||||
if child.type == "variable_assignment":
|
||||
var_node = child.child_by_field_name("name")
|
||||
if var_node:
|
||||
var = _read_text(var_node, source).strip()
|
||||
if var:
|
||||
var_nid = _make_id(stem, var)
|
||||
line = child.start_point[0] + 1
|
||||
add_node(var_nid, var, line)
|
||||
add_edge(file_nid, var_nid, "defines", line)
|
||||
return
|
||||
|
||||
for child in node.children:
|
||||
walk(child, parent_nid)
|
||||
|
||||
# Pre-pass: collect all defined function names so the source-command handler
|
||||
# in walk() can detect user-defined functions that shadow 'source' / '.'
|
||||
# regardless of definition order in the file.
|
||||
def _prescan_functions(node) -> None:
|
||||
if node.type == "function_definition":
|
||||
name = _bash_func_name(node)
|
||||
if name:
|
||||
defined_functions.add(name)
|
||||
for child in node.children:
|
||||
_prescan_functions(child)
|
||||
else:
|
||||
for child in node.children:
|
||||
_prescan_functions(child)
|
||||
|
||||
_prescan_functions(root)
|
||||
walk(root, file_nid)
|
||||
|
||||
# Second pass: cross-function calls
|
||||
top_seen: set = set()
|
||||
walk_calls(root, entry_nid, top_seen) # top-level calls attributed to the entrypoint
|
||||
for fn_nid, body in function_bodies:
|
||||
walk_calls(body, fn_nid, set())
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
@@ -0,0 +1,528 @@
|
||||
"""Dart extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from pathlib import Path
|
||||
from graphify.extractors.base import _file_stem, _make_id
|
||||
|
||||
|
||||
def extract_dart(path: Path) -> dict:
|
||||
"""Extract classes, mixins, functions, imports, generic calls, and annotations from a .dart file using regex."""
|
||||
try:
|
||||
src = path.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return {"error": f"cannot read {path}"}
|
||||
|
||||
# Remove inline and multi-line comments while leaving string literals untouched to prevent stripping URLs/paths inside strings
|
||||
comment_string_pattern = re.compile(
|
||||
r'"""(?:\\.|[\s\S])*?"""'
|
||||
r"|'''(?:\\.|[\s\S])*?'''"
|
||||
r'|"(?:\\.|[^"\\])*"'
|
||||
r"|'(?:\\.|[^'\\])*'"
|
||||
r"|/\*[\s\S]*?\*/"
|
||||
r"|//[^\n]*"
|
||||
)
|
||||
def _comment_replace(match: re.Match) -> str:
|
||||
token = match.group(0)
|
||||
if token.startswith("/"):
|
||||
return ""
|
||||
return token
|
||||
src_clean = comment_string_pattern.sub(_comment_replace, src)
|
||||
|
||||
stem = _file_stem(path)
|
||||
file_nid = _make_id(str(path))
|
||||
|
||||
# Check if this is a part-of file and redirect to parent
|
||||
part_of_match = re.search(r"^\s*part\s+of\s+['\"]([^'\"]+)['\"]", src_clean, re.MULTILINE)
|
||||
is_part = False
|
||||
if part_of_match:
|
||||
parent_ref = part_of_match.group(1)
|
||||
if parent_ref.endswith(".dart"):
|
||||
try:
|
||||
parent_path = (path.parent / parent_ref).resolve()
|
||||
if parent_path.exists():
|
||||
stem = _file_stem(parent_path)
|
||||
file_nid = _make_id(str(parent_path))
|
||||
is_part = True
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
nodes = []
|
||||
if not is_part:
|
||||
nodes.append({"id": file_nid, "label": path.name, "file_type": "code",
|
||||
"source_file": str(path), "source_location": None})
|
||||
edges = []
|
||||
defined: set[str] = set()
|
||||
|
||||
def add_node(nid: str, label: str, ftype: str = "code", source_file: str | None = str(path)) -> None:
|
||||
if nid not in defined:
|
||||
nodes.append({"id": nid, "label": label, "file_type": ftype,
|
||||
"source_file": source_file, "source_location": None})
|
||||
defined.add(nid)
|
||||
|
||||
def add_edge(src_id: str, tgt_id: str, relation: str, weight: float = 1.0, context: str | None = None) -> None:
|
||||
edge = {"source": src_id, "target": tgt_id, "relation": relation,
|
||||
"confidence": "EXTRACTED", "confidence_score": 1.0,
|
||||
"source_file": str(path), "source_location": None, "weight": weight}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
def _split_types(text: str) -> list[str]:
|
||||
parts = []
|
||||
current = []
|
||||
depth = 0
|
||||
for char in text:
|
||||
if char == "<":
|
||||
depth += 1
|
||||
current.append(char)
|
||||
elif char == ">":
|
||||
depth -= 1
|
||||
current.append(char)
|
||||
elif char == "," and depth == 0:
|
||||
parts.append("".join(current).strip())
|
||||
current = []
|
||||
else:
|
||||
current.append(char)
|
||||
if current:
|
||||
parts.append("".join(current).strip())
|
||||
return [p for p in parts if p]
|
||||
|
||||
def _find_matching_brace(text: str, start_pos: int) -> int:
|
||||
brace_count = 0
|
||||
in_double_quote = False
|
||||
in_single_quote = False
|
||||
escape = False
|
||||
|
||||
first_brace = text.find("{", start_pos)
|
||||
if first_brace == -1:
|
||||
return len(text)
|
||||
|
||||
brace_count = 1
|
||||
i = first_brace + 1
|
||||
n = len(text)
|
||||
while i < n:
|
||||
char = text[i]
|
||||
if escape:
|
||||
escape = False
|
||||
i += 1
|
||||
continue
|
||||
if char == "\\":
|
||||
escape = True
|
||||
i += 1
|
||||
continue
|
||||
if text[i:i+3] == '"""' and not in_single_quote:
|
||||
i += 3
|
||||
end = text.find('"""', i)
|
||||
i = end + 3 if end != -1 else n
|
||||
continue
|
||||
if text[i:i+3] == "'''" and not in_double_quote:
|
||||
i += 3
|
||||
end = text.find("'''", i)
|
||||
i = end + 3 if end != -1 else n
|
||||
continue
|
||||
if char == '"' and not in_single_quote:
|
||||
in_double_quote = not in_double_quote
|
||||
elif char == "'" and not in_double_quote:
|
||||
in_single_quote = not in_single_quote
|
||||
elif not in_double_quote and not in_single_quote:
|
||||
if char == "{":
|
||||
brace_count += 1
|
||||
elif char == "}":
|
||||
brace_count -= 1
|
||||
if brace_count == 0:
|
||||
return i + 1
|
||||
i += 1
|
||||
return len(text)
|
||||
|
||||
# 1. Classes, mixins, and enums declarations (with inheritance, mixins, interfaces, and generics)
|
||||
# Supports multiple combined modifiers (e.g., abstract base class, mixin class) without capturing "class" as a name
|
||||
class_pattern = r"^\s*(?:(?:abstract|sealed|base|interface|final|mixin)\s+)*(?:class|mixin|enum|extension\s+type)\s+(\w+)"
|
||||
for m in re.finditer(class_pattern, src_clean, re.MULTILINE):
|
||||
class_name = m.group(1)
|
||||
class_nid = _make_id(stem, class_name)
|
||||
add_node(class_nid, class_name)
|
||||
add_edge(file_nid, class_nid, "defines")
|
||||
|
||||
# Manually parse extends/on, with, and implements in header to handle nested generics brackets balanced
|
||||
start_idx = m.end()
|
||||
rest = src_clean[start_idx : start_idx + 500]
|
||||
|
||||
# Skip class generic parameters
|
||||
if rest.lstrip().startswith("<"):
|
||||
offset = rest.find("<")
|
||||
depth = 1
|
||||
i = offset + 1
|
||||
while i < len(rest) and depth > 0:
|
||||
if rest[i] == "<": depth += 1
|
||||
elif rest[i] == ">": depth -= 1
|
||||
i += 1
|
||||
rest = rest[i:]
|
||||
|
||||
# Skip primary constructor (e.g. extension type MyExt(int id))
|
||||
if rest.lstrip().startswith("("):
|
||||
offset = rest.find("(")
|
||||
depth = 1
|
||||
i = offset + 1
|
||||
while i < len(rest) and depth > 0:
|
||||
if rest[i] == "(": depth += 1
|
||||
elif rest[i] == ")": depth -= 1
|
||||
i += 1
|
||||
rest = rest[i:]
|
||||
|
||||
header_end = rest.find("{")
|
||||
if header_end == -1:
|
||||
header_end = rest.find(";")
|
||||
if header_end == -1:
|
||||
header_end = len(rest)
|
||||
header = rest[:header_end]
|
||||
|
||||
base_class = None
|
||||
generics = None
|
||||
mixins_list = []
|
||||
interfaces_list = []
|
||||
|
||||
# Parse extends or on
|
||||
extends_m = re.search(r"^\s*(?:extends|on)\s+([a-zA-Z0-9_.]+)", header)
|
||||
if extends_m:
|
||||
base_class = extends_m.group(1)
|
||||
rest_header = header[extends_m.end():]
|
||||
if rest_header.strip().startswith("<"):
|
||||
start_idx = rest_header.find("<")
|
||||
depth = 1
|
||||
i = start_idx + 1
|
||||
while i < len(rest_header) and depth > 0:
|
||||
if rest_header[i] == "<":
|
||||
depth += 1
|
||||
elif rest_header[i] == ">":
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
generics = rest_header[start_idx + 1 : i]
|
||||
break
|
||||
i += 1
|
||||
if generics is not None:
|
||||
header = rest_header[i + 1:]
|
||||
else:
|
||||
header = rest_header
|
||||
else:
|
||||
header = rest_header
|
||||
|
||||
# Parse with
|
||||
with_m = re.search(r"^\s*with\s+", header)
|
||||
if with_m:
|
||||
rest_header = header[with_m.end():]
|
||||
impl_idx = rest_header.find("implements")
|
||||
if impl_idx != -1:
|
||||
mixins_str = rest_header[:impl_idx]
|
||||
header = rest_header[impl_idx:]
|
||||
else:
|
||||
mixins_str = rest_header
|
||||
header = ""
|
||||
mixins_list = _split_types(mixins_str)
|
||||
|
||||
# Parse implements
|
||||
impl_m = re.search(r"^\s*implements\s+", header)
|
||||
if impl_m:
|
||||
interfaces_list = _split_types(header[impl_m.end():])
|
||||
|
||||
# Map extends inheritance relation
|
||||
if base_class:
|
||||
base_nid = _make_id(base_class)
|
||||
add_node(base_nid, base_class, source_file=None)
|
||||
add_edge(class_nid, base_nid, "inherits")
|
||||
|
||||
# Map generic type arguments (e.g. MyBloc extends Bloc<MyEvent, MyState>)
|
||||
if generics:
|
||||
for gen in _split_types(generics):
|
||||
gen_clean = gen.split("<")[0].strip()
|
||||
if gen_clean not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "void"}:
|
||||
gen_nid = _make_id(gen_clean)
|
||||
add_node(gen_nid, gen_clean, source_file=None)
|
||||
add_edge(class_nid, gen_nid, "references")
|
||||
|
||||
# Map mixins
|
||||
for mixin in mixins_list:
|
||||
mixin_clean = mixin.split("<")[0].strip()
|
||||
mixin_nid = _make_id(mixin_clean)
|
||||
add_node(mixin_nid, mixin_clean, source_file=None)
|
||||
add_edge(class_nid, mixin_nid, "mixes_in")
|
||||
|
||||
# Map interfaces
|
||||
for interface in interfaces_list:
|
||||
interface_clean = interface.split("<")[0].strip()
|
||||
interface_nid = _make_id(interface_clean)
|
||||
add_node(interface_nid, interface_clean, source_file=None)
|
||||
add_edge(class_nid, interface_nid, "implements")
|
||||
|
||||
# Extract class body for precise framework dependencies and event handling
|
||||
start_idx = m.start()
|
||||
brace_pos = src_clean.find("{", start_idx)
|
||||
semi_pos = src_clean.find(";", start_idx)
|
||||
|
||||
has_body = brace_pos != -1
|
||||
if has_body and semi_pos != -1 and semi_pos < brace_pos:
|
||||
has_body = False
|
||||
|
||||
if has_body:
|
||||
end_pos = _find_matching_brace(src_clean, start_idx)
|
||||
class_body = src_clean[brace_pos:end_pos]
|
||||
|
||||
# Bloc event registration: on<MyEvent>()
|
||||
for em in re.finditer(r"\bon<(\w+)>\s*\(", class_body):
|
||||
event_name = em.group(1)
|
||||
event_nid = _make_id(event_name)
|
||||
add_node(event_nid, event_name, source_file=None)
|
||||
add_edge(class_nid, event_nid, "calls", context="bloc_event")
|
||||
|
||||
# Bloc state emissions: emit(MyState) or yield MyState
|
||||
for sm in re.finditer(r"\b(?:emit|yield)\s*\(?\s*(?:const\s+)?([A-Z]\w*)\b", class_body):
|
||||
state_name = sm.group(1)
|
||||
if state_name not in {"String", "List", "Map", "Set", "Future", "Stream", "Object"}:
|
||||
state_nid = _make_id(state_name)
|
||||
add_node(state_nid, state_name, source_file=None)
|
||||
add_edge(class_nid, state_nid, "calls", context="emit_state")
|
||||
|
||||
# Bloc event additions: widget.add(MyEvent()) or bloc.add(MyEvent())
|
||||
for am in re.finditer(r"\b(?:\w*[Bb]loc\w*|context\.read<\w+>\(\))\.add\(\s*(?:const\s+)?([A-Z]\w*)\b", class_body):
|
||||
event_name = am.group(1)
|
||||
if event_name not in {"String", "List", "Map", "Set", "Future", "Stream", "Object"}:
|
||||
event_nid = _make_id(event_name)
|
||||
add_node(event_nid, event_name, source_file=None)
|
||||
add_edge(class_nid, event_nid, "calls", context="bloc_add_event")
|
||||
|
||||
# Riverpod provider references: ref.watch(provider)
|
||||
for rm in re.finditer(r"\bref\.(?:watch|read|listen)\s*\(\s*(\w+)\b", class_body):
|
||||
provider_name = rm.group(1)
|
||||
provider_nid = _make_id(provider_name)
|
||||
add_node(provider_nid, provider_name, source_file=None)
|
||||
add_edge(class_nid, provider_nid, "references", context="riverpod_reference")
|
||||
|
||||
# Widget to Bloc references: BlocBuilder<MyBloc, ...>
|
||||
for bm in re.finditer(r"\bBloc(?:Builder|Listener|Consumer|Provider|Selector)\s*<\s*([a-zA-Z0-9_]+)\b", class_body):
|
||||
bloc_name = bm.group(1)
|
||||
if bloc_name not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "void"}:
|
||||
bloc_nid = _make_id(bloc_name)
|
||||
add_node(bloc_nid, bloc_name, source_file=None)
|
||||
add_edge(class_nid, bloc_nid, "references", context="bloc_widget_binding")
|
||||
|
||||
# context.read<MyBloc>() or BlocProvider.of<MyBloc>(context)
|
||||
for lm in re.finditer(r"\b(?:read|watch|select|of)\s*<([a-zA-Z0-9_]+)>", class_body):
|
||||
bloc_name = lm.group(1)
|
||||
if bloc_name not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "void"}:
|
||||
bloc_nid = _make_id(bloc_name)
|
||||
add_node(bloc_nid, bloc_name, source_file=None)
|
||||
add_edge(class_nid, bloc_nid, "references", context="bloc_lookup")
|
||||
|
||||
# 2. Annotations mapping (class, mixin, enum, or function level annotations)
|
||||
# Support: @riverpod, @Riverpod(...), @injectable, @singleton, @RoutePage(), @HiveType(typeId: 0), @RestApi()
|
||||
# Matches `@annotation` and links it to the next class/mixin/enum/function declaration in the file
|
||||
annotation_pattern = r"@(\w+)(?:\([^)]*\))?"
|
||||
for am in re.finditer(annotation_pattern, src_clean):
|
||||
annotation_name = am.group(1)
|
||||
if annotation_name in {"override", "deprecated", "required", "protected", "mustCallSuper"}:
|
||||
continue
|
||||
annotation_pos = am.end()
|
||||
intervening_text = src_clean[annotation_pos : annotation_pos + 300]
|
||||
|
||||
class_m = re.search(r"^\s*(?:(?:abstract|sealed|base|interface|final|mixin)\s+)*(?:class|mixin|enum|extension\s+type)\s+(\w+)", intervening_text, re.MULTILINE)
|
||||
func_m = re.search(r"^\s*(?:factory\s+|static\s+|async\s+|external\s+|abstract\s+)?(?:\([^)]+\)|[a-zA-Z0-9_<>,.?]+)(?:\s+[a-zA-Z0-9_<>,.?]+){0,3}\s+(\w+)\s*\(", intervening_text, re.MULTILINE)
|
||||
|
||||
target_nid = None
|
||||
target_name = None
|
||||
target_type = None
|
||||
|
||||
if class_m and func_m:
|
||||
if class_m.start() < func_m.start():
|
||||
target_name = class_m.group(1)
|
||||
target_type = "class"
|
||||
target_nid = _make_id(stem, target_name)
|
||||
else:
|
||||
target_name = func_m.group(1)
|
||||
target_type = "function"
|
||||
target_nid = _make_id(stem, target_name)
|
||||
elif class_m:
|
||||
target_name = class_m.group(1)
|
||||
target_type = "class"
|
||||
target_nid = _make_id(stem, target_name)
|
||||
elif func_m:
|
||||
target_name = func_m.group(1)
|
||||
target_type = "function"
|
||||
target_nid = _make_id(stem, target_name)
|
||||
|
||||
if target_nid and target_name:
|
||||
actual_intervening = intervening_text[:min(class_m.start() if class_m else 300, func_m.start() if func_m else 300)]
|
||||
if ";" not in actual_intervening and "}" not in actual_intervening and "{" not in actual_intervening:
|
||||
annotation_nid = _make_id("annotation", annotation_name.lower())
|
||||
add_node(annotation_nid, f"@{annotation_name}", ftype="concept", source_file=None)
|
||||
add_edge(target_nid, annotation_nid, "configures")
|
||||
|
||||
# Riverpod specific provider generation mapping (supports camelCase class and functional providers)
|
||||
if annotation_name.lower() == "riverpod":
|
||||
if target_type == "class":
|
||||
provider_name = target_name[0].lower() + target_name[1:] + "Provider" if len(target_name) > 1 else target_name.lower() + "Provider"
|
||||
else:
|
||||
provider_name = target_name + "Provider"
|
||||
provider_nid = _make_id(provider_name)
|
||||
add_node(provider_nid, provider_name, ftype="concept", source_file=str(path))
|
||||
add_edge(target_nid, provider_nid, "defines", context="riverpod_provider")
|
||||
|
||||
# 2.5 Typedefs (Type Aliases)
|
||||
typedef_pattern = r"^\s*typedef\s+(\w+)\s*(?:<[^>]+>)?\s*=\s*([a-zA-Z0-9_<>,.?\s]+);"
|
||||
for m in re.finditer(typedef_pattern, src_clean, re.MULTILINE):
|
||||
typedef_name = m.group(1)
|
||||
target_type = m.group(2).split("<")[0].split(".")[-1].strip()
|
||||
if target_type not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "List", "Map", "Set", "void", "Function"}:
|
||||
typedef_nid = _make_id(stem, typedef_name)
|
||||
add_node(typedef_nid, typedef_name)
|
||||
add_edge(file_nid, typedef_nid, "defines")
|
||||
target_nid = _make_id(target_type)
|
||||
add_node(target_nid, target_type, source_file=None)
|
||||
add_edge(typedef_nid, target_nid, "references", context="typedef")
|
||||
|
||||
# 3. Extensions (extension MyExt on MyClass)
|
||||
ext_pattern = r"^\s{0,4}extension\s+(\w+)?(?:<[^>]+>)?\s+on\s+(\w+)"
|
||||
for m in re.finditer(ext_pattern, src_clean, re.MULTILINE):
|
||||
ext_name = m.group(1) or f"{stem}_anonymous_extension"
|
||||
target_class = m.group(2)
|
||||
|
||||
ext_nid = _make_id(stem, ext_name)
|
||||
label = m.group(1) or f"Extension on {target_class}"
|
||||
add_node(ext_nid, label)
|
||||
add_edge(file_nid, ext_nid, "defines")
|
||||
|
||||
target_nid = _make_id(target_class)
|
||||
add_node(target_nid, target_class, source_file=None)
|
||||
add_edge(ext_nid, target_nid, "extends")
|
||||
|
||||
# 4. Top-level and class-level variable declarations (generic variables, records, late, and destructuring)
|
||||
# Restrict indentation to 0-2 spaces to avoid matching local variables inside functions or switch expressions
|
||||
var_pattern = r"^\s{0,2}(?:late\s+)?(?:(?:final|const|var)\s+)?(?:\([^)]+\)\s+|([a-zA-Z0-9_<>,.?]+(?:\s+[a-zA-Z0-9_<>,.?]+){0,3})\s+)?(?:(\w+)|(?:\w+\s*)?\(([^)]+)\))\s*(?:=|$|;)"
|
||||
for m in re.finditer(var_pattern, src_clean, re.MULTILINE):
|
||||
var_type = m.group(1)
|
||||
single_name = m.group(2)
|
||||
destructured_names = m.group(3)
|
||||
|
||||
if not re.match(r"^\s*(?:late|final|const|var)\b", m.group(0)) and not var_type:
|
||||
continue
|
||||
|
||||
if single_name:
|
||||
if single_name not in {"if", "for", "while", "switch", "catch", "return"}:
|
||||
var_nid = _make_id(stem, single_name)
|
||||
add_node(var_nid, single_name)
|
||||
add_edge(file_nid, var_nid, "defines")
|
||||
|
||||
if var_type and var_type not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "List", "Map", "Set", "void"}:
|
||||
clean_type = var_type.split("<")[0].split(".")[-1].strip()
|
||||
type_nid = _make_id(clean_type)
|
||||
add_node(type_nid, clean_type, source_file=None)
|
||||
add_edge(file_nid, type_nid, "references", context="variable_type")
|
||||
elif destructured_names:
|
||||
for name in [n.strip() for n in destructured_names.split(",") if n.strip()]:
|
||||
if ":" in name:
|
||||
name = name.split(":")[-1].strip()
|
||||
if re.match(r"^[a-zA-Z_]\w*$", name) and not re.match(r"^[A-Z]", name):
|
||||
if name not in {"if", "for", "while", "switch", "catch", "return"}:
|
||||
var_nid = _make_id(stem, name)
|
||||
add_node(var_nid, name)
|
||||
add_edge(file_nid, var_nid, "defines")
|
||||
|
||||
# 5. Top-level and member functions/methods (supports typed/generic/record return types and Riverpod/Bloc references)
|
||||
# Restrict indentation to 0-2 spaces to avoid matching nested local functions or methods inside multiline switch statements
|
||||
method_pattern = r"^\s{0,2}(?:factory\s+|static\s+|async\s+|external\s+|abstract\s+)?(?:\([^)]+\)|[a-zA-Z0-9_<>,.?]+)(?:\s+[a-zA-Z0-9_<>,.?]+){0,3}\s+(\w+(?:\.\w+)?)\s*\("
|
||||
for m in re.finditer(method_pattern, src_clean, re.MULTILINE):
|
||||
raw_name = m.group(1)
|
||||
name = raw_name.split(".")[-1]
|
||||
if name in {"if", "for", "while", "switch", "catch", "return", "void", "dynamic", "final", "const", "get", "set"}:
|
||||
continue
|
||||
if re.match(r"^[A-Z]", name):
|
||||
continue
|
||||
nid = _make_id(stem, name)
|
||||
add_node(nid, name)
|
||||
add_edge(file_nid, nid, "defines")
|
||||
|
||||
# Get function body using matching brace to extract Riverpod reference patterns
|
||||
start_idx = m.start()
|
||||
brace_pos = src_clean.find("{", start_idx)
|
||||
semi_pos = src_clean.find(";", start_idx)
|
||||
arrow_pos = src_clean.find("=>", start_idx)
|
||||
|
||||
has_body = brace_pos != -1
|
||||
if has_body and semi_pos != -1 and semi_pos < brace_pos:
|
||||
has_body = False
|
||||
if has_body and arrow_pos != -1 and arrow_pos < brace_pos:
|
||||
has_body = False
|
||||
|
||||
if has_body:
|
||||
end_pos = _find_matching_brace(src_clean, start_idx)
|
||||
func_body = src_clean[brace_pos:end_pos]
|
||||
|
||||
# Extract Riverpod provider references: ref.watch(provider)
|
||||
for rm in re.finditer(r"\bref\.(?:watch|read|listen)\s*\(\s*(\w+)\b", func_body):
|
||||
provider_name = rm.group(1)
|
||||
provider_nid = _make_id(provider_name)
|
||||
add_node(provider_nid, provider_name, source_file=None)
|
||||
add_edge(nid, provider_nid, "references", context="riverpod_reference")
|
||||
|
||||
# Extract Bloc event additions: widget.add(MyEvent()) or bloc.add(MyEvent())
|
||||
for am in re.finditer(r"\b(?:\w*[Bb]loc\w*|context\.read<\w+>\(\))\.add\(\s*(?:const\s+)?([A-Z]\w*)\b", func_body):
|
||||
event_name = am.group(1)
|
||||
if event_name not in {"String", "List", "Map", "Set", "Future", "Stream", "Object"}:
|
||||
event_nid = _make_id(event_name)
|
||||
add_node(event_nid, event_name, source_file=None)
|
||||
add_edge(nid, event_nid, "calls", context="bloc_add_event")
|
||||
|
||||
# context.read<MyBloc>() or BlocProvider.of<MyBloc>(context)
|
||||
for lm in re.finditer(r"\b(?:read|watch|select|of)\s*<([a-zA-Z0-9_]+)>", func_body):
|
||||
bloc_name = lm.group(1)
|
||||
if bloc_name not in {"String", "int", "double", "bool", "num", "dynamic", "Object", "void"}:
|
||||
bloc_nid = _make_id(bloc_name)
|
||||
add_node(bloc_nid, bloc_name, source_file=None)
|
||||
add_edge(nid, bloc_nid, "references", context="bloc_lookup")
|
||||
|
||||
# Universal Navigation Patters (GoRouter, AutoRoute, Navigator)
|
||||
for nm in re.finditer(r"\b(?:go|push|goNamed|pushNamed|replace|replaceNamed)\s*\(\s*(?:context\s*,\s*)?['\"]([a-zA-Z0-9_/?=&%-]+)['\"]", func_body):
|
||||
route_path = nm.group(1)
|
||||
route_nid = _make_id("route", route_path.replace("/", "_").replace("?", "_").replace("=", "_").replace("&", "_"))
|
||||
add_node(route_nid, f"Route {route_path}", ftype="concept", source_file=None)
|
||||
add_edge(nid, route_nid, "navigates", context="route_path")
|
||||
|
||||
for cm in re.finditer(r"\b(?:go|push|goNamed|pushNamed|replace|replaceNamed)\s*\(\s*(?:context\s*,\s*)?([A-Z][a-zA-Z0-9_]*\.[a-zA-Z0-9_]+)", func_body):
|
||||
route_const = cm.group(1)
|
||||
route_nid = _make_id("route", route_const.replace(".", "_"))
|
||||
add_node(route_nid, route_const, ftype="concept", source_file=None)
|
||||
add_edge(nid, route_nid, "navigates", context="route_const")
|
||||
|
||||
for om in re.finditer(r"\b(?:push|replace)\s*\(\s*(?:context\s*,\s*)?.*?\b([A-Z]\w*(?:Route|Screen|Page))\b", func_body):
|
||||
route_class = om.group(1)
|
||||
route_nid = _make_id(route_class)
|
||||
add_node(route_nid, route_class, source_file=None)
|
||||
add_edge(nid, route_nid, "navigates", context="route_object")
|
||||
|
||||
# 6. Imports and Exports
|
||||
for m in re.finditer(r"""^\s*import\s+['"]([^'"]+)['"]""", src_clean, re.MULTILINE):
|
||||
pkg = m.group(1)
|
||||
tgt_nid = _make_id(pkg)
|
||||
add_node(tgt_nid, pkg, source_file=None)
|
||||
add_edge(file_nid, tgt_nid, "imports")
|
||||
|
||||
for m in re.finditer(r"""^\s*export\s+['"]([^'"]+)['"]""", src_clean, re.MULTILINE):
|
||||
pkg = m.group(1)
|
||||
tgt_nid = _make_id(pkg)
|
||||
add_node(tgt_nid, pkg, source_file=None)
|
||||
add_edge(file_nid, tgt_nid, "exports")
|
||||
|
||||
# 7. Generic Invocations / Type Lookups (Universal Dependency Lookup)
|
||||
# Matches any method call with type parameters: methodName<Type>() or object.methodName<Type>()
|
||||
# Automatically extracts GetIt, Injectable, Riverpod, Provider, BlocProvider, and InheritedWidget type lookups!
|
||||
generic_call_pattern = r"\b\w+<([a-zA-Z0-9_.]+(?:<[a-zA-Z0-9_.,\s<>]+>)?)\s*>\s*\("
|
||||
type_blacklist = {"String", "int", "double", "bool", "num", "dynamic", "Object", "List", "Map", "Set", "Future", "Stream", "void"}
|
||||
for m in re.finditer(generic_call_pattern, src_clean):
|
||||
type_name = m.group(1).split(".")[-1].strip()
|
||||
clean_name = type_name.split("<")[0].strip()
|
||||
if clean_name not in type_blacklist:
|
||||
target_nid = _make_id(clean_name)
|
||||
add_node(target_nid, clean_name, source_file=None)
|
||||
add_edge(file_nid, target_nid, "references", context="type_lookup")
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
@@ -0,0 +1,494 @@
|
||||
"""Dm extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from graphify.extractors.base import _file_stem, _make_id, _read_text
|
||||
|
||||
|
||||
def extract_dm(path: Path) -> dict:
|
||||
"""Extract types, procs, includes, and calls from a .dm/.dme file."""
|
||||
try:
|
||||
import tree_sitter_dm as tsdm
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree-sitter-dm not installed"}
|
||||
try:
|
||||
language = Language(tsdm.language())
|
||||
parser = Parser(language)
|
||||
source = path.read_bytes()
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
stem = _file_stem(path)
|
||||
str_path = str(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
function_bodies: list[tuple[str, Any, "str | None"]] = []
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid and nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({"id": nid, "label": label, "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line}"})
|
||||
|
||||
def add_edge(src: str, tgt: str, relation: str, line: int,
|
||||
confidence: str = "EXTRACTED", weight: float = 1.0,
|
||||
context: str | None = None) -> None:
|
||||
if not src or not tgt or src == tgt:
|
||||
return
|
||||
edge: dict = {"source": src, "target": tgt, "relation": relation,
|
||||
"confidence": confidence, "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": weight}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
def _type_path_text(node) -> str:
|
||||
return _read_text(node, source).strip()
|
||||
|
||||
def _ensure_type(path_text: str, line: int) -> str:
|
||||
nid = _make_id(stem, path_text)
|
||||
add_node(nid, path_text, line)
|
||||
return nid
|
||||
|
||||
def _find_child(node, type_name: str):
|
||||
for c in node.children:
|
||||
if c.type == type_name:
|
||||
return c
|
||||
return None
|
||||
|
||||
def _read_include_path(file_node) -> str:
|
||||
if file_node is None:
|
||||
return ""
|
||||
if file_node.type == "string_literal":
|
||||
parts = []
|
||||
for c in file_node.children:
|
||||
if c.type == "string_content":
|
||||
parts.append(_read_text(c, source))
|
||||
return "".join(parts)
|
||||
return _read_text(file_node, source).strip("'\"")
|
||||
|
||||
def walk(node, parent_type_path: "str | None" = None,
|
||||
parent_type_nid: "str | None" = None) -> None:
|
||||
t = node.type
|
||||
line = node.start_point[0] + 1
|
||||
|
||||
if t == "preproc_include":
|
||||
file_node = node.child_by_field_name("file")
|
||||
raw = _read_include_path(file_node)
|
||||
if raw:
|
||||
norm = raw.replace("\\", "/").lstrip("./")
|
||||
resolved = (path.parent / norm).resolve()
|
||||
edge: dict = {
|
||||
"source": file_nid,
|
||||
"target": _make_id(str(resolved)) if resolved.exists() else _make_id(norm),
|
||||
"relation": "imports_from" if resolved.exists() else "imports",
|
||||
"context": "import",
|
||||
"confidence": "EXTRACTED",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
"weight": 1.0,
|
||||
}
|
||||
if not resolved.exists():
|
||||
edge["external"] = True
|
||||
edges.append(edge)
|
||||
return
|
||||
|
||||
if t == "type_definition":
|
||||
tp_node = _find_child(node, "type_path")
|
||||
if tp_node is None:
|
||||
return
|
||||
type_path_str = _type_path_text(tp_node)
|
||||
type_nid = _ensure_type(type_path_str, line)
|
||||
add_edge(file_nid, type_nid, "contains", line)
|
||||
body = _find_child(node, "type_body")
|
||||
if body is not None:
|
||||
for c in body.children:
|
||||
walk(c, parent_type_path=type_path_str, parent_type_nid=type_nid)
|
||||
return
|
||||
|
||||
if t in ("type_body_intended", "type_body_braced"):
|
||||
for c in node.children:
|
||||
walk(c, parent_type_path, parent_type_nid)
|
||||
return
|
||||
|
||||
if t in ("type_proc_definition", "type_proc_override"):
|
||||
if parent_type_nid is None or parent_type_path is None:
|
||||
return
|
||||
name_node = node.child_by_field_name("name")
|
||||
if name_node is None:
|
||||
return
|
||||
proc_name = _read_text(name_node, source)
|
||||
proc_nid = _make_id(stem, parent_type_path, proc_name)
|
||||
add_node(proc_nid, f"{parent_type_path}/{proc_name}()", line)
|
||||
add_edge(parent_type_nid, proc_nid, "method", line)
|
||||
block = _find_child(node, "block")
|
||||
if block is not None:
|
||||
function_bodies.append((proc_nid, block, parent_type_path))
|
||||
return
|
||||
|
||||
if t in ("proc_definition", "proc_override"):
|
||||
tp_node = _find_child(node, "type_path")
|
||||
owner_path: "str | None" = None
|
||||
owner_nid: "str | None" = None
|
||||
if tp_node is not None:
|
||||
owner_path = _type_path_text(tp_node)
|
||||
owner_nid = _ensure_type(owner_path, line)
|
||||
add_edge(file_nid, owner_nid, "contains", line)
|
||||
name_node = node.child_by_field_name("name")
|
||||
if name_node is None:
|
||||
return
|
||||
proc_name = _read_text(name_node, source)
|
||||
if owner_path and owner_nid:
|
||||
proc_nid = _make_id(stem, owner_path, proc_name)
|
||||
add_node(proc_nid, f"{owner_path}/{proc_name}()", line)
|
||||
add_edge(owner_nid, proc_nid, "method", line)
|
||||
else:
|
||||
proc_nid = _make_id(stem, proc_name)
|
||||
add_node(proc_nid, f"{proc_name}()", line)
|
||||
add_edge(file_nid, proc_nid, "contains", line)
|
||||
block = _find_child(node, "block")
|
||||
if block is not None:
|
||||
function_bodies.append((proc_nid, block, owner_path))
|
||||
return
|
||||
|
||||
if t in ("operator_override", "type_operator_override"):
|
||||
return
|
||||
|
||||
for child in node.children:
|
||||
walk(child, parent_type_path, parent_type_nid)
|
||||
|
||||
walk(root)
|
||||
|
||||
label_to_nids: dict[str, list[str]] = {}
|
||||
path_to_nids: dict[str, list[str]] = {}
|
||||
for n in nodes:
|
||||
label = n["label"].strip("()")
|
||||
last = label.rsplit("/", 1)[-1] if "/" in label else label
|
||||
if last:
|
||||
label_to_nids.setdefault(last.lower(), []).append(n["id"])
|
||||
if label.startswith("/"):
|
||||
path_to_nids.setdefault(label.lower(), []).append(n["id"])
|
||||
|
||||
seen_call_pairs: set[tuple[str, str]] = set()
|
||||
raw_calls: list[dict] = []
|
||||
|
||||
def _emit_call(caller_nid: str, callee: str, line: int, is_member: bool) -> None:
|
||||
candidates = label_to_nids.get(callee.lower(), [])
|
||||
tgt_nid = candidates[0] if len(candidates) == 1 else None
|
||||
if tgt_nid and tgt_nid != caller_nid:
|
||||
pair = (caller_nid, tgt_nid)
|
||||
if pair in seen_call_pairs:
|
||||
return
|
||||
seen_call_pairs.add(pair)
|
||||
edges.append({
|
||||
"source": caller_nid, "target": tgt_nid, "relation": "calls",
|
||||
"context": "call", "confidence": "EXTRACTED",
|
||||
"source_file": str_path, "source_location": f"L{line}", "weight": 1.0,
|
||||
})
|
||||
else:
|
||||
raw_calls.append({
|
||||
"caller_nid": caller_nid, "callee": callee,
|
||||
"is_member_call": is_member, "source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
})
|
||||
|
||||
def walk_calls(body_node, caller_nid: str) -> None:
|
||||
if body_node is None:
|
||||
return
|
||||
t = body_node.type
|
||||
if t in ("proc_definition", "proc_override", "type_proc_definition",
|
||||
"type_proc_override", "type_definition"):
|
||||
return
|
||||
if t == "call_expression":
|
||||
name_node = body_node.child_by_field_name("name")
|
||||
if name_node is not None:
|
||||
callee = _read_text(name_node, source)
|
||||
if callee and callee != "..":
|
||||
_emit_call(caller_nid, callee, body_node.start_point[0] + 1,
|
||||
is_member=False)
|
||||
elif t == "field_proc_expression":
|
||||
proc_field = body_node.child_by_field_name("proc")
|
||||
if proc_field is not None:
|
||||
callee = _read_text(proc_field, source)
|
||||
if callee:
|
||||
_emit_call(caller_nid, callee, body_node.start_point[0] + 1,
|
||||
is_member=True)
|
||||
elif t == "new_expression":
|
||||
tp_node = _find_child(body_node, "type_path")
|
||||
if tp_node is not None:
|
||||
target_text = _type_path_text(tp_node)
|
||||
candidates = path_to_nids.get(target_text.lower(), [])
|
||||
tgt_nid = candidates[0] if len(candidates) == 1 else None
|
||||
if tgt_nid and tgt_nid != caller_nid:
|
||||
pair = (caller_nid, tgt_nid)
|
||||
if pair not in seen_call_pairs:
|
||||
seen_call_pairs.add(pair)
|
||||
edges.append({
|
||||
"source": caller_nid, "target": tgt_nid,
|
||||
"relation": "instantiates", "context": "call",
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{body_node.start_point[0] + 1}",
|
||||
"weight": 1.0,
|
||||
})
|
||||
for child in body_node.children:
|
||||
walk_calls(child, caller_nid)
|
||||
|
||||
for proc_nid, block, _owner_path in function_bodies:
|
||||
walk_calls(block, proc_nid)
|
||||
|
||||
return {"nodes": nodes, "edges": edges, "raw_calls": raw_calls}
|
||||
|
||||
def _read_dmi_description(data: bytes) -> str:
|
||||
"""Pull the BYOND metadata text out of a .dmi PNG, or empty string on failure."""
|
||||
import struct
|
||||
import zlib as _zlib
|
||||
if not data.startswith(b"\x89PNG\r\n\x1a\n"):
|
||||
return ""
|
||||
i = 8
|
||||
while i + 8 <= len(data):
|
||||
length = struct.unpack(">I", data[i:i + 4])[0]
|
||||
chunk_type = data[i + 4:i + 8]
|
||||
payload = data[i + 8:i + 8 + length]
|
||||
if chunk_type in (b"tEXt", b"zTXt"):
|
||||
try:
|
||||
null = payload.index(b"\x00")
|
||||
except ValueError:
|
||||
return ""
|
||||
keyword = payload[:null]
|
||||
if keyword == b"Description":
|
||||
if chunk_type == b"zTXt":
|
||||
return _zlib.decompressobj().decompress(payload[null + 2:], max_length=1024 * 1024).decode("utf-8", errors="replace")
|
||||
return payload[null + 1:].decode("utf-8", errors="replace")
|
||||
i += 8 + length + 4
|
||||
return ""
|
||||
|
||||
def extract_dmi(path: Path) -> dict:
|
||||
"""Extract icon state names from a .dmi (BYOND PNG icon sheet)."""
|
||||
try:
|
||||
data = path.read_bytes()
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
str_path = str(path)
|
||||
stem = _file_stem(path)
|
||||
file_nid = _make_id(str(path))
|
||||
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
|
||||
"source_file": str_path, "source_location": "L1"}]
|
||||
edges: list[dict] = []
|
||||
seen: set[str] = {file_nid}
|
||||
|
||||
description = _read_dmi_description(data)
|
||||
if not description:
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
|
||||
line_no = 0
|
||||
for raw_line in description.splitlines():
|
||||
line_no += 1
|
||||
stripped = raw_line.strip()
|
||||
if not stripped.startswith("state ="):
|
||||
continue
|
||||
value = stripped.split("=", 1)[1].strip()
|
||||
if value.startswith('"') and value.endswith('"') and len(value) >= 2:
|
||||
state_name = value[1:-1]
|
||||
else:
|
||||
state_name = value
|
||||
if not state_name:
|
||||
continue
|
||||
nid = _make_id(stem, "state", state_name)
|
||||
if nid in seen:
|
||||
continue
|
||||
seen.add(nid)
|
||||
nodes.append({"id": nid, "label": f'"{state_name}"', "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line_no}"})
|
||||
edges.append({"source": file_nid, "target": nid, "relation": "contains",
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{line_no}", "weight": 1.0})
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
|
||||
_DMM_GRID_RE = re.compile(r"^\(\s*\d+\s*,\s*\d+\s*,\s*\d+\s*\)\s*=", re.MULTILINE)
|
||||
|
||||
def _split_dmm_tile(body: str) -> list[str]:
|
||||
out: list[str] = []
|
||||
buf: list[str] = []
|
||||
depth = 0
|
||||
in_string = False
|
||||
escape = False
|
||||
for ch in body:
|
||||
if escape:
|
||||
buf.append(ch)
|
||||
escape = False
|
||||
continue
|
||||
if in_string:
|
||||
buf.append(ch)
|
||||
if ch == "\\":
|
||||
escape = True
|
||||
elif ch == '"':
|
||||
in_string = False
|
||||
continue
|
||||
if ch == '"':
|
||||
in_string = True
|
||||
buf.append(ch)
|
||||
elif ch in "({[":
|
||||
depth += 1
|
||||
buf.append(ch)
|
||||
elif ch in ")}]":
|
||||
depth -= 1
|
||||
buf.append(ch)
|
||||
elif ch == "," and depth == 0:
|
||||
out.append("".join(buf).strip())
|
||||
buf = []
|
||||
else:
|
||||
buf.append(ch)
|
||||
tail = "".join(buf).strip()
|
||||
if tail:
|
||||
out.append(tail)
|
||||
return out
|
||||
|
||||
def _dmm_type_path(entry: str) -> str:
|
||||
brace = entry.find("{")
|
||||
if brace != -1:
|
||||
entry = entry[:brace]
|
||||
return entry.strip()
|
||||
|
||||
def extract_dmm(path: Path) -> dict:
|
||||
"""Extract type-path references from a .dmm map file's tile dictionary."""
|
||||
try:
|
||||
if path.stat().st_size > 50 * 1024 * 1024:
|
||||
return {"nodes": [], "edges": [], "error": "file too large (>50 MB)"}
|
||||
text = path.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
str_path = str(path)
|
||||
file_nid = _make_id(str(path))
|
||||
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
|
||||
"source_file": str_path, "source_location": "L1"}]
|
||||
edges: list[dict] = []
|
||||
|
||||
grid_match = _DMM_GRID_RE.search(text)
|
||||
dict_text = text[:grid_match.start()] if grid_match else text
|
||||
|
||||
seen_targets: set[str] = set()
|
||||
buf: list[str] = []
|
||||
open_line = 0
|
||||
depth = 0
|
||||
in_string = False
|
||||
escape = False
|
||||
for line_idx, line in enumerate(dict_text.splitlines(), start=1):
|
||||
for ch in line:
|
||||
if escape:
|
||||
escape = False
|
||||
elif in_string:
|
||||
if ch == "\\":
|
||||
escape = True
|
||||
elif ch == '"':
|
||||
in_string = False
|
||||
elif ch == '"':
|
||||
in_string = True
|
||||
elif ch == "(":
|
||||
if depth == 0:
|
||||
open_line = line_idx
|
||||
depth += 1
|
||||
elif ch == ")":
|
||||
depth -= 1
|
||||
buf.append(ch)
|
||||
buf.append("\n")
|
||||
if depth == 0 and buf:
|
||||
chunk = "".join(buf)
|
||||
buf = []
|
||||
lp = chunk.find("(")
|
||||
rp = chunk.rfind(")")
|
||||
if lp == -1 or rp == -1 or rp <= lp:
|
||||
continue
|
||||
inner = chunk[lp + 1:rp]
|
||||
for entry in _split_dmm_tile(inner):
|
||||
tpath = _dmm_type_path(entry)
|
||||
if not tpath.startswith("/"):
|
||||
continue
|
||||
tgt = _make_id(tpath)
|
||||
if tgt in seen_targets:
|
||||
continue
|
||||
seen_targets.add(tgt)
|
||||
edges.append({"source": file_nid, "target": tgt, "relation": "uses",
|
||||
"context": "map", "confidence": "EXTRACTED",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{open_line}", "weight": 1.0})
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
|
||||
_DMF_WINDOW_RE = re.compile(r'^\s*window\s+"([^"]+)"\s*$')
|
||||
|
||||
_DMF_ELEM_RE = re.compile(r'^\s*elem\s+"([^"]+)"\s*$')
|
||||
|
||||
_DMF_TYPE_RE = re.compile(r'^\s*type\s*=\s*(\S+)\s*$')
|
||||
|
||||
def extract_dmf(path: Path) -> dict:
|
||||
"""Extract windows and controls from a .dmf interface file."""
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
str_path = str(path)
|
||||
stem = _file_stem(path)
|
||||
file_nid = _make_id(str(path))
|
||||
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
|
||||
"source_file": str_path, "source_location": "L1"}]
|
||||
edges: list[dict] = []
|
||||
seen: set[str] = {file_nid}
|
||||
|
||||
current_window_nid: str | None = None
|
||||
current_elem_nid: str | None = None
|
||||
current_elem_name: str | None = None
|
||||
|
||||
for line_idx, line in enumerate(text.splitlines(), start=1):
|
||||
m = _DMF_WINDOW_RE.match(line)
|
||||
if m:
|
||||
name = m.group(1)
|
||||
nid = _make_id(stem, "window", name)
|
||||
if nid not in seen:
|
||||
seen.add(nid)
|
||||
nodes.append({"id": nid, "label": f'window "{name}"', "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line_idx}"})
|
||||
edges.append({"source": file_nid, "target": nid, "relation": "contains",
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{line_idx}", "weight": 1.0})
|
||||
current_window_nid = nid
|
||||
current_elem_nid = None
|
||||
current_elem_name = None
|
||||
continue
|
||||
m = _DMF_ELEM_RE.match(line)
|
||||
if m and current_window_nid is not None:
|
||||
name = m.group(1)
|
||||
nid = _make_id(stem, "elem", current_window_nid, name)
|
||||
if nid not in seen:
|
||||
seen.add(nid)
|
||||
nodes.append({"id": nid, "label": f'elem "{name}"', "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line_idx}"})
|
||||
edges.append({"source": current_window_nid, "target": nid,
|
||||
"relation": "contains", "confidence": "EXTRACTED",
|
||||
"source_file": str_path, "source_location": f"L{line_idx}",
|
||||
"weight": 1.0})
|
||||
current_elem_nid = nid
|
||||
current_elem_name = name
|
||||
continue
|
||||
m = _DMF_TYPE_RE.match(line)
|
||||
if m and current_elem_nid is not None and current_elem_name is not None:
|
||||
ctype = m.group(1)
|
||||
for n in nodes:
|
||||
if n["id"] == current_elem_nid and " [" not in n["label"]:
|
||||
n["label"] = f'elem "{current_elem_name}" [{ctype}]'
|
||||
break
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
@@ -0,0 +1,309 @@
|
||||
"""Fortran extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
from pathlib import Path
|
||||
from graphify.extractors.base import _file_stem, _make_id, _read_text
|
||||
|
||||
|
||||
_FORTRAN_CPP_EXTS = {".F", ".F90", ".F95", ".F03", ".F08"}
|
||||
|
||||
def _cpp_preprocess(path: Path) -> bytes:
|
||||
"""Run cpp -w -P on a capital-F Fortran file and return preprocessed bytes.
|
||||
|
||||
Falls back to raw file bytes if cpp is not available. Capital-F extensions
|
||||
conventionally require C preprocessor expansion (#ifdef MPI, #define REAL8, etc.)
|
||||
before parsing.
|
||||
|
||||
Security (F-007): we pass `-nostdinc` and `-I /dev/null` so a malicious
|
||||
source file containing `#include "/home/victim/.ssh/id_rsa"` (or any other
|
||||
include directive) cannot inline arbitrary host files into the output that
|
||||
we then ship to an LLM. Without these flags `cpp` happily resolves any
|
||||
relative or absolute include path it can read, which is a corpus-side
|
||||
file-exfiltration vector.
|
||||
"""
|
||||
import shutil
|
||||
import subprocess
|
||||
if not shutil.which("cpp"):
|
||||
return path.read_bytes()
|
||||
try:
|
||||
# Pass an absolute path so a corpus file named like "-I/etc/x.F90" cannot
|
||||
# be parsed by cpp as an option (cpp does not accept a "--" end-of-options
|
||||
# terminator). An absolute path always begins with "/".
|
||||
result = subprocess.run(
|
||||
["cpp", "-w", "-P", "-nostdinc", "-I", "/dev/null", str(path.resolve())],
|
||||
capture_output=True,
|
||||
timeout=30,
|
||||
)
|
||||
if result.returncode == 0 and result.stdout:
|
||||
return result.stdout
|
||||
except Exception:
|
||||
pass
|
||||
return path.read_bytes()
|
||||
|
||||
def extract_fortran(path: Path) -> dict:
|
||||
"""Extract programs, modules, subroutines, functions, use statements, and calls from Fortran files.
|
||||
|
||||
Capital-F extensions (.F, .F90, etc.) are run through the C preprocessor before
|
||||
parsing so #ifdef/#define macros are resolved.
|
||||
"""
|
||||
try:
|
||||
import tree_sitter_fortran as tsfortran
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree-sitter-fortran not installed"}
|
||||
|
||||
try:
|
||||
language = Language(tsfortran.language())
|
||||
parser = Parser(language)
|
||||
source = _cpp_preprocess(path) if path.suffix in _FORTRAN_CPP_EXTS else path.read_bytes()
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
stem = _file_stem(path)
|
||||
str_path = str(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
scope_bodies: list[tuple[str, object]] = []
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid,
|
||||
"label": label,
|
||||
"file_type": "code",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
})
|
||||
|
||||
def add_edge(src: str, tgt: str, relation: str, line: int,
|
||||
confidence: str = "EXTRACTED", weight: float = 1.0,
|
||||
context: str | None = None) -> None:
|
||||
edge = {
|
||||
"source": src,
|
||||
"target": tgt,
|
||||
"relation": relation,
|
||||
"confidence": confidence,
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
"weight": weight,
|
||||
}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
def _fortran_name(stmt_node) -> str | None:
|
||||
"""Extract name from a *_statement node. Fortran is case-insensitive; lowercase."""
|
||||
for child in stmt_node.children:
|
||||
if child.type in ("name", "identifier"):
|
||||
return _read_text(child, source).lower()
|
||||
return None
|
||||
|
||||
def ensure_named_node(name: str, line: int) -> str:
|
||||
nid = _make_id(stem, name)
|
||||
if nid in seen_ids:
|
||||
return nid
|
||||
nid = _make_id(name)
|
||||
if nid not in seen_ids:
|
||||
# The name isn't defined in this file, so this is a cross-file reference
|
||||
# (e.g. a `Thing` type annotation imported from another module). Emit a
|
||||
# SOURCELESS stub — like the inheritance-base path below — so the
|
||||
# corpus-level rewire can collapse it onto the real definition. A sourced
|
||||
# stub here makes _disambiguate_colliding_node_ids bake the referencing
|
||||
# file's path (with extension) into the id and blocks the rewire, which is
|
||||
# the phantom-duplicate-node bug (#1402).
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid,
|
||||
"label": name,
|
||||
"file_type": "code",
|
||||
"source_file": "",
|
||||
"source_location": "",
|
||||
"origin_file": str_path,
|
||||
})
|
||||
return nid
|
||||
|
||||
def emit_signature_refs(scope_node, fn_nid: str, is_function: bool) -> None:
|
||||
"""Emit references[parameter_type] / references[return_type] edges for
|
||||
a subroutine/function based on its variable_declaration siblings."""
|
||||
stmt_type = "function_statement" if is_function else "subroutine_statement"
|
||||
stmt = next((c for c in scope_node.children if c.type == stmt_type), None)
|
||||
if stmt is None:
|
||||
return
|
||||
param_names: set[str] = set()
|
||||
params_node = next((c for c in stmt.children if c.type == "parameters"), None)
|
||||
if params_node is not None:
|
||||
for c in params_node.children:
|
||||
if c.type == "identifier":
|
||||
param_names.add(_read_text(c, source).lower())
|
||||
result_name: str | None = None
|
||||
if is_function:
|
||||
result_node = next((c for c in stmt.children if c.type == "function_result"), None)
|
||||
if result_node is not None:
|
||||
res_id = next((c for c in result_node.children if c.type == "identifier"), None)
|
||||
if res_id is not None:
|
||||
result_name = _read_text(res_id, source).lower()
|
||||
else:
|
||||
# implicit result variable: same name as the function
|
||||
result_name = _fortran_name(stmt)
|
||||
for child in scope_node.children:
|
||||
if child.type != "variable_declaration":
|
||||
continue
|
||||
derived = next((c for c in child.children if c.type == "derived_type"), None)
|
||||
if derived is None:
|
||||
continue
|
||||
type_name_node = next((c for c in derived.children if c.type == "type_name"), None)
|
||||
if type_name_node is None:
|
||||
continue
|
||||
type_name = _read_text(type_name_node, source).lower()
|
||||
for var in child.children:
|
||||
if var.type != "identifier":
|
||||
continue
|
||||
var_name = _read_text(var, source).lower()
|
||||
var_line = var.start_point[0] + 1
|
||||
if var_name in param_names:
|
||||
tgt = ensure_named_node(type_name, var_line)
|
||||
if tgt != fn_nid:
|
||||
add_edge(fn_nid, tgt, "references", var_line, context="parameter_type")
|
||||
elif is_function and var_name == result_name:
|
||||
tgt = ensure_named_node(type_name, var_line)
|
||||
if tgt != fn_nid:
|
||||
add_edge(fn_nid, tgt, "references", var_line, context="return_type")
|
||||
|
||||
def walk_calls(node, scope_nid: str) -> None:
|
||||
if node is None:
|
||||
return
|
||||
t = node.type
|
||||
if t in ("subroutine", "function", "module", "program", "internal_procedures"):
|
||||
return
|
||||
# call FOO(args) — tree-sitter-fortran uses subroutine_call
|
||||
if t == "subroutine_call":
|
||||
name_node = next((c for c in node.children if c.type == "identifier"), None)
|
||||
if name_node:
|
||||
callee = _read_text(name_node, source).lower()
|
||||
target_nid = _make_id(stem, callee)
|
||||
add_edge(scope_nid, target_nid, "calls", node.start_point[0] + 1,
|
||||
confidence="EXTRACTED", context="call")
|
||||
# x = compute(args) — function invocations are `call_expression`, which
|
||||
# shares Fortran's `name(...)` syntax with array indexing. Only emit a
|
||||
# call edge when the callee resolves to a procedure defined in this file
|
||||
# (an array variable produces no matching node), so array accesses can't
|
||||
# fabricate spurious `calls` edges.
|
||||
elif t == "call_expression":
|
||||
name_node = next((c for c in node.children if c.type == "identifier"), None)
|
||||
if name_node:
|
||||
callee = _read_text(name_node, source).lower()
|
||||
target_nid = _make_id(stem, callee)
|
||||
if target_nid in seen_ids and target_nid != scope_nid:
|
||||
add_edge(scope_nid, target_nid, "calls", node.start_point[0] + 1,
|
||||
confidence="EXTRACTED", context="call")
|
||||
for child in node.children:
|
||||
walk_calls(child, scope_nid)
|
||||
|
||||
def walk(node, scope_nid: str) -> None:
|
||||
t = node.type
|
||||
|
||||
if t == "program":
|
||||
stmt = next((c for c in node.children if c.type == "program_statement"), None)
|
||||
name = _fortran_name(stmt) if stmt else None
|
||||
if name:
|
||||
nid = _make_id(stem, name)
|
||||
line = node.start_point[0] + 1
|
||||
add_node(nid, name, line)
|
||||
add_edge(file_nid, nid, "defines", line)
|
||||
scope_bodies.append((nid, node))
|
||||
for child in node.children:
|
||||
walk(child, nid)
|
||||
return
|
||||
|
||||
if t == "module":
|
||||
stmt = next((c for c in node.children if c.type == "module_statement"), None)
|
||||
name = _fortran_name(stmt) if stmt else None
|
||||
if name:
|
||||
nid = _make_id(stem, name)
|
||||
line = node.start_point[0] + 1
|
||||
add_node(nid, name, line)
|
||||
add_edge(file_nid, nid, "defines", line)
|
||||
for child in node.children:
|
||||
walk(child, nid)
|
||||
return
|
||||
|
||||
# subroutines/functions inside a module live under internal_procedures
|
||||
if t == "internal_procedures":
|
||||
for child in node.children:
|
||||
walk(child, scope_nid)
|
||||
return
|
||||
|
||||
if t == "derived_type_definition":
|
||||
stmt = next((c for c in node.children if c.type == "derived_type_statement"), None)
|
||||
if stmt is not None:
|
||||
name_node = next((c for c in stmt.children if c.type == "type_name"), None)
|
||||
if name_node is not None:
|
||||
type_name = _read_text(name_node, source).lower()
|
||||
type_nid = _make_id(stem, type_name)
|
||||
line = node.start_point[0] + 1
|
||||
add_node(type_nid, type_name, line)
|
||||
add_edge(scope_nid, type_nid, "defines", line)
|
||||
return
|
||||
|
||||
if t == "subroutine":
|
||||
stmt = next((c for c in node.children if c.type == "subroutine_statement"), None)
|
||||
name = _fortran_name(stmt) if stmt else None
|
||||
if name:
|
||||
nid = _make_id(stem, name)
|
||||
line = node.start_point[0] + 1
|
||||
add_node(nid, f"{name}()", line)
|
||||
add_edge(scope_nid, nid, "defines", line)
|
||||
scope_bodies.append((nid, node))
|
||||
emit_signature_refs(node, nid, is_function=False)
|
||||
for child in node.children:
|
||||
walk(child, nid)
|
||||
return
|
||||
|
||||
if t == "function":
|
||||
stmt = next((c for c in node.children if c.type == "function_statement"), None)
|
||||
name = _fortran_name(stmt) if stmt else None
|
||||
if name:
|
||||
nid = _make_id(stem, name)
|
||||
line = node.start_point[0] + 1
|
||||
add_node(nid, f"{name}()", line)
|
||||
add_edge(scope_nid, nid, "defines", line)
|
||||
scope_bodies.append((nid, node))
|
||||
emit_signature_refs(node, nid, is_function=True)
|
||||
for child in node.children:
|
||||
walk(child, nid)
|
||||
return
|
||||
|
||||
if t == "use_statement":
|
||||
line = node.start_point[0] + 1
|
||||
# tree-sitter-fortran uses module_name node for the used module
|
||||
name_node = next((c for c in node.children if c.type in ("module_name", "name", "identifier")), None)
|
||||
if name_node:
|
||||
mod_name = _read_text(name_node, source).lower()
|
||||
imp_nid = _make_id(mod_name)
|
||||
add_node(imp_nid, mod_name, line)
|
||||
add_edge(scope_nid, imp_nid, "imports", line, context="use")
|
||||
return
|
||||
|
||||
for child in node.children:
|
||||
walk(child, scope_nid)
|
||||
|
||||
walk(root, file_nid)
|
||||
|
||||
_stmt_headers = {
|
||||
"subroutine_statement", "function_statement",
|
||||
"program_statement", "module_statement",
|
||||
}
|
||||
for scope_nid, body_node in scope_bodies:
|
||||
for child in body_node.children:
|
||||
if child.type not in _stmt_headers:
|
||||
walk_calls(child, scope_nid)
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
@@ -0,0 +1,396 @@
|
||||
"""Go extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
from pathlib import Path
|
||||
from graphify.extractors.base import _LANGUAGE_BUILTIN_GLOBALS, _file_stem, _make_id, _read_text
|
||||
|
||||
|
||||
_GO_PREDECLARED_TYPES = frozenset({
|
||||
"bool", "byte", "complex64", "complex128", "error", "float32", "float64",
|
||||
"int", "int8", "int16", "int32", "int64", "rune", "string",
|
||||
"uint", "uint8", "uint16", "uint32", "uint64", "uintptr", "any", "comparable",
|
||||
})
|
||||
|
||||
def _go_collect_type_refs(node, source: bytes, generic: bool, out: list[tuple[str, str]]) -> None:
|
||||
"""Walk a Go type expression; append (name, role) tuples."""
|
||||
if node is None:
|
||||
return
|
||||
t = node.type
|
||||
if t == "type_identifier":
|
||||
text = _read_text(node, source)
|
||||
if text and text not in _GO_PREDECLARED_TYPES:
|
||||
out.append((text, "generic_arg" if generic else "type"))
|
||||
return
|
||||
if t == "qualified_type":
|
||||
text = _read_text(node, source).rsplit(".", 1)[-1]
|
||||
if text and text not in _GO_PREDECLARED_TYPES:
|
||||
out.append((text, "generic_arg" if generic else "type"))
|
||||
return
|
||||
if t == "generic_type":
|
||||
type_field = node.child_by_field_name("type")
|
||||
if type_field is not None:
|
||||
sub: list[tuple[str, str]] = []
|
||||
_go_collect_type_refs(type_field, source, generic, sub)
|
||||
out.extend(sub)
|
||||
for c in node.children:
|
||||
if c.type == "type_arguments":
|
||||
for arg in c.children:
|
||||
if arg.is_named:
|
||||
_go_collect_type_refs(arg, source, True, out)
|
||||
return
|
||||
if t in ("pointer_type", "slice_type", "array_type", "map_type",
|
||||
"channel_type", "parenthesized_type"):
|
||||
for c in node.children:
|
||||
if c.is_named:
|
||||
_go_collect_type_refs(c, source, generic, out)
|
||||
return
|
||||
if node.is_named:
|
||||
for c in node.children:
|
||||
if c.is_named:
|
||||
_go_collect_type_refs(c, source, generic, out)
|
||||
|
||||
def extract_go(path: Path) -> dict:
|
||||
"""Extract functions, methods, type declarations, and imports from a .go file."""
|
||||
try:
|
||||
import tree_sitter_go as tsgo
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree-sitter-go not installed"}
|
||||
|
||||
try:
|
||||
language = Language(tsgo.language())
|
||||
parser = Parser(language)
|
||||
source = path.read_bytes()
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
stem = _file_stem(path)
|
||||
# Use directory name as package scope so methods on the same type across
|
||||
# multiple files in a package share one canonical type node.
|
||||
pkg_scope = path.parent.name or stem
|
||||
str_path = str(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
function_bodies: list[tuple[str, object]] = []
|
||||
go_imported_pkgs: set[str] = set() # local names of imported packages
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid,
|
||||
"label": label,
|
||||
"file_type": "code",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
})
|
||||
|
||||
def add_edge(src: str, tgt: str, relation: str, line: int,
|
||||
confidence: str = "EXTRACTED", weight: float = 1.0,
|
||||
context: str | None = None) -> None:
|
||||
edge = {
|
||||
"source": src,
|
||||
"target": tgt,
|
||||
"relation": relation,
|
||||
"confidence": confidence,
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
"weight": weight,
|
||||
}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
def ensure_named_node(name: str, line: int) -> str:
|
||||
nid = _make_id(pkg_scope, name)
|
||||
if nid in seen_ids:
|
||||
return nid
|
||||
nid = _make_id(name)
|
||||
if nid not in seen_ids:
|
||||
# The name isn't declared in this file, so this is a cross-file reference
|
||||
# (e.g. a type defined in another file of the package). Emit a SOURCELESS
|
||||
# stub — like the inheritance-base path in the other extractors — so the
|
||||
# corpus-level rewire can collapse it onto the real definition. A sourced
|
||||
# stub here makes _disambiguate_colliding_node_ids bake the referencing
|
||||
# file's path (with extension) into the id and blocks the rewire, which is
|
||||
# the phantom-duplicate-node bug (#1402).
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid,
|
||||
"label": name,
|
||||
"file_type": "code",
|
||||
"source_file": "",
|
||||
"source_location": "",
|
||||
"origin_file": str_path,
|
||||
})
|
||||
return nid
|
||||
|
||||
def emit_go_method_refs(func_node, func_nid: str, line: int) -> None:
|
||||
params = func_node.child_by_field_name("parameters")
|
||||
if params is not None:
|
||||
for p in params.children:
|
||||
if p.type != "parameter_declaration":
|
||||
continue
|
||||
type_node = p.child_by_field_name("type")
|
||||
refs: list[tuple[str, str]] = []
|
||||
_go_collect_type_refs(type_node, source, False, refs)
|
||||
for ref_name, role in refs:
|
||||
ctx = "generic_arg" if role == "generic_arg" else "parameter_type"
|
||||
tgt = ensure_named_node(ref_name, line)
|
||||
if tgt != func_nid:
|
||||
add_edge(func_nid, tgt, "references", line, context=ctx)
|
||||
result = func_node.child_by_field_name("result")
|
||||
if result is not None:
|
||||
if result.type == "parameter_list":
|
||||
for p in result.children:
|
||||
if p.type != "parameter_declaration":
|
||||
continue
|
||||
type_node = p.child_by_field_name("type")
|
||||
if type_node is None:
|
||||
for c in p.children:
|
||||
if c.is_named:
|
||||
type_node = c
|
||||
break
|
||||
refs = []
|
||||
_go_collect_type_refs(type_node, source, False, refs)
|
||||
for ref_name, role in refs:
|
||||
ctx = "generic_arg" if role == "generic_arg" else "return_type"
|
||||
tgt = ensure_named_node(ref_name, line)
|
||||
if tgt != func_nid:
|
||||
add_edge(func_nid, tgt, "references", line, context=ctx)
|
||||
else:
|
||||
refs = []
|
||||
_go_collect_type_refs(result, source, False, refs)
|
||||
for ref_name, role in refs:
|
||||
ctx = "generic_arg" if role == "generic_arg" else "return_type"
|
||||
tgt = ensure_named_node(ref_name, line)
|
||||
if tgt != func_nid:
|
||||
add_edge(func_nid, tgt, "references", line, context=ctx)
|
||||
|
||||
def walk(node) -> None:
|
||||
t = node.type
|
||||
|
||||
if t == "function_declaration":
|
||||
name_node = node.child_by_field_name("name")
|
||||
if name_node:
|
||||
func_name = _read_text(name_node, source)
|
||||
line = node.start_point[0] + 1
|
||||
func_nid = _make_id(stem, func_name)
|
||||
add_node(func_nid, f"{func_name}()", line)
|
||||
add_edge(file_nid, func_nid, "contains", line)
|
||||
emit_go_method_refs(node, func_nid, line)
|
||||
body = node.child_by_field_name("body")
|
||||
if body:
|
||||
function_bodies.append((func_nid, body))
|
||||
return
|
||||
|
||||
if t == "method_declaration":
|
||||
receiver = node.child_by_field_name("receiver")
|
||||
receiver_type: str | None = None
|
||||
if receiver:
|
||||
for param in receiver.children:
|
||||
if param.type == "parameter_declaration":
|
||||
type_node = param.child_by_field_name("type")
|
||||
if type_node:
|
||||
receiver_type = _read_text(type_node, source).lstrip("*").strip()
|
||||
break
|
||||
name_node = node.child_by_field_name("name")
|
||||
if not name_node:
|
||||
return
|
||||
method_name = _read_text(name_node, source)
|
||||
line = node.start_point[0] + 1
|
||||
|
||||
if receiver_type:
|
||||
parent_nid = _make_id(pkg_scope, receiver_type)
|
||||
add_node(parent_nid, receiver_type, line)
|
||||
method_nid = _make_id(parent_nid, method_name)
|
||||
add_node(method_nid, f".{method_name}()", line)
|
||||
add_edge(parent_nid, method_nid, "method", line)
|
||||
else:
|
||||
method_nid = _make_id(stem, method_name)
|
||||
add_node(method_nid, f"{method_name}()", line)
|
||||
add_edge(file_nid, method_nid, "contains", line)
|
||||
|
||||
emit_go_method_refs(node, method_nid, line)
|
||||
body = node.child_by_field_name("body")
|
||||
if body:
|
||||
function_bodies.append((method_nid, body))
|
||||
return
|
||||
|
||||
if t == "type_declaration":
|
||||
for child in node.children:
|
||||
if child.type != "type_spec":
|
||||
continue
|
||||
name_node = child.child_by_field_name("name")
|
||||
if not name_node:
|
||||
continue
|
||||
type_name = _read_text(name_node, source)
|
||||
line = child.start_point[0] + 1
|
||||
type_nid = _make_id(pkg_scope, type_name)
|
||||
add_node(type_nid, type_name, line)
|
||||
add_edge(file_nid, type_nid, "contains", line)
|
||||
# Type body: struct fields (with embeds) or interface embedding.
|
||||
type_body = None
|
||||
for tc in child.children:
|
||||
if tc.type in ("struct_type", "interface_type"):
|
||||
type_body = tc
|
||||
break
|
||||
if type_body is None:
|
||||
continue
|
||||
if type_body.type == "struct_type":
|
||||
for fdl in type_body.children:
|
||||
if fdl.type != "field_declaration_list":
|
||||
continue
|
||||
for field in fdl.children:
|
||||
if field.type != "field_declaration":
|
||||
continue
|
||||
has_name = any(
|
||||
fc.type == "field_identifier" for fc in field.children
|
||||
)
|
||||
type_node = field.child_by_field_name("type")
|
||||
if type_node is None:
|
||||
for fc in field.children:
|
||||
if fc.is_named and fc.type != "field_identifier":
|
||||
type_node = fc
|
||||
break
|
||||
refs: list[tuple[str, str]] = []
|
||||
_go_collect_type_refs(type_node, source, False, refs)
|
||||
for ref_name, role in refs:
|
||||
tgt = ensure_named_node(ref_name, field.start_point[0] + 1)
|
||||
if tgt == type_nid:
|
||||
continue
|
||||
if not has_name and role == "type":
|
||||
add_edge(type_nid, tgt, "embeds",
|
||||
field.start_point[0] + 1)
|
||||
else:
|
||||
ctx = "generic_arg" if role == "generic_arg" else "field"
|
||||
add_edge(type_nid, tgt, "references",
|
||||
field.start_point[0] + 1, context=ctx)
|
||||
elif type_body.type == "interface_type":
|
||||
for elem in type_body.children:
|
||||
if elem.type != "type_elem":
|
||||
continue
|
||||
refs = []
|
||||
for sub in elem.children:
|
||||
if sub.is_named:
|
||||
_go_collect_type_refs(sub, source, False, refs)
|
||||
for ref_name, role in refs:
|
||||
tgt = ensure_named_node(ref_name, elem.start_point[0] + 1)
|
||||
if tgt == type_nid:
|
||||
continue
|
||||
if role == "type":
|
||||
add_edge(type_nid, tgt, "embeds",
|
||||
elem.start_point[0] + 1)
|
||||
else:
|
||||
add_edge(type_nid, tgt, "references",
|
||||
elem.start_point[0] + 1, context="generic_arg")
|
||||
return
|
||||
|
||||
if t == "import_declaration":
|
||||
for child in node.children:
|
||||
if child.type == "import_spec_list":
|
||||
for spec in child.children:
|
||||
if spec.type == "import_spec":
|
||||
path_node = spec.child_by_field_name("path")
|
||||
if path_node:
|
||||
raw = _read_text(path_node, source).strip('"')
|
||||
# Prefix with go_pkg_ so stdlib names (e.g. "context")
|
||||
# don't collide with local files of the same basename.
|
||||
tgt_nid = _make_id("go", "pkg", raw)
|
||||
add_edge(file_nid, tgt_nid, "imports_from", spec.start_point[0] + 1, context="import")
|
||||
# Track local name (alias or last path segment)
|
||||
alias = spec.child_by_field_name("name")
|
||||
local_name = _read_text(alias, source) if alias else raw.split("/")[-1]
|
||||
if local_name and local_name != "_" and local_name != ".":
|
||||
go_imported_pkgs.add(local_name)
|
||||
elif child.type == "import_spec":
|
||||
path_node = child.child_by_field_name("path")
|
||||
if path_node:
|
||||
raw = _read_text(path_node, source).strip('"')
|
||||
tgt_nid = _make_id("go", "pkg", raw)
|
||||
add_edge(file_nid, tgt_nid, "imports_from", child.start_point[0] + 1, context="import")
|
||||
alias = child.child_by_field_name("name")
|
||||
local_name = _read_text(alias, source) if alias else raw.split("/")[-1]
|
||||
if local_name and local_name != "_" and local_name != ".":
|
||||
go_imported_pkgs.add(local_name)
|
||||
return
|
||||
|
||||
for child in node.children:
|
||||
walk(child)
|
||||
|
||||
walk(root)
|
||||
|
||||
label_to_nid: dict[str, str] = {}
|
||||
for n in nodes:
|
||||
raw = n["label"]
|
||||
normalised = raw.strip("()").lstrip(".")
|
||||
label_to_nid[normalised] = n["id"]
|
||||
|
||||
seen_call_pairs: set[tuple[str, str]] = set()
|
||||
raw_calls: list[dict] = []
|
||||
|
||||
def walk_calls(node, caller_nid: str) -> None:
|
||||
if node.type in ("function_declaration", "method_declaration"):
|
||||
return
|
||||
if node.type == "call_expression":
|
||||
func_node = node.child_by_field_name("function")
|
||||
callee_name: str | None = None
|
||||
is_member_call: bool = False
|
||||
if func_node:
|
||||
if func_node.type == "identifier":
|
||||
callee_name = _read_text(func_node, source)
|
||||
elif func_node.type == "selector_expression":
|
||||
field = func_node.child_by_field_name("field")
|
||||
operand = func_node.child_by_field_name("operand")
|
||||
receiver_name = _read_text(operand, source) if operand else ""
|
||||
# Package-qualified call (e.g. fmt.Println) → allow cross-file resolution.
|
||||
# Receiver method call (e.g. s.logger.Log) → skip, no import evidence.
|
||||
is_member_call = receiver_name not in go_imported_pkgs
|
||||
if field:
|
||||
callee_name = _read_text(field, source)
|
||||
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
|
||||
tgt_nid = label_to_nid.get(callee_name)
|
||||
if tgt_nid and tgt_nid != caller_nid:
|
||||
pair = (caller_nid, tgt_nid)
|
||||
if pair not in seen_call_pairs:
|
||||
seen_call_pairs.add(pair)
|
||||
line = node.start_point[0] + 1
|
||||
edges.append({
|
||||
"source": caller_nid,
|
||||
"target": tgt_nid,
|
||||
"relation": "calls",
|
||||
"context": "call",
|
||||
"confidence": "EXTRACTED",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
"weight": 1.0,
|
||||
})
|
||||
elif callee_name:
|
||||
raw_calls.append({
|
||||
"caller_nid": caller_nid,
|
||||
"callee": callee_name,
|
||||
"is_member_call": is_member_call,
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{node.start_point[0] + 1}",
|
||||
})
|
||||
for child in node.children:
|
||||
walk_calls(child, caller_nid)
|
||||
|
||||
for caller_nid, body_node in function_bodies:
|
||||
walk_calls(body_node, caller_nid)
|
||||
|
||||
valid_ids = seen_ids
|
||||
clean_edges = []
|
||||
for edge in edges:
|
||||
src, tgt = edge["source"], edge["target"]
|
||||
if src in valid_ids and (tgt in valid_ids or edge["relation"] in ("imports", "imports_from")):
|
||||
clean_edges.append(edge)
|
||||
|
||||
return {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls}
|
||||
@@ -0,0 +1,206 @@
|
||||
"""Json_config extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
from pathlib import Path
|
||||
from graphify.extractors.base import _file_stem, _make_id, _read_text
|
||||
|
||||
|
||||
_CONFIG_JSON_NAMES = frozenset({
|
||||
"package.json", "tsconfig.json", "jsconfig.json", "composer.json",
|
||||
"deno.json", "deno.jsonc", "bower.json", "manifest.json",
|
||||
"app.json", "now.json", "vercel.json", "angular.json", "nest-cli.json",
|
||||
"biome.json", "biome.jsonc", "renovate.json", ".babelrc", ".babelrc.json",
|
||||
".eslintrc.json", ".prettierrc.json", ".prettierrc", "babel.config.json",
|
||||
})
|
||||
|
||||
_CONFIG_JSON_KEYS = frozenset({
|
||||
"dependencies", "devDependencies", "peerDependencies",
|
||||
"optionalDependencies", "bundleDependencies", "bundledDependencies",
|
||||
"extends", "$ref", "$schema", "compilerOptions",
|
||||
})
|
||||
|
||||
def _is_config_json(path: Path, obj_node, source: bytes) -> bool:
|
||||
"""True if a .json file is a recognized config/manifest worth AST-extracting.
|
||||
|
||||
Matches by filename first (cheap), then falls back to a top-level key probe
|
||||
so arbitrarily-named config files (e.g. ``api.tsconfig.json``,
|
||||
``foo.eslintrc.json``) are still picked up. Returns False for data JSON so it
|
||||
is skipped by the structural pass (#1224)."""
|
||||
name = path.name.casefold()
|
||||
if name in _CONFIG_JSON_NAMES:
|
||||
return True
|
||||
# Common compound config names: *.eslintrc.json, *.prettierrc.json, etc.
|
||||
if name.endswith((".eslintrc.json", ".prettierrc.json", ".babelrc.json",
|
||||
"tsconfig.json", "jsconfig.json")):
|
||||
return True
|
||||
# Top-level key probe: scan the root object's immediate keys (no deep walk).
|
||||
for top_key in obj_node.children:
|
||||
if top_key.type != "pair":
|
||||
continue
|
||||
key_node = top_key.child_by_field_name("key")
|
||||
if key_node is None:
|
||||
continue
|
||||
kc = key_node.child_by_field_name("string_content")
|
||||
text = _read_text(kc, source) if kc else _read_text(key_node, source).strip('"\'')
|
||||
if text in _CONFIG_JSON_KEYS:
|
||||
return True
|
||||
return False
|
||||
|
||||
def extract_json(path: Path) -> dict:
|
||||
"""Extract structure and dependency edges from a *config/manifest* .json file.
|
||||
|
||||
Data-shaped JSON (eval fixtures, datasets, GeoJSON, API response dumps) is
|
||||
deliberately skipped — AST-walking it produced hundreds of orphan key-nodes
|
||||
and duplicate communities that swamped real structure (#1224). Recognition
|
||||
is by filename (package.json, tsconfig.json, …) or a top-level key probe
|
||||
(dependencies / extends / $ref / $schema / compilerOptions)."""
|
||||
_JSON_MAX_BYTES = 1_048_576 # 1 MiB — skip large fixture dumps / GeoJSON blobs
|
||||
|
||||
try:
|
||||
import tree_sitter_json as tsjson
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree-sitter-json not installed"}
|
||||
|
||||
try:
|
||||
# Bounded read instead of stat()+read() to eliminate TOCTOU (J-1):
|
||||
# read one byte beyond the limit so we can detect oversized files even
|
||||
# if the file grows between stat and read.
|
||||
with path.open("rb") as _f:
|
||||
source = _f.read(_JSON_MAX_BYTES + 1)
|
||||
if len(source) > _JSON_MAX_BYTES:
|
||||
return {"nodes": [], "edges": [], "error": "json file too large to index"}
|
||||
language = Language(tsjson.language())
|
||||
parser = Parser(language)
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
stem = _file_stem(path)
|
||||
str_path = str(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
|
||||
# Keys whose string values become imports (package.json dep blocks)
|
||||
_DEP_KEYS = frozenset({
|
||||
"dependencies", "devDependencies", "peerDependencies",
|
||||
"optionalDependencies", "bundleDependencies", "bundledDependencies",
|
||||
})
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid and nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({"id": nid, "label": label, "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line}"})
|
||||
|
||||
def add_edge(src: str, tgt: str, relation: str, line: int,
|
||||
context: str | None = None) -> None:
|
||||
if not src or not tgt or src == tgt:
|
||||
return
|
||||
edge = {"source": src, "target": tgt, "relation": relation,
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": 1.0}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
def _key_text(pair_node) -> str | None:
|
||||
"""Extract the string content of a pair's key."""
|
||||
key_node = pair_node.child_by_field_name("key")
|
||||
if key_node is None:
|
||||
return None
|
||||
if key_node.type == "string":
|
||||
content = key_node.child_by_field_name("string_content")
|
||||
if content:
|
||||
return _read_text(content, source)
|
||||
# fallback: strip surrounding quotes
|
||||
raw = _read_text(key_node, source)
|
||||
return raw.strip('"\'')
|
||||
return _read_text(key_node, source)
|
||||
|
||||
def _val_node(pair_node):
|
||||
return pair_node.child_by_field_name("value")
|
||||
|
||||
def walk_object(obj_node, parent_nid: str, parent_key: str | None,
|
||||
depth: int, pair_count: list) -> None:
|
||||
if depth > 6:
|
||||
return
|
||||
for child in obj_node.children:
|
||||
if child.type != "pair":
|
||||
continue
|
||||
if pair_count[0] >= 500: # check per-pair so the cap is honoured exactly (J-3)
|
||||
return
|
||||
pair_count[0] += 1
|
||||
key = _key_text(child)
|
||||
if not key:
|
||||
continue
|
||||
key_nid = _make_id(stem, *(([parent_key] if parent_key else []) + [key]))
|
||||
if not key_nid:
|
||||
continue
|
||||
line = child.start_point[0] + 1
|
||||
add_node(key_nid, key, line)
|
||||
add_edge(parent_nid, key_nid, "contains", line)
|
||||
|
||||
val = _val_node(child)
|
||||
if val is None:
|
||||
continue
|
||||
|
||||
if val.type == "object":
|
||||
walk_object(val, key_nid, key, depth + 1, pair_count)
|
||||
|
||||
elif val.type == "array":
|
||||
# For "extends" arrays (tsconfig, eslint): each string element.
|
||||
# Prefix with "ref_" so external refs don't collide with real
|
||||
# code/file node IDs that share the same collapsed _make_id (J-4).
|
||||
for item in val.children:
|
||||
if item.type == "string":
|
||||
content = item.child_by_field_name("string_content")
|
||||
ref = _read_text(content, source) if content else _read_text(item, source).strip('"\'')
|
||||
if ref:
|
||||
ref_nid = _make_id("ref", ref)
|
||||
if ref_nid:
|
||||
add_edge(key_nid, ref_nid, "extends", line, context="import")
|
||||
|
||||
elif val.type == "string":
|
||||
content = val.child_by_field_name("string_content")
|
||||
val_text = _read_text(content, source) if content else _read_text(val, source).strip('"\'')
|
||||
|
||||
if key == "extends" and val_text:
|
||||
# Namespace external refs to avoid ID collision with file nodes (J-4)
|
||||
ref_nid = _make_id("ref", val_text)
|
||||
if ref_nid:
|
||||
add_edge(file_nid, ref_nid, "extends", line, context="import")
|
||||
|
||||
elif key == "$ref" and val_text:
|
||||
# Namespace $ref values to prevent edge hijacking into code nodes (J-4)
|
||||
ref_nid = _make_id("ref", val_text)
|
||||
if ref_nid:
|
||||
add_edge(parent_nid, ref_nid, "references", line)
|
||||
|
||||
elif parent_key in _DEP_KEYS and val_text:
|
||||
dep_nid = _make_id(key)
|
||||
if dep_nid:
|
||||
add_edge(key_nid, dep_nid, "imports", line, context="import")
|
||||
|
||||
# Entry: find root document → object
|
||||
doc = root
|
||||
if doc.type == "document" and doc.child_count > 0:
|
||||
doc = doc.children[0]
|
||||
if doc.type == "object":
|
||||
# Only AST-extract recognized config/manifest JSON. Data JSON (fixtures,
|
||||
# datasets, GeoJSON, API dumps) is skipped so it doesn't explode into
|
||||
# orphan key-nodes (#1224); it's left to the LLM semantic pass.
|
||||
if not _is_config_json(path, doc, source):
|
||||
return {"nodes": [], "edges": [], "skipped": "data json (not a config/manifest)"}
|
||||
walk_object(doc, file_nid, None, 0, [0])
|
||||
else:
|
||||
# Top-level array or scalar => data JSON, never a config/manifest.
|
||||
return {"nodes": [], "edges": [], "skipped": "data json (non-object root)"}
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
@@ -0,0 +1,196 @@
|
||||
"""Pascal_forms extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from graphify.extractors.base import _file_stem, _make_id
|
||||
|
||||
|
||||
def extract_lazarus_form(path: Path) -> dict:
|
||||
"""Extract component hierarchy from Lazarus .lfm form files.
|
||||
|
||||
.lfm is a text-based declarative format for UI component trees, structured as:
|
||||
object ComponentName: TClassName
|
||||
PropertyName = Value
|
||||
OnEvent = HandlerName
|
||||
object ChildName: TChildClass
|
||||
...
|
||||
end
|
||||
end
|
||||
|
||||
Produces nodes for:
|
||||
- The form file itself
|
||||
- Each component class encountered (TForm1, TButton, TPanel, ...)
|
||||
- Event handler names referenced by OnXxx properties
|
||||
|
||||
Produces edges for:
|
||||
- file --contains--> root form class
|
||||
- parent component --contains--> child component class
|
||||
- component --references--> event handler (context: "event")
|
||||
"""
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
import re
|
||||
str_path = str(path)
|
||||
stem = _file_stem(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
seen_edge_pairs: set[tuple[str, str, str]] = set()
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid, "label": label, "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line}",
|
||||
})
|
||||
|
||||
def add_edge(
|
||||
src: str, tgt: str, relation: str, line: int,
|
||||
context: str | None = None,
|
||||
) -> None:
|
||||
key = (src, tgt, relation)
|
||||
if key in seen_edge_pairs:
|
||||
return
|
||||
seen_edge_pairs.add(key)
|
||||
edge: dict[str, Any] = {
|
||||
"source": src, "target": tgt, "relation": relation,
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": 1.0,
|
||||
}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
obj_re = re.compile(r"^\s*object\s+\w+\s*:\s*(\w+)", re.IGNORECASE)
|
||||
event_re = re.compile(r"^\s*On\w+\s*=\s*(\w+)", re.IGNORECASE)
|
||||
end_re = re.compile(r"^\s*end\s*$", re.IGNORECASE)
|
||||
|
||||
# Stack of node IDs representing the nesting of object...end blocks
|
||||
stack: list[str] = [file_nid]
|
||||
|
||||
for lineno, line in enumerate(text.splitlines(), 1):
|
||||
m = obj_re.match(line)
|
||||
if m:
|
||||
class_name = m.group(1)
|
||||
nid = _make_id(stem, class_name)
|
||||
add_node(nid, class_name, lineno)
|
||||
add_edge(stack[-1], nid, "contains", lineno)
|
||||
stack.append(nid)
|
||||
continue
|
||||
|
||||
m = event_re.match(line)
|
||||
if m and len(stack) > 1:
|
||||
handler = m.group(1)
|
||||
handler_nid = _make_id(stem, handler)
|
||||
add_node(handler_nid, f"{handler}()", lineno)
|
||||
add_edge(stack[-1], handler_nid, "references", lineno, context="event")
|
||||
continue
|
||||
|
||||
if end_re.match(line) and len(stack) > 1:
|
||||
stack.pop()
|
||||
|
||||
return {"nodes": nodes, "edges": edges, "input_tokens": 0, "output_tokens": 0}
|
||||
|
||||
def extract_delphi_form(path: Path) -> dict:
|
||||
"""Extract component hierarchy from Delphi .dfm form files.
|
||||
|
||||
.dfm files come in two formats:
|
||||
- Text (same `object Name: TClassName ... end` syntax as .lfm)
|
||||
- Binary (starts with a TPF0/FF0A magic header — unreadable as text)
|
||||
|
||||
Binary .dfm files are skipped gracefully: an empty result is returned
|
||||
so the rest of the pipeline is unaffected. Convert binary forms to
|
||||
text in the Delphi IDE via File → Save As (Text DFM) if you want them
|
||||
indexed.
|
||||
|
||||
Text .dfm files are parsed identically to .lfm: component containment
|
||||
(`contains`) and event handler references (`references`, context "event").
|
||||
"""
|
||||
try:
|
||||
raw = path.read_bytes()
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
# Detect binary DFM: Delphi binary resource streams start with FF 0A
|
||||
if raw[:2] == b"\xff\x0a":
|
||||
return {
|
||||
"nodes": [], "edges": [],
|
||||
"error": f"binary DFM (convert to text in Delphi IDE to index): {path.name}",
|
||||
}
|
||||
|
||||
# Text DFM — delegate to the shared form parser (same syntax as .lfm)
|
||||
try:
|
||||
text = raw.decode("utf-8", errors="replace")
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
import re
|
||||
str_path = str(path)
|
||||
stem = _file_stem(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
seen_edge_pairs: set[tuple[str, str, str]] = set()
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid, "label": label, "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line}",
|
||||
})
|
||||
|
||||
def add_edge(
|
||||
src: str, tgt: str, relation: str, line: int,
|
||||
context: str | None = None,
|
||||
) -> None:
|
||||
key = (src, tgt, relation)
|
||||
if key in seen_edge_pairs:
|
||||
return
|
||||
seen_edge_pairs.add(key)
|
||||
edge: dict[str, Any] = {
|
||||
"source": src, "target": tgt, "relation": relation,
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": 1.0,
|
||||
}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
obj_re = re.compile(r"^\s*object\s+\w+\s*:\s*(\w+)", re.IGNORECASE)
|
||||
event_re = re.compile(r"^\s*On\w+\s*=\s*(\w+)", re.IGNORECASE)
|
||||
end_re = re.compile(r"^\s*end\s*$", re.IGNORECASE)
|
||||
stack: list[str] = [file_nid]
|
||||
|
||||
for lineno, line in enumerate(text.splitlines(), 1):
|
||||
m = obj_re.match(line)
|
||||
if m:
|
||||
class_name = m.group(1)
|
||||
nid = _make_id(stem, class_name)
|
||||
add_node(nid, class_name, lineno)
|
||||
add_edge(stack[-1], nid, "contains", lineno)
|
||||
stack.append(nid)
|
||||
continue
|
||||
m = event_re.match(line)
|
||||
if m and len(stack) > 1:
|
||||
handler = m.group(1)
|
||||
handler_nid = _make_id(stem, handler)
|
||||
add_node(handler_nid, f"{handler}()", lineno)
|
||||
add_edge(stack[-1], handler_nid, "references", lineno, context="event")
|
||||
continue
|
||||
if end_re.match(line) and len(stack) > 1:
|
||||
stack.pop()
|
||||
|
||||
return {"nodes": nodes, "edges": edges, "input_tokens": 0, "output_tokens": 0}
|
||||
@@ -0,0 +1,496 @@
|
||||
"""Powershell extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from graphify.extractors.base import _file_stem, _make_id, _read_text
|
||||
|
||||
|
||||
def extract_powershell(path: Path) -> dict:
|
||||
"""Extract functions, classes, methods, and using statements from a .ps1 file."""
|
||||
try:
|
||||
import tree_sitter_powershell as tsps
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree_sitter_powershell not installed"}
|
||||
|
||||
try:
|
||||
language = Language(tsps.language())
|
||||
parser = Parser(language)
|
||||
source = path.read_bytes()
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
stem = _file_stem(path)
|
||||
str_path = str(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
function_bodies: list[tuple[str, Any]] = []
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({"id": nid, "label": label, "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line}"})
|
||||
|
||||
def add_edge(src: str, tgt: str, relation: str, line: int,
|
||||
confidence: str = "EXTRACTED", weight: float = 1.0,
|
||||
context: str | None = None) -> None:
|
||||
edge = {"source": src, "target": tgt, "relation": relation,
|
||||
"confidence": confidence, "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": weight}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
_PS_SKIP = frozenset({
|
||||
"using", "return", "if", "else", "elseif", "foreach", "for",
|
||||
"while", "do", "switch", "try", "catch", "finally", "throw",
|
||||
"break", "continue", "exit", "param", "begin", "process", "end",
|
||||
# Import commands — handled as import edges, not function calls
|
||||
"import-module",
|
||||
})
|
||||
|
||||
def _find_script_block_body(node):
|
||||
for child in node.children:
|
||||
if child.type == "script_block":
|
||||
for sc in child.children:
|
||||
if sc.type == "script_block_body":
|
||||
return sc
|
||||
return child
|
||||
return None
|
||||
|
||||
def ensure_named_node(name: str, line: int) -> str:
|
||||
nid = _make_id(stem, name)
|
||||
if nid in seen_ids:
|
||||
return nid
|
||||
nid = _make_id(name)
|
||||
if nid not in seen_ids:
|
||||
# The name isn't defined in this file, so this is a cross-file reference
|
||||
# (e.g. a `Thing` type annotation imported from another module). Emit a
|
||||
# SOURCELESS stub — like the inheritance-base path below — so the
|
||||
# corpus-level rewire can collapse it onto the real definition. A sourced
|
||||
# stub here makes _disambiguate_colliding_node_ids bake the referencing
|
||||
# file's path (with extension) into the id and blocks the rewire, which is
|
||||
# the phantom-duplicate-node bug (#1402).
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid,
|
||||
"label": name,
|
||||
"file_type": "code",
|
||||
"source_file": "",
|
||||
"source_location": "",
|
||||
"origin_file": str_path,
|
||||
})
|
||||
return nid
|
||||
|
||||
def _ps_type_name(type_literal_node) -> str | None:
|
||||
"""Drill into a type_literal node and return the inner type_identifier text."""
|
||||
if type_literal_node is None:
|
||||
return None
|
||||
for spec in type_literal_node.children:
|
||||
if spec.type != "type_spec":
|
||||
continue
|
||||
for tname in spec.children:
|
||||
if tname.type != "type_name":
|
||||
continue
|
||||
for tid in tname.children:
|
||||
if tid.type == "type_identifier":
|
||||
return _read_text(tid, source)
|
||||
return None
|
||||
|
||||
def walk(node, parent_class_nid: str | None = None) -> None:
|
||||
t = node.type
|
||||
|
||||
if t == "function_statement":
|
||||
name_node = next((c for c in node.children if c.type == "function_name"), None)
|
||||
if name_node:
|
||||
func_name = _read_text(name_node, source)
|
||||
line = node.start_point[0] + 1
|
||||
func_nid = _make_id(stem, func_name)
|
||||
add_node(func_nid, f"{func_name}()", line)
|
||||
add_edge(file_nid, func_nid, "contains", line)
|
||||
body = _find_script_block_body(node)
|
||||
if body:
|
||||
function_bodies.append((func_nid, body))
|
||||
# Also walk the body during the main pass so that
|
||||
# Import-Module / dot-source inside functions emit
|
||||
# file-level imports_from edges (#1331).
|
||||
walk(body, parent_class_nid)
|
||||
return
|
||||
|
||||
if t == "class_statement":
|
||||
name_node = next((c for c in node.children if c.type == "simple_name"), None)
|
||||
if name_node:
|
||||
class_name = _read_text(name_node, source)
|
||||
line = node.start_point[0] + 1
|
||||
class_nid = _make_id(stem, class_name)
|
||||
add_node(class_nid, class_name, line)
|
||||
add_edge(file_nid, class_nid, "contains", line)
|
||||
# Base type(s) after ':'. PowerShell has no syntactic base vs
|
||||
# interface split, so (matching the C# convention) treat the
|
||||
# first base as the superclass (inherits) and the rest as
|
||||
# interfaces (implements). Bases are the simple_name children
|
||||
# after the ':' token.
|
||||
colon_seen = False
|
||||
base_index = 0
|
||||
for child in node.children:
|
||||
if child.type == ":":
|
||||
colon_seen = True
|
||||
elif colon_seen and child.type == "simple_name":
|
||||
base_nid = ensure_named_node(_read_text(child, source), line)
|
||||
if base_nid != class_nid:
|
||||
rel = "inherits" if base_index == 0 else "implements"
|
||||
add_edge(class_nid, base_nid, rel, line)
|
||||
base_index += 1
|
||||
for child in node.children:
|
||||
walk(child, parent_class_nid=class_nid)
|
||||
return
|
||||
|
||||
if t == "class_property_definition" and parent_class_nid:
|
||||
type_literal = next((c for c in node.children if c.type == "type_literal"), None)
|
||||
type_name = _ps_type_name(type_literal)
|
||||
if type_name:
|
||||
line = node.start_point[0] + 1
|
||||
target_nid = ensure_named_node(type_name, line)
|
||||
if target_nid != parent_class_nid:
|
||||
add_edge(parent_class_nid, target_nid, "references",
|
||||
line, context="field")
|
||||
return
|
||||
|
||||
if t == "class_method_definition":
|
||||
name_node = next((c for c in node.children if c.type == "simple_name"), None)
|
||||
if name_node:
|
||||
method_name = _read_text(name_node, source)
|
||||
line = node.start_point[0] + 1
|
||||
if parent_class_nid:
|
||||
method_nid = _make_id(parent_class_nid, method_name)
|
||||
add_node(method_nid, f".{method_name}()", line)
|
||||
add_edge(parent_class_nid, method_nid, "method", line)
|
||||
else:
|
||||
method_nid = _make_id(stem, method_name)
|
||||
add_node(method_nid, f"{method_name}()", line)
|
||||
add_edge(file_nid, method_nid, "contains", line)
|
||||
# Return type: type_literal sibling of simple_name
|
||||
return_type_literal = next(
|
||||
(c for c in node.children if c.type == "type_literal"), None)
|
||||
return_type_name = _ps_type_name(return_type_literal)
|
||||
if return_type_name:
|
||||
target_nid = ensure_named_node(return_type_name, line)
|
||||
if target_nid != method_nid:
|
||||
add_edge(method_nid, target_nid, "references",
|
||||
line, context="return_type")
|
||||
# Parameter types: class_method_parameter_list
|
||||
param_list = next(
|
||||
(c for c in node.children if c.type == "class_method_parameter_list"), None)
|
||||
if param_list is not None:
|
||||
for p in param_list.children:
|
||||
if p.type != "class_method_parameter":
|
||||
continue
|
||||
ptype_literal = next(
|
||||
(c for c in p.children if c.type == "type_literal"), None)
|
||||
ptype_name = _ps_type_name(ptype_literal)
|
||||
if not ptype_name:
|
||||
continue
|
||||
p_line = p.start_point[0] + 1
|
||||
target_nid = ensure_named_node(ptype_name, p_line)
|
||||
if target_nid != method_nid:
|
||||
add_edge(method_nid, target_nid, "references",
|
||||
p_line, context="parameter_type")
|
||||
body = _find_script_block_body(node)
|
||||
if body:
|
||||
function_bodies.append((method_nid, body))
|
||||
return
|
||||
|
||||
if t == "command":
|
||||
# Dot-sourcing: `. ./Shared.psm1`
|
||||
# Uses command_invokation_operator '.' + command_name_expr (not command_name)
|
||||
invoke_op = next(
|
||||
(c for c in node.children if c.type == "command_invokation_operator"), None
|
||||
)
|
||||
if invoke_op is not None and _read_text(invoke_op, source).strip() == ".":
|
||||
name_expr = next(
|
||||
(c for c in node.children if c.type == "command_name_expr"), None
|
||||
)
|
||||
if name_expr is not None:
|
||||
name_node = next(
|
||||
(c for c in name_expr.children if c.type == "command_name"), None
|
||||
)
|
||||
if name_node:
|
||||
raw_path = _read_text(name_node, source)
|
||||
# Strip relative path prefix (./ or .\ or just the dot)
|
||||
module_stem = re.sub(r'^[./\\]+', '', raw_path)
|
||||
# Drop extension to get bare module name
|
||||
module_stem = re.sub(r'\.[^.]+$', '', module_stem).replace('\\', '/')
|
||||
module_name = module_stem.split('/')[-1]
|
||||
if module_name:
|
||||
add_edge(file_nid, _make_id(module_name), "imports_from",
|
||||
node.start_point[0] + 1)
|
||||
return
|
||||
|
||||
cmd_name_node = next((c for c in node.children if c.type == "command_name"), None)
|
||||
if cmd_name_node:
|
||||
cmd_text = _read_text(cmd_name_node, source).lower()
|
||||
if cmd_text == "using":
|
||||
tokens = []
|
||||
for child in node.children:
|
||||
if child.type == "command_elements":
|
||||
for el in child.children:
|
||||
if el.type == "generic_token":
|
||||
tokens.append(_read_text(el, source))
|
||||
module_tokens = [t for t in tokens
|
||||
if t.lower() not in ("namespace", "module", "assembly")]
|
||||
if module_tokens:
|
||||
module_name = module_tokens[-1].split(".")[-1]
|
||||
add_edge(file_nid, _make_id(module_name), "imports_from",
|
||||
node.start_point[0] + 1)
|
||||
elif cmd_text == "import-module":
|
||||
# Collect generic_token args; skip command_parameter flags like -Name
|
||||
# The module name is the first generic_token (or the one after -Name)
|
||||
module_name: str | None = None
|
||||
expect_name = False
|
||||
for child in node.children:
|
||||
if child.type != "command_elements":
|
||||
continue
|
||||
for el in child.children:
|
||||
if el.type == "command_parameter":
|
||||
param_text = _read_text(el, source).lstrip("-").lower()
|
||||
expect_name = param_text in ("name", "n")
|
||||
elif el.type == "generic_token":
|
||||
token = _read_text(el, source)
|
||||
if module_name is None or expect_name:
|
||||
module_name = token
|
||||
expect_name = False
|
||||
if module_name:
|
||||
# Strip extension; keep only the stem for the node ID
|
||||
bare = re.sub(r'\.[^.]+$', '', module_name).split('/')[-1].split('\\')[-1]
|
||||
if bare:
|
||||
add_edge(file_nid, _make_id(bare), "imports_from",
|
||||
node.start_point[0] + 1)
|
||||
return
|
||||
|
||||
for child in node.children:
|
||||
walk(child, parent_class_nid)
|
||||
|
||||
walk(root)
|
||||
|
||||
label_to_nid = {n["label"].strip("()").lstrip(".").lower(): n["id"] for n in nodes}
|
||||
seen_call_pairs: set[tuple[str, str]] = set()
|
||||
raw_calls: list[dict] = []
|
||||
|
||||
def walk_calls(node, caller_nid: str) -> None:
|
||||
if node.type in ("function_statement", "class_statement"):
|
||||
return
|
||||
if node.type == "command":
|
||||
cmd_name_node = next((c for c in node.children if c.type == "command_name"), None)
|
||||
if cmd_name_node:
|
||||
cmd_text = _read_text(cmd_name_node, source)
|
||||
if cmd_text.lower() not in _PS_SKIP:
|
||||
tgt_nid = label_to_nid.get(cmd_text.lower())
|
||||
if tgt_nid and tgt_nid != caller_nid:
|
||||
pair = (caller_nid, tgt_nid)
|
||||
if pair not in seen_call_pairs:
|
||||
seen_call_pairs.add(pair)
|
||||
add_edge(caller_nid, tgt_nid, "calls",
|
||||
node.start_point[0] + 1,
|
||||
confidence="EXTRACTED", weight=1.0)
|
||||
elif cmd_text:
|
||||
raw_calls.append({
|
||||
"caller_nid": caller_nid,
|
||||
"callee": cmd_text,
|
||||
"is_member_call": False,
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{node.start_point[0] + 1}",
|
||||
})
|
||||
for child in node.children:
|
||||
walk_calls(child, caller_nid)
|
||||
|
||||
for caller_nid, body_node in function_bodies:
|
||||
walk_calls(body_node, caller_nid)
|
||||
|
||||
clean_edges = [e for e in edges if e["source"] in seen_ids and
|
||||
(e["target"] in seen_ids or e["relation"] in ("imports_from", "imports"))]
|
||||
return {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls}
|
||||
|
||||
_PSD1_IMPORT_KEYS = frozenset({"RootModule", "NestedModules", "RequiredModules"})
|
||||
|
||||
def _psd1_collect_string_literals(node, source: bytes) -> list[str]:
|
||||
"""Recursively collect all string_literal text values under *node*."""
|
||||
results: list[str] = []
|
||||
|
||||
def _walk(n) -> None:
|
||||
if n.type == "string_literal":
|
||||
raw = source[n.start_byte:n.end_byte].decode(errors="replace")
|
||||
# Strip surrounding quote chars (' or ")
|
||||
results.append(raw.strip("'\""))
|
||||
return
|
||||
for child in n.children:
|
||||
_walk(child)
|
||||
|
||||
_walk(node)
|
||||
return results
|
||||
|
||||
def _psd1_module_name(raw: str) -> str:
|
||||
"""Derive a bare module name from a raw string value.
|
||||
|
||||
e.g. 'MyModule.psm1' → 'MyModule', './sub/Util.psm1' → 'Util', 'PSReadLine' → 'PSReadLine'
|
||||
"""
|
||||
# Strip path prefix and extension
|
||||
name = raw.replace("\\", "/").split("/")[-1]
|
||||
name = re.sub(r"\.[^.]+$", "", name) # remove last extension
|
||||
return name.strip()
|
||||
|
||||
def extract_powershell_manifest(path: Path) -> dict:
|
||||
"""Extract module dependency edges from a PowerShell .psd1 manifest file.
|
||||
|
||||
.psd1 files are PowerShell data hashtables, not scripts. tree-sitter-powershell
|
||||
parses them correctly (they are syntactically valid PS). We walk the AST looking
|
||||
for RootModule, NestedModules, and RequiredModules keys and emit imports_from
|
||||
edges for every referenced module.
|
||||
|
||||
RequiredModules supports two forms:
|
||||
- Simple string: 'PSReadLine'
|
||||
- Module specification: @{ ModuleName = 'Pester'; ModuleVersion = '5.0' }
|
||||
For the hashtable form we only follow the ModuleName key.
|
||||
"""
|
||||
try:
|
||||
import tree_sitter_powershell as tsps
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree_sitter_powershell not installed"}
|
||||
|
||||
try:
|
||||
language = Language(tsps.language())
|
||||
parser = Parser(language)
|
||||
source = path.read_bytes()
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
str_path = str(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({"id": nid, "label": label, "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line}"})
|
||||
|
||||
def add_import_edge(src: str, module_raw: str, line: int) -> None:
|
||||
name = _psd1_module_name(module_raw)
|
||||
if not name:
|
||||
return
|
||||
tgt_nid = _make_id(name)
|
||||
edges.append({
|
||||
"source": src,
|
||||
"target": tgt_nid,
|
||||
"relation": "imports_from",
|
||||
"confidence": "EXTRACTED",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
"weight": 1.0,
|
||||
"context": "import",
|
||||
})
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
def walk_manifest(node) -> None:
|
||||
"""Walk the AST and emit edges for import-relevant hash_entry nodes."""
|
||||
if node.type != "hash_entry":
|
||||
for child in node.children:
|
||||
walk_manifest(child)
|
||||
return
|
||||
|
||||
# Identify the key
|
||||
key_node = next((c for c in node.children if c.type == "key_expression"), None)
|
||||
if key_node is None:
|
||||
return
|
||||
key_text = source[key_node.start_byte:key_node.end_byte].decode(errors="replace").strip()
|
||||
|
||||
if key_text not in _PSD1_IMPORT_KEYS:
|
||||
# Still recurse in case there are nested hashes (e.g. ModuleVersion entries
|
||||
# contain sub-hashes, but we only care about top-level keys for imports)
|
||||
return
|
||||
|
||||
line = node.start_point[0] + 1
|
||||
value_node = next((c for c in node.children if c.type == "pipeline"), None)
|
||||
if value_node is None:
|
||||
return
|
||||
|
||||
if key_text == "RootModule":
|
||||
# Value is a single string
|
||||
strings = _psd1_collect_string_literals(value_node, source)
|
||||
for s in strings:
|
||||
add_import_edge(file_nid, s, line)
|
||||
|
||||
elif key_text == "NestedModules":
|
||||
# Value is a string or @('a', 'b', ...) array — collect all string literals
|
||||
strings = _psd1_collect_string_literals(value_node, source)
|
||||
for s in strings:
|
||||
add_import_edge(file_nid, s, line)
|
||||
|
||||
elif key_text == "RequiredModules":
|
||||
# Two forms:
|
||||
# 1) 'SimpleModule' — direct string literals in the array
|
||||
# 2) @{ ModuleName = 'Foo'; ModuleVersion = '2.0' } — use ModuleName only
|
||||
#
|
||||
# Strategy: walk the value for hash_entry nodes whose key is 'ModuleName';
|
||||
# collect their string values. For the remaining string_literal nodes that
|
||||
# are NOT inside a hash_entry subtree, treat them as simple module names.
|
||||
module_name_strings: list[str] = []
|
||||
inside_hash_entries: set[int] = set() # byte offsets of handled strings
|
||||
|
||||
def find_modulename_entries(n) -> None:
|
||||
if n.type == "hash_entry":
|
||||
sub_key = next((c for c in n.children if c.type == "key_expression"), None)
|
||||
if sub_key is not None:
|
||||
sk_text = source[sub_key.start_byte:sub_key.end_byte].decode(errors="replace").strip()
|
||||
# Collect strings inside *all* sub-keys so we can exclude them
|
||||
for c in n.children:
|
||||
if c.type == "pipeline":
|
||||
for s_node in _collect_string_nodes(c):
|
||||
inside_hash_entries.add(s_node.start_byte)
|
||||
if sk_text == "ModuleName":
|
||||
for c in n.children:
|
||||
if c.type == "pipeline":
|
||||
for s in _psd1_collect_string_literals(c, source):
|
||||
module_name_strings.append(s)
|
||||
return # don't recurse further into this hash_entry
|
||||
for child in n.children:
|
||||
find_modulename_entries(child)
|
||||
|
||||
def _collect_string_nodes(n):
|
||||
"""Return all string_literal nodes in subtree."""
|
||||
if n.type == "string_literal":
|
||||
yield n
|
||||
return
|
||||
for child in n.children:
|
||||
yield from _collect_string_nodes(child)
|
||||
|
||||
find_modulename_entries(value_node)
|
||||
|
||||
# Now gather direct string literals not inside hash entries
|
||||
direct_strings: list[str] = []
|
||||
for s_node in _collect_string_nodes(value_node):
|
||||
if s_node.start_byte not in inside_hash_entries:
|
||||
raw = source[s_node.start_byte:s_node.end_byte].decode(errors="replace")
|
||||
direct_strings.append(raw.strip("'\""))
|
||||
|
||||
for s in direct_strings + module_name_strings:
|
||||
add_import_edge(file_nid, s, line)
|
||||
|
||||
walk_manifest(root)
|
||||
|
||||
return {"nodes": nodes, "edges": edges, "raw_calls": []}
|
||||
@@ -0,0 +1,410 @@
|
||||
"""Rust extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
from pathlib import Path
|
||||
from graphify.extractors.base import _LANGUAGE_BUILTIN_GLOBALS, _file_stem, _make_id, _read_text
|
||||
|
||||
|
||||
def _rust_collect_type_refs(node, source: bytes, generic: bool, out: list[tuple[str, str]]) -> None:
|
||||
"""Walk a Rust type expression; append (name, role) tuples."""
|
||||
if node is None:
|
||||
return
|
||||
t = node.type
|
||||
if t == "primitive_type":
|
||||
return
|
||||
if t == "type_identifier":
|
||||
text = _read_text(node, source)
|
||||
if text:
|
||||
out.append((text, "generic_arg" if generic else "type"))
|
||||
return
|
||||
if t == "scoped_type_identifier":
|
||||
text = _read_text(node, source).rsplit("::", 1)[-1]
|
||||
if text:
|
||||
out.append((text, "generic_arg" if generic else "type"))
|
||||
return
|
||||
if t == "generic_type":
|
||||
name_node = node.child_by_field_name("type")
|
||||
if name_node is None:
|
||||
for c in node.children:
|
||||
if c.type in ("type_identifier", "scoped_type_identifier"):
|
||||
name_node = c
|
||||
break
|
||||
if name_node is not None:
|
||||
text = _read_text(name_node, source).rsplit("::", 1)[-1]
|
||||
if text:
|
||||
out.append((text, "generic_arg" if generic else "type"))
|
||||
for c in node.children:
|
||||
if c.type == "type_arguments":
|
||||
for arg in c.children:
|
||||
if arg.is_named:
|
||||
_rust_collect_type_refs(arg, source, True, out)
|
||||
return
|
||||
if t in ("reference_type", "pointer_type", "array_type", "tuple_type", "slice_type"):
|
||||
for c in node.children:
|
||||
if c.is_named:
|
||||
_rust_collect_type_refs(c, source, generic, out)
|
||||
return
|
||||
if node.is_named:
|
||||
for c in node.children:
|
||||
if c.is_named:
|
||||
_rust_collect_type_refs(c, source, generic, out)
|
||||
|
||||
_RUST_TRAIT_METHOD_BLOCKLIST: frozenset[str] = frozenset({
|
||||
"new", "default", "parse", "from_str", "now", "clone", "into", "from",
|
||||
"to_string", "to_owned", "len", "is_empty", "iter", "next", "build",
|
||||
"start", "run", "init", "app", "get", "set", "push", "pop", "insert",
|
||||
"remove", "contains", "collect", "map", "filter", "unwrap", "expect",
|
||||
"ok", "err", "some", "none", "send", "recv", "lock", "read", "write",
|
||||
})
|
||||
|
||||
def extract_rust(path: Path) -> dict:
|
||||
"""Extract functions, structs, enums, traits, impl methods, and use declarations from a .rs file."""
|
||||
try:
|
||||
import tree_sitter_rust as tsrust
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree-sitter-rust not installed"}
|
||||
|
||||
try:
|
||||
language = Language(tsrust.language())
|
||||
parser = Parser(language)
|
||||
source = path.read_bytes()
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
stem = _file_stem(path)
|
||||
str_path = str(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
function_bodies: list[tuple[str, object]] = []
|
||||
|
||||
def add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid,
|
||||
"label": label,
|
||||
"file_type": "code",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
})
|
||||
|
||||
def add_edge(src: str, tgt: str, relation: str, line: int,
|
||||
confidence: str = "EXTRACTED", weight: float = 1.0,
|
||||
context: str | None = None) -> None:
|
||||
edge = {
|
||||
"source": src,
|
||||
"target": tgt,
|
||||
"relation": relation,
|
||||
"confidence": confidence,
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
"weight": weight,
|
||||
}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
add_node(file_nid, path.name, 1)
|
||||
|
||||
def ensure_named_node(name: str, line: int) -> str:
|
||||
nid = _make_id(stem, name)
|
||||
if nid in seen_ids:
|
||||
return nid
|
||||
nid = _make_id(name)
|
||||
if nid not in seen_ids:
|
||||
# The name isn't defined in this file, so this is a cross-file reference
|
||||
# (e.g. a `Thing` type annotation imported from another module). Emit a
|
||||
# SOURCELESS stub — like the inheritance-base path below — so the
|
||||
# corpus-level rewire can collapse it onto the real definition. A sourced
|
||||
# stub here makes _disambiguate_colliding_node_ids bake the referencing
|
||||
# file's path (with extension) into the id and blocks the rewire, which is
|
||||
# the phantom-duplicate-node bug (#1402).
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid,
|
||||
"label": name,
|
||||
"file_type": "code",
|
||||
"source_file": "",
|
||||
"source_location": "",
|
||||
"origin_file": str_path,
|
||||
})
|
||||
return nid
|
||||
|
||||
def emit_param_return_refs(func_node, func_nid: str, line: int) -> None:
|
||||
params = func_node.child_by_field_name("parameters")
|
||||
if params is not None:
|
||||
for p in params.children:
|
||||
if p.type != "parameter":
|
||||
continue
|
||||
type_node = p.child_by_field_name("type")
|
||||
refs: list[tuple[str, str]] = []
|
||||
_rust_collect_type_refs(type_node, source, False, refs)
|
||||
for ref_name, role in refs:
|
||||
ctx = "generic_arg" if role == "generic_arg" else "parameter_type"
|
||||
tgt = ensure_named_node(ref_name, line)
|
||||
if tgt != func_nid:
|
||||
add_edge(func_nid, tgt, "references", line, context=ctx)
|
||||
return_type = func_node.child_by_field_name("return_type")
|
||||
if return_type is not None:
|
||||
refs = []
|
||||
_rust_collect_type_refs(return_type, source, False, refs)
|
||||
for ref_name, role in refs:
|
||||
ctx = "generic_arg" if role == "generic_arg" else "return_type"
|
||||
tgt = ensure_named_node(ref_name, line)
|
||||
if tgt != func_nid:
|
||||
add_edge(func_nid, tgt, "references", line, context=ctx)
|
||||
|
||||
def walk(node, parent_impl_nid: str | None = None) -> None:
|
||||
t = node.type
|
||||
|
||||
if t == "function_item":
|
||||
name_node = node.child_by_field_name("name")
|
||||
if name_node:
|
||||
func_name = _read_text(name_node, source)
|
||||
line = node.start_point[0] + 1
|
||||
if parent_impl_nid:
|
||||
func_nid = _make_id(parent_impl_nid, func_name)
|
||||
add_node(func_nid, f".{func_name}()", line)
|
||||
add_edge(parent_impl_nid, func_nid, "method", line)
|
||||
else:
|
||||
func_nid = _make_id(stem, func_name)
|
||||
add_node(func_nid, f"{func_name}()", line)
|
||||
add_edge(file_nid, func_nid, "contains", line)
|
||||
emit_param_return_refs(node, func_nid, line)
|
||||
body = node.child_by_field_name("body")
|
||||
if body:
|
||||
function_bodies.append((func_nid, body))
|
||||
return
|
||||
|
||||
if t in ("struct_item", "enum_item", "trait_item"):
|
||||
name_node = node.child_by_field_name("name")
|
||||
if name_node:
|
||||
item_name = _read_text(name_node, source)
|
||||
line = node.start_point[0] + 1
|
||||
item_nid = _make_id(stem, item_name)
|
||||
add_node(item_nid, item_name, line)
|
||||
add_edge(file_nid, item_nid, "contains", line)
|
||||
if t == "trait_item":
|
||||
for c in node.children:
|
||||
if c.type != "trait_bounds":
|
||||
continue
|
||||
for sub in c.children:
|
||||
if not sub.is_named:
|
||||
continue
|
||||
refs: list[tuple[str, str]] = []
|
||||
_rust_collect_type_refs(sub, source, False, refs)
|
||||
for idx, (ref_name, _role) in enumerate(refs):
|
||||
tgt = ensure_named_node(ref_name, line)
|
||||
if tgt == item_nid:
|
||||
continue
|
||||
rel = "inherits" if idx == 0 else "references"
|
||||
if rel == "inherits":
|
||||
add_edge(item_nid, tgt, "inherits", line)
|
||||
else:
|
||||
add_edge(item_nid, tgt, "references", line,
|
||||
context="generic_arg")
|
||||
if t == "struct_item":
|
||||
for c in node.children:
|
||||
if c.type != "field_declaration_list":
|
||||
continue
|
||||
for field in c.children:
|
||||
if field.type != "field_declaration":
|
||||
continue
|
||||
type_node = field.child_by_field_name("type")
|
||||
if type_node is None:
|
||||
for fc in field.children:
|
||||
if fc.type in ("type_identifier", "generic_type",
|
||||
"scoped_type_identifier",
|
||||
"reference_type", "primitive_type"):
|
||||
type_node = fc
|
||||
break
|
||||
refs = []
|
||||
_rust_collect_type_refs(type_node, source, False, refs)
|
||||
for ref_name, role in refs:
|
||||
ctx = "generic_arg" if role == "generic_arg" else "field"
|
||||
tgt = ensure_named_node(ref_name, field.start_point[0] + 1)
|
||||
if tgt != item_nid:
|
||||
add_edge(item_nid, tgt, "references",
|
||||
field.start_point[0] + 1, context=ctx)
|
||||
# Tuple structs (`struct Wrapper(pub Logger, Config);`) nest their
|
||||
# positional field types directly under ordered_field_declaration_list
|
||||
# with no field_declaration wrapper -- the same shape handled for tuple
|
||||
# enum variants below. Without this branch these field type references
|
||||
# are silently dropped.
|
||||
for c in node.children:
|
||||
if c.type != "ordered_field_declaration_list":
|
||||
continue
|
||||
fline = c.start_point[0] + 1
|
||||
for tc in c.children:
|
||||
if tc.type not in ("type_identifier", "generic_type",
|
||||
"scoped_type_identifier", "reference_type",
|
||||
"primitive_type", "tuple_type", "array_type"):
|
||||
continue
|
||||
refs = []
|
||||
_rust_collect_type_refs(tc, source, False, refs)
|
||||
for ref_name, role in refs:
|
||||
ctx = "generic_arg" if role == "generic_arg" else "field"
|
||||
tgt = ensure_named_node(ref_name, fline)
|
||||
if tgt != item_nid:
|
||||
add_edge(item_nid, tgt, "references", fline, context=ctx)
|
||||
if t == "enum_item":
|
||||
# Variant payload types nest under enum_variant_list ->
|
||||
# enum_variant -> ordered_field_declaration_list (tuple variant,
|
||||
# `Click(Logger)`) | field_declaration_list (struct variant,
|
||||
# `Resize { size: Dim }`). Neither was traversed, so every
|
||||
# enum-variant type reference was silently dropped.
|
||||
_TYPE_NODES = ("type_identifier", "generic_type",
|
||||
"scoped_type_identifier", "reference_type",
|
||||
"primitive_type", "tuple_type", "array_type")
|
||||
|
||||
def _emit_enum_type(type_node, at_line):
|
||||
if type_node is None:
|
||||
return
|
||||
refs2: list[tuple[str, str]] = []
|
||||
_rust_collect_type_refs(type_node, source, False, refs2)
|
||||
for ref_name, role in refs2:
|
||||
ctx = "generic_arg" if role == "generic_arg" else "field"
|
||||
tgt = ensure_named_node(ref_name, at_line)
|
||||
if tgt != item_nid:
|
||||
add_edge(item_nid, tgt, "references", at_line, context=ctx)
|
||||
|
||||
for c in node.children:
|
||||
if c.type != "enum_variant_list":
|
||||
continue
|
||||
for variant in c.children:
|
||||
if variant.type != "enum_variant":
|
||||
continue
|
||||
vline = variant.start_point[0] + 1
|
||||
for vc in variant.children:
|
||||
if vc.type == "ordered_field_declaration_list":
|
||||
for tc in vc.children:
|
||||
if tc.type in _TYPE_NODES:
|
||||
_emit_enum_type(tc, vline)
|
||||
elif vc.type == "field_declaration_list":
|
||||
for field in vc.children:
|
||||
if field.type != "field_declaration":
|
||||
continue
|
||||
type_node = field.child_by_field_name("type")
|
||||
_emit_enum_type(type_node, field.start_point[0] + 1)
|
||||
return
|
||||
|
||||
if t == "impl_item":
|
||||
type_node = node.child_by_field_name("type")
|
||||
trait_node = node.child_by_field_name("trait")
|
||||
impl_nid: str | None = None
|
||||
if type_node:
|
||||
type_name = _read_text(type_node, source).strip()
|
||||
impl_nid = _make_id(stem, type_name)
|
||||
add_node(impl_nid, type_name, node.start_point[0] + 1)
|
||||
if trait_node is not None and impl_nid is not None:
|
||||
refs: list[tuple[str, str]] = []
|
||||
_rust_collect_type_refs(trait_node, source, False, refs)
|
||||
for idx, (ref_name, _role) in enumerate(refs):
|
||||
tgt = ensure_named_node(ref_name, node.start_point[0] + 1)
|
||||
if tgt == impl_nid:
|
||||
continue
|
||||
if idx == 0:
|
||||
add_edge(impl_nid, tgt, "implements", node.start_point[0] + 1)
|
||||
else:
|
||||
add_edge(impl_nid, tgt, "references", node.start_point[0] + 1,
|
||||
context="generic_arg")
|
||||
body = node.child_by_field_name("body")
|
||||
if body:
|
||||
for child in body.children:
|
||||
walk(child, parent_impl_nid=impl_nid)
|
||||
return
|
||||
|
||||
if t == "use_declaration":
|
||||
arg = node.child_by_field_name("argument")
|
||||
if arg:
|
||||
raw = _read_text(arg, source)
|
||||
clean = raw.split("{")[0].rstrip(":").rstrip("*").rstrip(":")
|
||||
module_name = clean.split("::")[-1].strip()
|
||||
if module_name:
|
||||
tgt_nid = _make_id(module_name)
|
||||
add_edge(file_nid, tgt_nid, "imports_from", node.start_point[0] + 1, context="import")
|
||||
return
|
||||
|
||||
for child in node.children:
|
||||
walk(child, parent_impl_nid=None)
|
||||
|
||||
walk(root)
|
||||
|
||||
label_to_nid: dict[str, str] = {}
|
||||
for n in nodes:
|
||||
raw = n["label"]
|
||||
normalised = raw.strip("()").lstrip(".")
|
||||
label_to_nid[normalised] = n["id"]
|
||||
|
||||
seen_call_pairs: set[tuple[str, str]] = set()
|
||||
raw_calls: list[dict] = []
|
||||
|
||||
def walk_calls(node, caller_nid: str) -> None:
|
||||
if node.type == "function_item":
|
||||
return
|
||||
if node.type == "call_expression":
|
||||
func_node = node.child_by_field_name("function")
|
||||
callee_name: str | None = None
|
||||
is_member_call: bool = False
|
||||
is_scoped_call: bool = False
|
||||
if func_node:
|
||||
if func_node.type == "identifier":
|
||||
callee_name = _read_text(func_node, source)
|
||||
elif func_node.type == "field_expression":
|
||||
is_member_call = True
|
||||
field = func_node.child_by_field_name("field")
|
||||
if field:
|
||||
callee_name = _read_text(field, source)
|
||||
elif func_node.type == "scoped_identifier":
|
||||
# Type::method() — still allow in-file EXTRACTED match, but
|
||||
# skip cross-file resolution: bare last-segment lookup ignores
|
||||
# crate boundaries and produces spurious INFERRED edges (#908).
|
||||
is_scoped_call = True
|
||||
name = func_node.child_by_field_name("name")
|
||||
if name:
|
||||
callee_name = _read_text(name, source)
|
||||
if callee_name and callee_name not in _LANGUAGE_BUILTIN_GLOBALS:
|
||||
tgt_nid = label_to_nid.get(callee_name)
|
||||
if tgt_nid and tgt_nid != caller_nid:
|
||||
pair = (caller_nid, tgt_nid)
|
||||
if pair not in seen_call_pairs:
|
||||
seen_call_pairs.add(pair)
|
||||
line = node.start_point[0] + 1
|
||||
edges.append({
|
||||
"source": caller_nid,
|
||||
"target": tgt_nid,
|
||||
"relation": "calls",
|
||||
"context": "call",
|
||||
"confidence": "EXTRACTED",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
"weight": 1.0,
|
||||
})
|
||||
elif not is_scoped_call and callee_name.lower() not in _RUST_TRAIT_METHOD_BLOCKLIST:
|
||||
raw_calls.append({
|
||||
"caller_nid": caller_nid,
|
||||
"callee": callee_name,
|
||||
"is_member_call": is_member_call,
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{node.start_point[0] + 1}",
|
||||
})
|
||||
for child in node.children:
|
||||
walk_calls(child, caller_nid)
|
||||
|
||||
for caller_nid, body_node in function_bodies:
|
||||
walk_calls(body_node, caller_nid)
|
||||
|
||||
valid_ids = seen_ids
|
||||
clean_edges = []
|
||||
for edge in edges:
|
||||
src, tgt = edge["source"], edge["target"]
|
||||
if src in valid_ids and (tgt in valid_ids or edge["relation"] in ("imports", "imports_from")):
|
||||
clean_edges.append(edge)
|
||||
|
||||
return {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls}
|
||||
@@ -0,0 +1,81 @@
|
||||
"""Sln extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from pathlib import Path
|
||||
from graphify.extractors.base import _make_id
|
||||
|
||||
|
||||
def extract_sln(path: Path) -> dict:
|
||||
"""Extract projects and inter-project dependencies from a .sln file."""
|
||||
try:
|
||||
src = path.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return {"nodes": [], "edges": [], "error": f"cannot read {path}"}
|
||||
|
||||
file_nid = _make_id(str(path))
|
||||
str_path = str(path)
|
||||
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
|
||||
"source_file": str_path, "source_location": None}]
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
seen_ids.add(file_nid)
|
||||
|
||||
_PROJECT_RE = re.compile(
|
||||
r'Project\("[^"]*"\)\s*=\s*"([^"]+)"\s*,\s*"([^"]+)"\s*,\s*"([^"]*)"'
|
||||
)
|
||||
_DEP_RE = re.compile(r'\{([0-9a-fA-F-]+)\}\s*=\s*\{([0-9a-fA-F-]+)\}')
|
||||
|
||||
guid_to_nid: dict[str, str] = {}
|
||||
|
||||
for m in _PROJECT_RE.finditer(src):
|
||||
proj_name = m.group(1)
|
||||
proj_path = m.group(2).replace("\\", "/")
|
||||
proj_guid = m.group(3).strip("{}")
|
||||
|
||||
try:
|
||||
abs_proj = str((path.parent / proj_path).resolve())
|
||||
except Exception:
|
||||
abs_proj = proj_path
|
||||
proj_nid = _make_id(abs_proj)
|
||||
if proj_nid and proj_nid not in seen_ids:
|
||||
seen_ids.add(proj_nid)
|
||||
nodes.append({"id": proj_nid, "label": proj_name,
|
||||
"file_type": "code", "source_file": abs_proj,
|
||||
"source_location": None})
|
||||
edges.append({"source": file_nid, "target": proj_nid,
|
||||
"relation": "contains", "confidence": "EXTRACTED",
|
||||
"source_file": str_path, "weight": 1.0})
|
||||
if proj_guid:
|
||||
guid_to_nid[proj_guid.lower()] = proj_nid
|
||||
|
||||
in_dep_section = False
|
||||
current_proj_guid: str | None = None
|
||||
_PROJECT_LINE_RE = re.compile(r'Project\("[^"]*"\)\s*=\s*"[^"]+"\s*,\s*"[^"]+"\s*,\s*"\{([^}]+)\}"')
|
||||
for line in src.splitlines():
|
||||
proj_line_m = _PROJECT_LINE_RE.search(line)
|
||||
if proj_line_m:
|
||||
current_proj_guid = proj_line_m.group(1).lower()
|
||||
continue
|
||||
if line.strip() == "EndProject":
|
||||
current_proj_guid = None
|
||||
continue
|
||||
if "ProjectSection(ProjectDependencies)" in line:
|
||||
in_dep_section = True
|
||||
continue
|
||||
if in_dep_section and "EndProjectSection" in line:
|
||||
in_dep_section = False
|
||||
continue
|
||||
if in_dep_section and current_proj_guid:
|
||||
dep_m = _DEP_RE.search(line)
|
||||
if dep_m:
|
||||
to_guid = dep_m.group(1).lower()
|
||||
from_nid = guid_to_nid.get(current_proj_guid)
|
||||
to_nid = guid_to_nid.get(to_guid)
|
||||
if from_nid and to_nid and from_nid != to_nid:
|
||||
edges.append({"source": from_nid, "target": to_nid,
|
||||
"relation": "imports", "confidence": "EXTRACTED",
|
||||
"source_file": str_path, "weight": 1.0})
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
@@ -0,0 +1,276 @@
|
||||
"""Sql extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from pathlib import Path
|
||||
from graphify.extractors.base import _file_stem, _make_id
|
||||
|
||||
|
||||
def extract_sql(path: Path, content: str | bytes | None = None) -> dict:
|
||||
"""Extract tables, views, functions, and relationships from .sql files via tree-sitter."""
|
||||
try:
|
||||
import tree_sitter_sql as tssql
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree_sitter_sql not installed. Run: pip install tree-sitter-sql"}
|
||||
|
||||
try:
|
||||
language = Language(tssql.language())
|
||||
parser = Parser(language)
|
||||
source = (
|
||||
content.encode("utf-8") if isinstance(content, str)
|
||||
else content if content is not None
|
||||
else path.read_bytes()
|
||||
)
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
|
||||
stem = _file_stem(path)
|
||||
str_path = str(path)
|
||||
file_nid = _make_id(str_path)
|
||||
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
|
||||
"source_file": str_path, "source_location": None}]
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = {file_nid}
|
||||
table_nids: dict[str, str] = {} # name → nid for reference resolution
|
||||
|
||||
def _read(n) -> str:
|
||||
return source[n.start_byte:n.end_byte].decode("utf-8", errors="replace")
|
||||
|
||||
def _obj_name(n) -> str | None:
|
||||
for c in n.children:
|
||||
if c.type == "object_reference":
|
||||
return _read(c)
|
||||
return None
|
||||
|
||||
def _add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({"id": nid, "label": label, "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line}"})
|
||||
edges.append({"source": file_nid, "target": nid, "relation": "contains",
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": 1.0})
|
||||
|
||||
def _add_edge(src: str, tgt: str, relation: str, line: int) -> None:
|
||||
edges.append({"source": src, "target": tgt, "relation": relation,
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": 1.0})
|
||||
|
||||
def walk(node) -> None:
|
||||
t = node.type
|
||||
line = node.start_point[0] + 1
|
||||
|
||||
if t == "create_table":
|
||||
name = _obj_name(node)
|
||||
if name:
|
||||
nid = _make_id(stem, name)
|
||||
_add_node(nid, name, line)
|
||||
table_nids[name.lower()] = nid
|
||||
# Foreign key REFERENCES
|
||||
for col in node.children:
|
||||
if col.type == "column_definitions":
|
||||
has_error = any(cd.type == "ERROR" for cd in col.children)
|
||||
seen_refs: set[str] = set()
|
||||
for cd in col.children:
|
||||
if cd.type == "column_definition":
|
||||
# Inline column-level REFERENCES
|
||||
ref_name: str | None = None
|
||||
found_ref = False
|
||||
for cc in cd.children:
|
||||
if cc.type == "keyword_references":
|
||||
found_ref = True
|
||||
elif found_ref and cc.type == "object_reference":
|
||||
ref_name = _read(cc)
|
||||
break
|
||||
if ref_name:
|
||||
ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name)
|
||||
_add_edge(nid, ref_nid, "references", line)
|
||||
seen_refs.add(ref_name.lower())
|
||||
elif cd.type == "constraints":
|
||||
# Table-level FOREIGN KEY ... REFERENCES ... constraints
|
||||
for constraint in cd.children:
|
||||
if constraint.type != "constraint":
|
||||
continue
|
||||
ref_name = None
|
||||
found_ref = False
|
||||
for cc in constraint.children:
|
||||
if cc.type == "keyword_references":
|
||||
found_ref = True
|
||||
elif found_ref and cc.type == "object_reference":
|
||||
ref_name = _read(cc)
|
||||
break
|
||||
if ref_name:
|
||||
ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name)
|
||||
_add_edge(nid, ref_nid, "references", line)
|
||||
seen_refs.add(ref_name.lower())
|
||||
if has_error:
|
||||
# Dialect-specific syntax (e.g. Firebird COMPUTED BY) causes ERROR
|
||||
# nodes that make the parser drop the trailing constraints block.
|
||||
# Regex-scan the raw column_definitions text as fallback.
|
||||
col_text = _read(col)
|
||||
for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", col_text, re.IGNORECASE):
|
||||
ref_name = rm.group(1)
|
||||
if ref_name.lower() not in seen_refs:
|
||||
ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name)
|
||||
_add_edge(nid, ref_nid, "references", line)
|
||||
seen_refs.add(ref_name.lower())
|
||||
|
||||
elif t == "create_view":
|
||||
name = _obj_name(node)
|
||||
if name:
|
||||
nid = _make_id(stem, name)
|
||||
_add_node(nid, name, line)
|
||||
table_nids[name.lower()] = nid
|
||||
# FROM/JOIN table references inside view body
|
||||
_walk_from_refs(node, nid, line)
|
||||
|
||||
elif t == "create_function":
|
||||
name = _obj_name(node)
|
||||
if name:
|
||||
nid = _make_id(stem, name)
|
||||
_add_node(nid, f"{name}()", line)
|
||||
_walk_from_refs(node, nid, line)
|
||||
|
||||
elif t == "create_procedure":
|
||||
name = _obj_name(node)
|
||||
if name:
|
||||
nid = _make_id(stem, name)
|
||||
_add_node(nid, f"{name}()", line)
|
||||
_walk_from_refs(node, nid, line)
|
||||
|
||||
elif t == "alter_table":
|
||||
name = _obj_name(node)
|
||||
if name:
|
||||
src_nid = table_nids.get(name.lower())
|
||||
if not src_nid:
|
||||
src_nid = _make_id(stem, name)
|
||||
_add_node(src_nid, name, line)
|
||||
table_nids[name.lower()] = src_nid
|
||||
for child in node.children:
|
||||
if child.type == "add_constraint":
|
||||
for cc in child.children:
|
||||
if cc.type != "constraint":
|
||||
continue
|
||||
found_ref = False
|
||||
ref_name: str | None = None
|
||||
for ccc in cc.children:
|
||||
if ccc.type == "keyword_references":
|
||||
found_ref = True
|
||||
elif found_ref and ccc.type == "object_reference":
|
||||
ref_name = _read(ccc)
|
||||
break
|
||||
if ref_name:
|
||||
ref_nid = table_nids.get(ref_name.lower())
|
||||
if not ref_nid:
|
||||
ref_nid = _make_id(stem, ref_name)
|
||||
_add_edge(src_nid, ref_nid, "references", line)
|
||||
|
||||
elif t == "create_trigger":
|
||||
trig_name: str | None = None
|
||||
tbl_name: str | None = None
|
||||
after_trigger = False
|
||||
after_for = False
|
||||
for c in node.children:
|
||||
if c.type == "keyword_trigger":
|
||||
after_trigger = True
|
||||
elif after_trigger and not trig_name and c.type == "object_reference":
|
||||
trig_name = _read(c)
|
||||
elif c.type == "keyword_for":
|
||||
after_for = True
|
||||
elif after_for and not tbl_name and c.type == "object_reference":
|
||||
tbl_name = _read(c)
|
||||
if trig_name:
|
||||
trig_nid = _make_id(stem, trig_name)
|
||||
_add_node(trig_nid, trig_name, line)
|
||||
if tbl_name:
|
||||
tbl_nid = table_nids.get(tbl_name.lower()) or _make_id(stem, tbl_name)
|
||||
_add_edge(trig_nid, tbl_nid, "triggers", line)
|
||||
|
||||
elif t == "fb_proc_or_trigger":
|
||||
text = _read(node)
|
||||
m = re.match(
|
||||
r"CREATE\s+(?:OR\s+(?:REPLACE|ALTER)\s+)?"
|
||||
r"(PROCEDURE|TRIGGER|FUNCTION)\s+([\w$]+)",
|
||||
text, re.IGNORECASE,
|
||||
)
|
||||
if m:
|
||||
obj_type = m.group(1).upper()
|
||||
obj_name = m.group(2)
|
||||
obj_nid = _make_id(stem, obj_name)
|
||||
label = obj_name if obj_type == "TRIGGER" else f"{obj_name}()"
|
||||
_add_node(obj_nid, label, line)
|
||||
if obj_type == "TRIGGER":
|
||||
fm = re.search(r"\bFOR\s+([\w$]+)", text, re.IGNORECASE)
|
||||
if fm:
|
||||
tbl = fm.group(1)
|
||||
tbl_nid = table_nids.get(tbl.lower()) or _make_id(stem, tbl)
|
||||
_add_edge(obj_nid, tbl_nid, "triggers", line)
|
||||
_NON_TABLES = {
|
||||
"select", "where", "set", "dual", "null", "true", "false",
|
||||
"first", "skip", "rows", "next", "only", "lateral",
|
||||
}
|
||||
seen_tbls: set[str] = set()
|
||||
for rm in re.finditer(r"\b(?:FROM|JOIN|INTO)\s+([\w$]+)", text, re.IGNORECASE):
|
||||
tbl = rm.group(1)
|
||||
if tbl.lower() not in _NON_TABLES and tbl.lower() not in seen_tbls:
|
||||
seen_tbls.add(tbl.lower())
|
||||
tbl_nid = table_nids.get(tbl.lower()) or _make_id(stem, tbl)
|
||||
_add_edge(obj_nid, tbl_nid, "reads_from", line)
|
||||
for rm in re.finditer(r"\bUPDATE\s+([\w$]+)", text, re.IGNORECASE):
|
||||
tbl = rm.group(1)
|
||||
if tbl.lower() not in _NON_TABLES and tbl.lower() not in seen_tbls:
|
||||
seen_tbls.add(tbl.lower())
|
||||
tbl_nid = table_nids.get(tbl.lower()) or _make_id(stem, tbl)
|
||||
_add_edge(obj_nid, tbl_nid, "reads_from", line)
|
||||
|
||||
for child in node.children:
|
||||
walk(child)
|
||||
|
||||
def _walk_from_refs(node, caller_nid: str, line: int) -> None:
|
||||
"""Recursively find FROM/JOIN table references inside a node."""
|
||||
if node.type in ("from", "join"):
|
||||
for c in node.children:
|
||||
if c.type == "relation":
|
||||
for cc in c.children:
|
||||
if cc.type == "object_reference":
|
||||
tbl = _read(cc)
|
||||
tbl_nid = _make_id(stem, tbl)
|
||||
_add_edge(caller_nid, tbl_nid, "reads_from",
|
||||
c.start_point[0] + 1)
|
||||
for child in node.children:
|
||||
_walk_from_refs(child, caller_nid, line)
|
||||
|
||||
for stmt in root.children:
|
||||
if stmt.type == "statement":
|
||||
for child in stmt.children:
|
||||
walk(child)
|
||||
elif stmt.type in ("fb_proc_or_trigger", "set_term", "declare_external_function"):
|
||||
walk(stmt)
|
||||
|
||||
# Global regex fallback: catch any REFERENCES missed due to ERROR nodes in the parse tree
|
||||
# (e.g. Firebird COMPUTED BY columns push constraints out of the tree entirely).
|
||||
# Snapshot after tree walk so we don't re-emit edges already captured above.
|
||||
emitted = {(e["source"], e["target"]) for e in edges if e["relation"] == "references"}
|
||||
src_text = source.decode("utf-8", errors="replace")
|
||||
for m in re.finditer(r"CREATE\s+TABLE\s+([\w$]+)\s*\(", src_text, re.IGNORECASE):
|
||||
tbl_name = m.group(1)
|
||||
tbl_nid = table_nids.get(tbl_name.lower())
|
||||
if tbl_nid is None:
|
||||
continue
|
||||
tbl_line = src_text[: m.start()].count("\n") + 1
|
||||
tail = src_text[m.start():]
|
||||
end = re.search(r"(?:^|\n)(?:CREATE|SET\s+TERM|ALTER)\s", tail[1:], re.IGNORECASE)
|
||||
block = tail[: end.start() + 1] if end else tail
|
||||
for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", block, re.IGNORECASE):
|
||||
ref_name = rm.group(1)
|
||||
ref_nid = table_nids.get(ref_name.lower()) or _make_id(stem, ref_name)
|
||||
if (tbl_nid, ref_nid) not in emitted:
|
||||
_add_edge(tbl_nid, ref_nid, "references", tbl_line)
|
||||
emitted.add((tbl_nid, ref_nid))
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
@@ -0,0 +1,181 @@
|
||||
"""Terraform extractor. Moved verbatim from graphify/extract.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
from pathlib import Path
|
||||
from graphify.extractors.base import _make_id
|
||||
|
||||
|
||||
_TF_META_HEADS = frozenset({"count", "each", "self", "path", "terraform"})
|
||||
|
||||
def extract_terraform(path: Path) -> dict:
|
||||
"""Extract Terraform/HCL blocks and the references between them via tree-sitter.
|
||||
|
||||
Nodes: resources, data sources, modules, variables, outputs, providers, and
|
||||
locals. Edges: `contains` (file -> block), `references` (block -> the blocks
|
||||
it interpolates, e.g. `aws_instance.web` -> `var.region`), and `depends_on`
|
||||
(explicit dependency edges).
|
||||
|
||||
Node IDs are scoped by the parent directory, not the file stem, because
|
||||
Terraform resources are module(directory)-scoped: a resource defined in
|
||||
main.tf is referenced from other .tf files in the same directory. Directory
|
||||
scoping lets those cross-file references resolve when per-file extractions
|
||||
are merged (stem scoping would split a definition from its references).
|
||||
"""
|
||||
try:
|
||||
import tree_sitter_hcl as tshcl
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {"nodes": [], "edges": [], "error": "tree_sitter_hcl not installed. Run: pip install tree-sitter-hcl"}
|
||||
|
||||
try:
|
||||
language = Language(tshcl.language())
|
||||
parser = Parser(language)
|
||||
source = path.read_bytes()
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
|
||||
str_path = str(path)
|
||||
file_nid = _make_id(str_path)
|
||||
scope = path.parent.name or "tf"
|
||||
|
||||
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
|
||||
"source_file": str_path, "source_location": None}]
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = {file_nid}
|
||||
seen_edges: set[tuple[str, str, str]] = set()
|
||||
|
||||
def _read(n) -> str:
|
||||
return source[n.start_byte:n.end_byte].decode("utf-8", errors="replace")
|
||||
|
||||
def _label_text(n) -> str:
|
||||
return _read(n).strip().strip('"')
|
||||
|
||||
def _add_node(address: str, label: str, line: int) -> str:
|
||||
nid = _make_id(scope, address)
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({"id": nid, "label": label, "file_type": "code",
|
||||
"source_file": str_path, "source_location": f"L{line}"})
|
||||
edges.append({"source": file_nid, "target": nid, "relation": "contains",
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": 1.0})
|
||||
return nid
|
||||
|
||||
def _add_edge(src: str, address: str, relation: str, line: int) -> None:
|
||||
tgt = _make_id(scope, address)
|
||||
if src == tgt:
|
||||
return
|
||||
key = (src, tgt, relation)
|
||||
if key in seen_edges:
|
||||
return
|
||||
seen_edges.add(key)
|
||||
edges.append({"source": src, "target": tgt, "relation": relation,
|
||||
"confidence": "EXTRACTED", "source_file": str_path,
|
||||
"source_location": f"L{line}", "weight": 1.0})
|
||||
|
||||
def _block_parts(block) -> tuple:
|
||||
btype = None
|
||||
labels: list[str] = []
|
||||
for c in block.children:
|
||||
if c.type in ("block_start", "body", "block_end"):
|
||||
break
|
||||
if c.type == "identifier" and btype is None:
|
||||
btype = _read(c)
|
||||
elif c.type in ("string_lit", "identifier"):
|
||||
labels.append(_label_text(c))
|
||||
return btype, labels
|
||||
|
||||
def _ref_address(expr):
|
||||
head = _read(expr)
|
||||
parent = expr.parent
|
||||
attrs: list[str] = []
|
||||
if parent is not None:
|
||||
seen_self = False
|
||||
for c in parent.children:
|
||||
if c.id == expr.id:
|
||||
seen_self = True
|
||||
continue
|
||||
if seen_self and c.type == "get_attr":
|
||||
name = None
|
||||
for gc in c.children:
|
||||
if gc.type == "identifier":
|
||||
name = _read(gc)
|
||||
break
|
||||
if name is None:
|
||||
break
|
||||
attrs.append(name)
|
||||
elif seen_self and c.type not in ("get_attr",):
|
||||
break
|
||||
if head in _TF_META_HEADS or not head:
|
||||
return None
|
||||
if head == "var":
|
||||
return f"var.{attrs[0]}" if attrs else None
|
||||
if head == "local":
|
||||
return f"local.{attrs[0]}" if attrs else None
|
||||
if head == "module":
|
||||
return f"module.{attrs[0]}" if attrs else None
|
||||
if head == "data":
|
||||
return f"data.{attrs[0]}.{attrs[1]}" if len(attrs) >= 2 else None
|
||||
return f"{head}.{attrs[0]}" if attrs else None
|
||||
|
||||
def _collect_refs(node, owner_nid: str, relation: str) -> None:
|
||||
rel = relation
|
||||
if node.type == "attribute":
|
||||
key_node = node.child_by_field_name("key") or (
|
||||
node.children[0] if node.children else None
|
||||
)
|
||||
if key_node is not None and _read(key_node) == "depends_on":
|
||||
rel = "depends_on"
|
||||
if node.type == "variable_expr":
|
||||
addr = _ref_address(node)
|
||||
if addr:
|
||||
_add_edge(owner_nid, addr, rel, node.start_point[0] + 1)
|
||||
for c in node.children:
|
||||
if c.is_named:
|
||||
_collect_refs(c, owner_nid, rel)
|
||||
|
||||
def _body_of(block):
|
||||
for c in block.children:
|
||||
if c.type == "body":
|
||||
return c
|
||||
return None
|
||||
|
||||
body = next((c for c in root.children if c.type == "body"), root)
|
||||
for block in body.children:
|
||||
if block.type != "block":
|
||||
continue
|
||||
btype, labels = _block_parts(block)
|
||||
line = block.start_point[0] + 1
|
||||
blk_body = _body_of(block)
|
||||
if btype == "resource" and len(labels) >= 2:
|
||||
owner = _add_node(f"{labels[0]}.{labels[1]}", f"{labels[0]}.{labels[1]}", line)
|
||||
elif btype == "data" and len(labels) >= 2:
|
||||
owner = _add_node(f"data.{labels[0]}.{labels[1]}", f"data.{labels[0]}.{labels[1]}", line)
|
||||
elif btype == "module" and labels:
|
||||
owner = _add_node(f"module.{labels[0]}", f"module.{labels[0]}", line)
|
||||
elif btype == "variable" and labels:
|
||||
owner = _add_node(f"var.{labels[0]}", f"var.{labels[0]}", line)
|
||||
elif btype == "output" and labels:
|
||||
owner = _add_node(f"output.{labels[0]}", f"output.{labels[0]}", line)
|
||||
elif btype == "provider" and labels:
|
||||
owner = _add_node(f"provider.{labels[0]}", f"provider.{labels[0]}", line)
|
||||
elif btype == "locals" and blk_body is not None:
|
||||
for attr in blk_body.children:
|
||||
if attr.type != "attribute":
|
||||
continue
|
||||
key_node = attr.children[0] if attr.children else None
|
||||
if key_node is None:
|
||||
continue
|
||||
key = _read(key_node)
|
||||
lnid = _add_node(f"local.{key}", f"local.{key}", attr.start_point[0] + 1)
|
||||
_collect_refs(attr, lnid, "references")
|
||||
continue
|
||||
else:
|
||||
continue
|
||||
if blk_body is not None:
|
||||
_collect_refs(blk_body, owner, "references")
|
||||
|
||||
return {"nodes": nodes, "edges": edges}
|
||||
Reference in New Issue
Block a user