From 8e6483511dc30ad096fcb10cc79cf9cc682b4a83 Mon Sep 17 00:00:00 2001 From: Safi Date: Sat, 9 May 2026 15:52:17 +0100 Subject: [PATCH] add Pascal regex fallback - extraction works without tree-sitter-pascal tree-sitter-pascal is not on PyPI so extract_pascal() fell back to an empty error result for all users. _extract_pascal_regex() now handles unit/program/library headers, uses clauses, class/interface declarations with inheritance, forward method decls, qualified impl headers, balanced begin/end body extraction, and intra-file calls edges. All 15 previously skipped pascal_required tests now run unconditionally and pass. Co-Authored-By: Claude Sonnet 4.6 --- graphify/extract.py | 305 +++++++++++++++++++++++++++++++++++++++++-- tests/test_pascal.py | 30 ----- 2 files changed, 297 insertions(+), 38 deletions(-) diff --git a/graphify/extract.py b/graphify/extract.py index b5b4ef7..0233083 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -4620,6 +4620,297 @@ def _pascal_resolve_class(from_path: Path, class_name: str) -> str | None: return None +_PAS_TOKEN_RE = re.compile( + r"'(?:''|[^'])*'" + r"|\{[^}]*\}" + r"|\(\*.*?\*\)" + r"|//[^\n]*", + re.DOTALL, +) +_PAS_MODULE_RE = re.compile( + r"\b(unit|program|library)\s+([A-Za-z_][\w.]*)\s*;", + re.IGNORECASE, +) +_PAS_USES_RE = re.compile( + r"\buses\b\s*([^;]+);", + re.IGNORECASE | re.DOTALL, +) +_PAS_TYPE_HEADER_RE = re.compile( + r"\b(?P[A-Za-z_]\w*)(?:\s*<[^>]+>)?\s*=\s*(?:packed\s+)?" + r"(?Pclass|interface)\b" + r"(?:\s*\(\s*(?P[^)]*)\s*\))?", + re.IGNORECASE, +) +_PAS_END_SEMI_RE = re.compile(r"\bend\s*;", re.IGNORECASE) +_PAS_METHOD_DECL_RE = re.compile( + r"\b(?:procedure|function|constructor|destructor)\s+" + r"(?P[A-Za-z_]\w*)" + r"(?:\s*\([^)]*\))?" + r"(?:\s*:\s*[\w<>,\s.]+)?" + r"\s*;", + re.IGNORECASE, +) +_PAS_IMPL_HEADER_RE = re.compile( + r"\b(?:procedure|function|constructor|destructor)\s+" + r"(?P[A-Za-z_]\w*(?:\.[A-Za-z_]\w*)?)" + r"(?:\s*<[^>]+>)?" + r"(?:\s*\([^)]*\))?" + r"(?:\s*:\s*[\w<>,\s.]+)?" + r"\s*;", + re.IGNORECASE, +) +_PAS_BEGIN_END_TOKEN_RE = re.compile( + r"\b(begin|end|case|try|asm|record)\b", re.IGNORECASE +) +_PAS_CALL_RE = re.compile(r"\b([A-Za-z_]\w*(?:\.[A-Za-z_]\w*)*)\s*[(;]") +_PAS_KEYWORDS = frozenset({ + "begin", "end", "if", "then", "else", "while", "do", "for", "to", + "downto", "repeat", "until", "case", "of", "try", "finally", "except", + "with", "inherited", "result", "var", "const", "type", "nil", "true", + "false", "exit", "break", "continue", "uses", "unit", "program", + "library", "interface", "implementation", "initialization", "finalization", + "procedure", "function", "constructor", "destructor", "class", "record", + "object", "array", "string", "integer", "boolean", "real", "char", + "writeln", "write", "readln", "read", "assigned", "length", "high", + "low", "inc", "dec", "new", "dispose", "setlength", "copy", "pos", + "trim", "format", "inttostr", "strtoint", "ord", "chr", "sizeof", + "create", "free", "destroy", +}) + + +def _pascal_strip_comments(text: str) -> str: + """Strip Pascal comments ({}, (* *), //) while preserving newlines.""" + def _sub(m: re.Match) -> str: + tok = m.group(0) + if tok.startswith("'"): + return tok + return "".join(c if c == "\n" else " " for c in tok) + return _PAS_TOKEN_RE.sub(_sub, text) + + +def _pascal_split_sections(text: str) -> tuple[str, int, str, int]: + """Split into (iface_text, iface_offset, impl_text, impl_offset). + Files without interface/implementation sections (dpr/lpr/inc) return + the whole text as impl with offset 0. + """ + iface_m = re.search(r"\binterface\b", text, re.IGNORECASE) + impl_m = re.search(r"\bimplementation\b", text, re.IGNORECASE) + if iface_m and impl_m: + iface_off = iface_m.end() + impl_off = impl_m.end() + end_m = re.search( + r"\b(initialization|finalization)\b", text[impl_off:], re.IGNORECASE + ) + impl_end = impl_off + end_m.start() if end_m else len(text) + return text[iface_off:impl_m.start()], iface_off, text[impl_off:impl_end], impl_off + return "", 0, text, 0 + + +def _pascal_split_uses(s: str) -> list[str]: + """Split a uses list string, handling 'Foo in ''bar.pas''' syntax.""" + out = [] + for chunk in s.split(","): + name = re.split(r"\s+in\s+", chunk.strip(), maxsplit=1, flags=re.IGNORECASE)[0] + name = name.strip().strip(";") + if name and re.match(r"[A-Za-z_][\w.]*$", name): + out.append(name) + return out + + +def _pascal_split_bases(s: str) -> list[str]: + """Split inheritance list, handling generics like TList.""" + out, depth, buf = [], 0, [] + for ch in s: + if ch == "<": + depth += 1 + buf.append(ch) + elif ch == ">": + depth -= 1 + buf.append(ch) + elif ch == "," and depth == 0: + name = re.sub(r"<.*$", "", "".join(buf).strip()) + if name: + out.append(name) + buf = [] + else: + buf.append(ch) + name = re.sub(r"<.*$", "", "".join(buf).strip()) + if name: + out.append(name) + return [n for n in out if re.match(r"[A-Za-z_]\w*$", n)] + + +def _pascal_find_body(text: str, start: int) -> tuple[int, int]: + """Find balanced begin..end after start. Returns (body_start, body_end). + Returns (0, 0) if no begin found. + """ + m = re.search(r"\bbegin\b", text[start:], re.IGNORECASE) + if not m: + return (0, 0) + body_start = start + m.end() + depth = 1 + for tok in _PAS_BEGIN_END_TOKEN_RE.finditer(text, body_start): + kw = tok.group(1).lower() + if kw in ("begin", "case", "try", "asm", "record"): + depth += 1 + elif kw == "end": + depth -= 1 + if depth == 0: + return (body_start, tok.start()) + return (body_start, len(text)) + + +def _extract_pascal_regex(path: Path) -> dict: + """Regex fallback for Pascal/Delphi extraction when tree-sitter-pascal + is unavailable. Produces the same node/edge schema as the tree-sitter pass. + """ + try: + raw = path.read_text(encoding="utf-8", errors="replace") + except Exception as exc: + return {"nodes": [], "edges": [], "error": str(exc)} + + str_path = str(path) + stem = _file_stem(path) + nodes: list[dict] = [] + edges: list[dict] = [] + seen_ids: set[str] = set() + seen_call_pairs: set[tuple[str, str]] = set() + + def _add_node(nid: str, label: str, line: int) -> None: + if nid not in seen_ids: + seen_ids.add(nid) + nodes.append({ + "id": nid, + "label": label, + "file_type": "code", + "source_file": str_path, + "source_location": f"L{line}", + }) + + def _add_edge(src: str, tgt: str, relation: str, line: int, context: str | None = None) -> None: + edge: dict = { + "source": src, + "target": tgt, + "relation": relation, + "confidence": "EXTRACTED", + "source_file": str_path, + "source_location": f"L{line}", + "weight": 1.0, + } + if context: + edge["context"] = context + edges.append(edge) + + def _lineno(text: str, offset: int) -> int: + return text.count("\n", 0, offset) + 1 + + file_nid = _make_id(str_path) + _add_node(file_nid, path.name, 1) + + stripped = _pascal_strip_comments(raw) + + # Module header + module_nid = file_nid + mod_m = _PAS_MODULE_RE.search(stripped) + if mod_m: + mod_name = mod_m.group(2) + module_nid = _make_id(stem, mod_name) + _add_node(module_nid, mod_name, _lineno(stripped, mod_m.start())) + _add_edge(file_nid, module_nid, "contains", _lineno(stripped, mod_m.start())) + + iface_text, iface_off, impl_text, impl_off = _pascal_split_sections(stripped) + + # Uses clauses + for section_text, section_off in ((iface_text, iface_off), (impl_text, impl_off)): + for um in _PAS_USES_RE.finditer(section_text): + line = _lineno(stripped, section_off + um.start()) + for unit_name in _pascal_split_uses(um.group(1)): + tgt_nid = _pascal_resolve_unit(path, unit_name) + _add_edge(module_nid, tgt_nid, "imports", line, context="import") + + # Type declarations (classes / interfaces) in interface section + search_text = iface_text if iface_text else stripped + search_off = iface_off if iface_text else 0 + pos = 0 + while pos < len(search_text): + hm = _PAS_TYPE_HEADER_RE.search(search_text, pos) + if not hm: + break + type_name = hm.group("name") + bases_raw = hm.group("bases") or "" + line = _lineno(stripped, search_off + hm.start()) + cls_nid = _make_id(stem, type_name) + _add_node(cls_nid, type_name, line) + _add_edge(module_nid, cls_nid, "contains", line) + + for base_name in _pascal_split_bases(bases_raw): + resolved = _pascal_resolve_class(path, base_name) + base_nid = resolved if resolved else _make_id(base_name) + if base_nid not in seen_ids: + _add_node(base_nid, base_name, line) + _add_edge(cls_nid, base_nid, "inherits", line) + + # Find class body (up to next end;) + end_m = _PAS_END_SEMI_RE.search(search_text, hm.end()) + body_text = search_text[hm.end():end_m.start()] if end_m else "" + body_off = search_off + hm.end() + + # Forward method declarations inside the class body + for mm in _PAS_METHOD_DECL_RE.finditer(body_text): + mname = mm.group("name") + mline = _lineno(stripped, body_off + mm.start()) + method_nid = _make_id(cls_nid, mname) + _add_node(method_nid, f"{mname}()", mline) + _add_edge(cls_nid, method_nid, "method", mline) + + pos = end_m.end() if end_m else len(search_text) + + # Implementation headers (procedure/function/constructor/destructor) + impl_records: list[tuple[str, int, str]] = [] + for fm in _PAS_IMPL_HEADER_RE.finditer(impl_text): + qualified = fm.group("qual") + line = _lineno(stripped, impl_off + fm.start()) + if "." in qualified: + cls_part, method_part = qualified.split(".", 1) + cls_nid = _make_id(stem, cls_part) + container = cls_nid if cls_nid in seen_ids else module_nid + relation = "method" if cls_nid in seen_ids else "contains" + label = f"{method_part}()" + else: + container, relation = module_nid, "contains" + label = f"{qualified}()" + proc_nid = _make_id(stem, qualified) + _add_node(proc_nid, label, line) + _add_edge(container, proc_nid, relation, line) + + body_start, body_end = _pascal_find_body(impl_text, fm.end()) + body_text = impl_text[body_start:body_end] if body_start else "" + impl_records.append((proc_nid, line, body_text)) + + # Intra-file call edges + all_procs: dict[str, str] = { + n["label"].removesuffix("()").lower(): n["id"] + for n in nodes + if n["id"] != file_nid and n["label"].endswith("()") + } + for caller_nid, caller_line, body_text in impl_records: + for cm in _PAS_CALL_RE.finditer(body_text): + callee_name = cm.group(1).split(".")[-1].lower() + if callee_name in _PAS_KEYWORDS: + continue + callee_nid = all_procs.get(callee_name) + if not callee_nid or callee_nid == caller_nid: + continue + pair = (caller_nid, callee_nid) + if pair in seen_call_pairs: + continue + seen_call_pairs.add(pair) + call_line = caller_line + body_text.count("\n", 0, cm.start()) + _add_edge(caller_nid, callee_nid, "calls", call_line, context="call") + + return {"nodes": nodes, "edges": edges, "input_tokens": 0, "output_tokens": 0} + + def extract_pascal(path: Path) -> dict: """Extract units, classes, procedures, uses-imports, and calls from Pascal/Delphi files. @@ -4637,17 +4928,15 @@ def extract_pascal(path: Path) -> dict: - class/module --contains--> procedure/function implementation - procedure --calls--> other procedure (within the same file) - Requires tree-sitter-pascal: pip install tree-sitter-pascal - (https://github.com/Isopod/tree-sitter-pascal) + Uses tree-sitter-pascal when available; falls back to a regex-based extractor + (_extract_pascal_regex) when it isn't installed or fails to parse, so Pascal + extraction works out of the box without an extra pip install. """ try: import tree_sitter_pascal as tspascal from tree_sitter import Language, Parser except ImportError: - return { - "nodes": [], "edges": [], - "error": "tree_sitter_pascal not installed. Run: pip install tree-sitter-pascal", - } + return _extract_pascal_regex(path) try: language = Language(tspascal.language()) @@ -4655,8 +4944,8 @@ def extract_pascal(path: Path) -> dict: source = path.read_bytes() tree = parser.parse(source) root = tree.root_node - except Exception as e: - return {"nodes": [], "edges": [], "error": str(e)} + except Exception: + return _extract_pascal_regex(path) stem = _file_stem(path) str_path = str(path) diff --git a/tests/test_pascal.py b/tests/test_pascal.py index 5b38dbd..36c1b87 100644 --- a/tests/test_pascal.py +++ b/tests/test_pascal.py @@ -1,25 +1,10 @@ """Tests for the Pascal/Delphi extractor.""" from __future__ import annotations -import pytest from pathlib import Path FIXTURES = Path(__file__).parent / "fixtures" -def _try_import(): - try: - import tree_sitter_pascal # noqa: F401 - return True - except ImportError: - return False - - -pascal_required = pytest.mark.skipif( - not _try_import(), - reason="tree_sitter_pascal not installed", -) - - def _labels(r): return [n["label"] for n in r["nodes"]] @@ -32,21 +17,18 @@ def _edges_with_relation(r, *relations): return [e for e in r["edges"] if e["relation"] in relations] -@pascal_required def test_pascal_no_error(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") assert "error" not in r -@pascal_required def test_pascal_finds_unit(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") assert any("SampleUnit" in l for l in _labels(r)) -@pascal_required def test_pascal_finds_classes(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") @@ -55,14 +37,12 @@ def test_pascal_finds_classes(): assert any("TDataProcessor" in l for l in labels) -@pascal_required def test_pascal_finds_interface(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") assert any("IProcessor" in l for l in _labels(r)) -@pascal_required def test_pascal_finds_methods(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") @@ -73,14 +53,12 @@ def test_pascal_finds_methods(): assert any("Reset" in l for l in labels) -@pascal_required def test_pascal_finds_imports(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") assert "imports" in _relations(r) -@pascal_required def test_pascal_import_edges_have_import_context(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") @@ -89,14 +67,12 @@ def test_pascal_import_edges_have_import_context(): assert all(e.get("context") == "import" for e in import_edges) -@pascal_required def test_pascal_finds_inherits(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") assert "inherits" in _relations(r) -@pascal_required def test_pascal_inherits_from_base(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") @@ -109,14 +85,12 @@ def test_pascal_inherits_from_base(): assert found, "TDataProcessor should have at least one inherits edge" -@pascal_required def test_pascal_finds_calls(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") assert "calls" in _relations(r) -@pascal_required def test_pascal_call_edges_have_call_context(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") @@ -125,7 +99,6 @@ def test_pascal_call_edges_have_call_context(): assert all(e.get("context") == "call" for e in call_edges) -@pascal_required def test_pascal_all_edges_extracted(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") @@ -135,7 +108,6 @@ def test_pascal_all_edges_extracted(): assert e["confidence"] == "EXTRACTED", f"Expected EXTRACTED: {e}" -@pascal_required def test_pascal_no_dangling_edges(): from graphify.extract import extract_pascal r = extract_pascal(FIXTURES / "sample.pas") @@ -148,7 +120,6 @@ def test_pascal_no_dangling_edges(): assert e["target"] in node_ids, f"Dangling target: {e}" -@pascal_required def test_pascal_dispatch_registered(): from graphify.extract import _DISPATCH assert ".pas" in _DISPATCH @@ -161,7 +132,6 @@ def test_pascal_dispatch_registered(): assert ".lpk" in _DISPATCH -@pascal_required def test_pascal_detect_extensions_registered(): from graphify.detect import CODE_EXTENSIONS assert ".pas" in CODE_EXTENSIONS