add Pascal regex fallback - extraction works without tree-sitter-pascal
tree-sitter-pascal is not on PyPI so extract_pascal() fell back to an empty error result for all users. _extract_pascal_regex() now handles unit/program/library headers, uses clauses, class/interface declarations with inheritance, forward method decls, qualified impl headers, balanced begin/end body extraction, and intra-file calls edges. All 15 previously skipped pascal_required tests now run unconditionally and pass. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 4.6
parent
32bf8b4a37
commit
8e6483511d
+297
-8
@@ -4620,6 +4620,297 @@ def _pascal_resolve_class(from_path: Path, class_name: str) -> str | None:
|
||||
return None
|
||||
|
||||
|
||||
_PAS_TOKEN_RE = re.compile(
|
||||
r"'(?:''|[^'])*'"
|
||||
r"|\{[^}]*\}"
|
||||
r"|\(\*.*?\*\)"
|
||||
r"|//[^\n]*",
|
||||
re.DOTALL,
|
||||
)
|
||||
_PAS_MODULE_RE = re.compile(
|
||||
r"\b(unit|program|library)\s+([A-Za-z_][\w.]*)\s*;",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_PAS_USES_RE = re.compile(
|
||||
r"\buses\b\s*([^;]+);",
|
||||
re.IGNORECASE | re.DOTALL,
|
||||
)
|
||||
_PAS_TYPE_HEADER_RE = re.compile(
|
||||
r"\b(?P<name>[A-Za-z_]\w*)(?:\s*<[^>]+>)?\s*=\s*(?:packed\s+)?"
|
||||
r"(?P<kind>class|interface)\b"
|
||||
r"(?:\s*\(\s*(?P<bases>[^)]*)\s*\))?",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_PAS_END_SEMI_RE = re.compile(r"\bend\s*;", re.IGNORECASE)
|
||||
_PAS_METHOD_DECL_RE = re.compile(
|
||||
r"\b(?:procedure|function|constructor|destructor)\s+"
|
||||
r"(?P<name>[A-Za-z_]\w*)"
|
||||
r"(?:\s*\([^)]*\))?"
|
||||
r"(?:\s*:\s*[\w<>,\s.]+)?"
|
||||
r"\s*;",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_PAS_IMPL_HEADER_RE = re.compile(
|
||||
r"\b(?:procedure|function|constructor|destructor)\s+"
|
||||
r"(?P<qual>[A-Za-z_]\w*(?:\.[A-Za-z_]\w*)?)"
|
||||
r"(?:\s*<[^>]+>)?"
|
||||
r"(?:\s*\([^)]*\))?"
|
||||
r"(?:\s*:\s*[\w<>,\s.]+)?"
|
||||
r"\s*;",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_PAS_BEGIN_END_TOKEN_RE = re.compile(
|
||||
r"\b(begin|end|case|try|asm|record)\b", re.IGNORECASE
|
||||
)
|
||||
_PAS_CALL_RE = re.compile(r"\b([A-Za-z_]\w*(?:\.[A-Za-z_]\w*)*)\s*[(;]")
|
||||
_PAS_KEYWORDS = frozenset({
|
||||
"begin", "end", "if", "then", "else", "while", "do", "for", "to",
|
||||
"downto", "repeat", "until", "case", "of", "try", "finally", "except",
|
||||
"with", "inherited", "result", "var", "const", "type", "nil", "true",
|
||||
"false", "exit", "break", "continue", "uses", "unit", "program",
|
||||
"library", "interface", "implementation", "initialization", "finalization",
|
||||
"procedure", "function", "constructor", "destructor", "class", "record",
|
||||
"object", "array", "string", "integer", "boolean", "real", "char",
|
||||
"writeln", "write", "readln", "read", "assigned", "length", "high",
|
||||
"low", "inc", "dec", "new", "dispose", "setlength", "copy", "pos",
|
||||
"trim", "format", "inttostr", "strtoint", "ord", "chr", "sizeof",
|
||||
"create", "free", "destroy",
|
||||
})
|
||||
|
||||
|
||||
def _pascal_strip_comments(text: str) -> str:
|
||||
"""Strip Pascal comments ({}, (* *), //) while preserving newlines."""
|
||||
def _sub(m: re.Match) -> str:
|
||||
tok = m.group(0)
|
||||
if tok.startswith("'"):
|
||||
return tok
|
||||
return "".join(c if c == "\n" else " " for c in tok)
|
||||
return _PAS_TOKEN_RE.sub(_sub, text)
|
||||
|
||||
|
||||
def _pascal_split_sections(text: str) -> tuple[str, int, str, int]:
|
||||
"""Split into (iface_text, iface_offset, impl_text, impl_offset).
|
||||
Files without interface/implementation sections (dpr/lpr/inc) return
|
||||
the whole text as impl with offset 0.
|
||||
"""
|
||||
iface_m = re.search(r"\binterface\b", text, re.IGNORECASE)
|
||||
impl_m = re.search(r"\bimplementation\b", text, re.IGNORECASE)
|
||||
if iface_m and impl_m:
|
||||
iface_off = iface_m.end()
|
||||
impl_off = impl_m.end()
|
||||
end_m = re.search(
|
||||
r"\b(initialization|finalization)\b", text[impl_off:], re.IGNORECASE
|
||||
)
|
||||
impl_end = impl_off + end_m.start() if end_m else len(text)
|
||||
return text[iface_off:impl_m.start()], iface_off, text[impl_off:impl_end], impl_off
|
||||
return "", 0, text, 0
|
||||
|
||||
|
||||
def _pascal_split_uses(s: str) -> list[str]:
|
||||
"""Split a uses list string, handling 'Foo in ''bar.pas''' syntax."""
|
||||
out = []
|
||||
for chunk in s.split(","):
|
||||
name = re.split(r"\s+in\s+", chunk.strip(), maxsplit=1, flags=re.IGNORECASE)[0]
|
||||
name = name.strip().strip(";")
|
||||
if name and re.match(r"[A-Za-z_][\w.]*$", name):
|
||||
out.append(name)
|
||||
return out
|
||||
|
||||
|
||||
def _pascal_split_bases(s: str) -> list[str]:
|
||||
"""Split inheritance list, handling generics like TList<T, U>."""
|
||||
out, depth, buf = [], 0, []
|
||||
for ch in s:
|
||||
if ch == "<":
|
||||
depth += 1
|
||||
buf.append(ch)
|
||||
elif ch == ">":
|
||||
depth -= 1
|
||||
buf.append(ch)
|
||||
elif ch == "," and depth == 0:
|
||||
name = re.sub(r"<.*$", "", "".join(buf).strip())
|
||||
if name:
|
||||
out.append(name)
|
||||
buf = []
|
||||
else:
|
||||
buf.append(ch)
|
||||
name = re.sub(r"<.*$", "", "".join(buf).strip())
|
||||
if name:
|
||||
out.append(name)
|
||||
return [n for n in out if re.match(r"[A-Za-z_]\w*$", n)]
|
||||
|
||||
|
||||
def _pascal_find_body(text: str, start: int) -> tuple[int, int]:
|
||||
"""Find balanced begin..end after start. Returns (body_start, body_end).
|
||||
Returns (0, 0) if no begin found.
|
||||
"""
|
||||
m = re.search(r"\bbegin\b", text[start:], re.IGNORECASE)
|
||||
if not m:
|
||||
return (0, 0)
|
||||
body_start = start + m.end()
|
||||
depth = 1
|
||||
for tok in _PAS_BEGIN_END_TOKEN_RE.finditer(text, body_start):
|
||||
kw = tok.group(1).lower()
|
||||
if kw in ("begin", "case", "try", "asm", "record"):
|
||||
depth += 1
|
||||
elif kw == "end":
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
return (body_start, tok.start())
|
||||
return (body_start, len(text))
|
||||
|
||||
|
||||
def _extract_pascal_regex(path: Path) -> dict:
|
||||
"""Regex fallback for Pascal/Delphi extraction when tree-sitter-pascal
|
||||
is unavailable. Produces the same node/edge schema as the tree-sitter pass.
|
||||
"""
|
||||
try:
|
||||
raw = path.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception as exc:
|
||||
return {"nodes": [], "edges": [], "error": str(exc)}
|
||||
|
||||
str_path = str(path)
|
||||
stem = _file_stem(path)
|
||||
nodes: list[dict] = []
|
||||
edges: list[dict] = []
|
||||
seen_ids: set[str] = set()
|
||||
seen_call_pairs: set[tuple[str, str]] = set()
|
||||
|
||||
def _add_node(nid: str, label: str, line: int) -> None:
|
||||
if nid not in seen_ids:
|
||||
seen_ids.add(nid)
|
||||
nodes.append({
|
||||
"id": nid,
|
||||
"label": label,
|
||||
"file_type": "code",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
})
|
||||
|
||||
def _add_edge(src: str, tgt: str, relation: str, line: int, context: str | None = None) -> None:
|
||||
edge: dict = {
|
||||
"source": src,
|
||||
"target": tgt,
|
||||
"relation": relation,
|
||||
"confidence": "EXTRACTED",
|
||||
"source_file": str_path,
|
||||
"source_location": f"L{line}",
|
||||
"weight": 1.0,
|
||||
}
|
||||
if context:
|
||||
edge["context"] = context
|
||||
edges.append(edge)
|
||||
|
||||
def _lineno(text: str, offset: int) -> int:
|
||||
return text.count("\n", 0, offset) + 1
|
||||
|
||||
file_nid = _make_id(str_path)
|
||||
_add_node(file_nid, path.name, 1)
|
||||
|
||||
stripped = _pascal_strip_comments(raw)
|
||||
|
||||
# Module header
|
||||
module_nid = file_nid
|
||||
mod_m = _PAS_MODULE_RE.search(stripped)
|
||||
if mod_m:
|
||||
mod_name = mod_m.group(2)
|
||||
module_nid = _make_id(stem, mod_name)
|
||||
_add_node(module_nid, mod_name, _lineno(stripped, mod_m.start()))
|
||||
_add_edge(file_nid, module_nid, "contains", _lineno(stripped, mod_m.start()))
|
||||
|
||||
iface_text, iface_off, impl_text, impl_off = _pascal_split_sections(stripped)
|
||||
|
||||
# Uses clauses
|
||||
for section_text, section_off in ((iface_text, iface_off), (impl_text, impl_off)):
|
||||
for um in _PAS_USES_RE.finditer(section_text):
|
||||
line = _lineno(stripped, section_off + um.start())
|
||||
for unit_name in _pascal_split_uses(um.group(1)):
|
||||
tgt_nid = _pascal_resolve_unit(path, unit_name)
|
||||
_add_edge(module_nid, tgt_nid, "imports", line, context="import")
|
||||
|
||||
# Type declarations (classes / interfaces) in interface section
|
||||
search_text = iface_text if iface_text else stripped
|
||||
search_off = iface_off if iface_text else 0
|
||||
pos = 0
|
||||
while pos < len(search_text):
|
||||
hm = _PAS_TYPE_HEADER_RE.search(search_text, pos)
|
||||
if not hm:
|
||||
break
|
||||
type_name = hm.group("name")
|
||||
bases_raw = hm.group("bases") or ""
|
||||
line = _lineno(stripped, search_off + hm.start())
|
||||
cls_nid = _make_id(stem, type_name)
|
||||
_add_node(cls_nid, type_name, line)
|
||||
_add_edge(module_nid, cls_nid, "contains", line)
|
||||
|
||||
for base_name in _pascal_split_bases(bases_raw):
|
||||
resolved = _pascal_resolve_class(path, base_name)
|
||||
base_nid = resolved if resolved else _make_id(base_name)
|
||||
if base_nid not in seen_ids:
|
||||
_add_node(base_nid, base_name, line)
|
||||
_add_edge(cls_nid, base_nid, "inherits", line)
|
||||
|
||||
# Find class body (up to next end;)
|
||||
end_m = _PAS_END_SEMI_RE.search(search_text, hm.end())
|
||||
body_text = search_text[hm.end():end_m.start()] if end_m else ""
|
||||
body_off = search_off + hm.end()
|
||||
|
||||
# Forward method declarations inside the class body
|
||||
for mm in _PAS_METHOD_DECL_RE.finditer(body_text):
|
||||
mname = mm.group("name")
|
||||
mline = _lineno(stripped, body_off + mm.start())
|
||||
method_nid = _make_id(cls_nid, mname)
|
||||
_add_node(method_nid, f"{mname}()", mline)
|
||||
_add_edge(cls_nid, method_nid, "method", mline)
|
||||
|
||||
pos = end_m.end() if end_m else len(search_text)
|
||||
|
||||
# Implementation headers (procedure/function/constructor/destructor)
|
||||
impl_records: list[tuple[str, int, str]] = []
|
||||
for fm in _PAS_IMPL_HEADER_RE.finditer(impl_text):
|
||||
qualified = fm.group("qual")
|
||||
line = _lineno(stripped, impl_off + fm.start())
|
||||
if "." in qualified:
|
||||
cls_part, method_part = qualified.split(".", 1)
|
||||
cls_nid = _make_id(stem, cls_part)
|
||||
container = cls_nid if cls_nid in seen_ids else module_nid
|
||||
relation = "method" if cls_nid in seen_ids else "contains"
|
||||
label = f"{method_part}()"
|
||||
else:
|
||||
container, relation = module_nid, "contains"
|
||||
label = f"{qualified}()"
|
||||
proc_nid = _make_id(stem, qualified)
|
||||
_add_node(proc_nid, label, line)
|
||||
_add_edge(container, proc_nid, relation, line)
|
||||
|
||||
body_start, body_end = _pascal_find_body(impl_text, fm.end())
|
||||
body_text = impl_text[body_start:body_end] if body_start else ""
|
||||
impl_records.append((proc_nid, line, body_text))
|
||||
|
||||
# Intra-file call edges
|
||||
all_procs: dict[str, str] = {
|
||||
n["label"].removesuffix("()").lower(): n["id"]
|
||||
for n in nodes
|
||||
if n["id"] != file_nid and n["label"].endswith("()")
|
||||
}
|
||||
for caller_nid, caller_line, body_text in impl_records:
|
||||
for cm in _PAS_CALL_RE.finditer(body_text):
|
||||
callee_name = cm.group(1).split(".")[-1].lower()
|
||||
if callee_name in _PAS_KEYWORDS:
|
||||
continue
|
||||
callee_nid = all_procs.get(callee_name)
|
||||
if not callee_nid or callee_nid == caller_nid:
|
||||
continue
|
||||
pair = (caller_nid, callee_nid)
|
||||
if pair in seen_call_pairs:
|
||||
continue
|
||||
seen_call_pairs.add(pair)
|
||||
call_line = caller_line + body_text.count("\n", 0, cm.start())
|
||||
_add_edge(caller_nid, callee_nid, "calls", call_line, context="call")
|
||||
|
||||
return {"nodes": nodes, "edges": edges, "input_tokens": 0, "output_tokens": 0}
|
||||
|
||||
|
||||
def extract_pascal(path: Path) -> dict:
|
||||
"""Extract units, classes, procedures, uses-imports, and calls from Pascal/Delphi files.
|
||||
|
||||
@@ -4637,17 +4928,15 @@ def extract_pascal(path: Path) -> dict:
|
||||
- class/module --contains--> procedure/function implementation
|
||||
- procedure --calls--> other procedure (within the same file)
|
||||
|
||||
Requires tree-sitter-pascal: pip install tree-sitter-pascal
|
||||
(https://github.com/Isopod/tree-sitter-pascal)
|
||||
Uses tree-sitter-pascal when available; falls back to a regex-based extractor
|
||||
(_extract_pascal_regex) when it isn't installed or fails to parse, so Pascal
|
||||
extraction works out of the box without an extra pip install.
|
||||
"""
|
||||
try:
|
||||
import tree_sitter_pascal as tspascal
|
||||
from tree_sitter import Language, Parser
|
||||
except ImportError:
|
||||
return {
|
||||
"nodes": [], "edges": [],
|
||||
"error": "tree_sitter_pascal not installed. Run: pip install tree-sitter-pascal",
|
||||
}
|
||||
return _extract_pascal_regex(path)
|
||||
|
||||
try:
|
||||
language = Language(tspascal.language())
|
||||
@@ -4655,8 +4944,8 @@ def extract_pascal(path: Path) -> dict:
|
||||
source = path.read_bytes()
|
||||
tree = parser.parse(source)
|
||||
root = tree.root_node
|
||||
except Exception as e:
|
||||
return {"nodes": [], "edges": [], "error": str(e)}
|
||||
except Exception:
|
||||
return _extract_pascal_regex(path)
|
||||
|
||||
stem = _file_stem(path)
|
||||
str_path = str(path)
|
||||
|
||||
@@ -1,25 +1,10 @@
|
||||
"""Tests for the Pascal/Delphi extractor."""
|
||||
from __future__ import annotations
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
|
||||
FIXTURES = Path(__file__).parent / "fixtures"
|
||||
|
||||
|
||||
def _try_import():
|
||||
try:
|
||||
import tree_sitter_pascal # noqa: F401
|
||||
return True
|
||||
except ImportError:
|
||||
return False
|
||||
|
||||
|
||||
pascal_required = pytest.mark.skipif(
|
||||
not _try_import(),
|
||||
reason="tree_sitter_pascal not installed",
|
||||
)
|
||||
|
||||
|
||||
def _labels(r):
|
||||
return [n["label"] for n in r["nodes"]]
|
||||
|
||||
@@ -32,21 +17,18 @@ def _edges_with_relation(r, *relations):
|
||||
return [e for e in r["edges"] if e["relation"] in relations]
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_no_error():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
assert "error" not in r
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_finds_unit():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
assert any("SampleUnit" in l for l in _labels(r))
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_finds_classes():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
@@ -55,14 +37,12 @@ def test_pascal_finds_classes():
|
||||
assert any("TDataProcessor" in l for l in labels)
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_finds_interface():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
assert any("IProcessor" in l for l in _labels(r))
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_finds_methods():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
@@ -73,14 +53,12 @@ def test_pascal_finds_methods():
|
||||
assert any("Reset" in l for l in labels)
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_finds_imports():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
assert "imports" in _relations(r)
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_import_edges_have_import_context():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
@@ -89,14 +67,12 @@ def test_pascal_import_edges_have_import_context():
|
||||
assert all(e.get("context") == "import" for e in import_edges)
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_finds_inherits():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
assert "inherits" in _relations(r)
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_inherits_from_base():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
@@ -109,14 +85,12 @@ def test_pascal_inherits_from_base():
|
||||
assert found, "TDataProcessor should have at least one inherits edge"
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_finds_calls():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
assert "calls" in _relations(r)
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_call_edges_have_call_context():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
@@ -125,7 +99,6 @@ def test_pascal_call_edges_have_call_context():
|
||||
assert all(e.get("context") == "call" for e in call_edges)
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_all_edges_extracted():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
@@ -135,7 +108,6 @@ def test_pascal_all_edges_extracted():
|
||||
assert e["confidence"] == "EXTRACTED", f"Expected EXTRACTED: {e}"
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_no_dangling_edges():
|
||||
from graphify.extract import extract_pascal
|
||||
r = extract_pascal(FIXTURES / "sample.pas")
|
||||
@@ -148,7 +120,6 @@ def test_pascal_no_dangling_edges():
|
||||
assert e["target"] in node_ids, f"Dangling target: {e}"
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_dispatch_registered():
|
||||
from graphify.extract import _DISPATCH
|
||||
assert ".pas" in _DISPATCH
|
||||
@@ -161,7 +132,6 @@ def test_pascal_dispatch_registered():
|
||||
assert ".lpk" in _DISPATCH
|
||||
|
||||
|
||||
@pascal_required
|
||||
def test_pascal_detect_extensions_registered():
|
||||
from graphify.detect import CODE_EXTENSIONS
|
||||
assert ".pas" in CODE_EXTENSIONS
|
||||
|
||||
Reference in New Issue
Block a user