From 8fe624ef97473f7299ae9417cb7838fa1894e035 Mon Sep 17 00:00:00 2001 From: Anthony Date: Tue, 22 Sep 2026 23:16:25 +1000 Subject: [PATCH 1/5] feat: add a GDScript extractor for Godot 4 scripts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Nodes: the script (its class_name, or its autoload name, or the file name), func, signal, const, enum and inner class. Edges: extends -> inherits to the parent script; preload()/load() of a .gd -> imports; signal connect/emit -> uses (plus the connect callback as a reference); calls resolved through the same file, the extends chain, the project's [autoload] table, preload const aliases and the class_name index - all EXTRACTED, because each names its target in source; a call none of those resolves is counted, never guessed, and never handed to the shared name-matching pass. A .md §N.N in a comment becomes a references edge to a section node of that page; a bare §N.N inherits the last page the file named (INFERRED). The project index is built once per project.godot root and cached per process. Registered in the dispatch, the language-family table, the extractor registry and detect's code set; grammar tree-sitter-gdscript 6.x. Version carries the +gms1 local label so the fork build is observable from --version. --- graphify/detect.py | 2 +- graphify/extract.py | 3 + graphify/extractors/__init__.py | 2 + graphify/extractors/gdscript.py | 589 ++++++++++++++++++++++++++++++++ pyproject.toml | 3 +- uv.lock | 15 +- 6 files changed, 611 insertions(+), 3 deletions(-) create mode 100644 graphify/extractors/gdscript.py diff --git a/graphify/detect.py b/graphify/detect.py index eec8aaf1d2..8262921d22 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -41,7 +41,7 @@ class FileType(str, Enum): _MTIME_COARSE_S = 2.0 _MTIME_SUBSECOND_S = 0.05 -CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.mts', '.cts', '.js', '.jsx', '.mjs', '.cjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.cu', '.cuh', '.metal', '.rb', '.rake', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.psm1', '.psd1', '.ex', '.exs', '.m', '.mm', '.ml', '.mli', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.svh', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.tf', '.tfvars', '.hcl', '.dm', '.dme', '.dmi', '.dmm', '.dmf', '.sln', '.slnx', '.csproj', '.fsproj', '.vbproj', '.xaml', '.razor', '.cshtml', '.cls', '.trigger', '.lisp', '.cl', '.lsp', '.asd', '.robot', '.resource'} +CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.mts', '.cts', '.js', '.jsx', '.mjs', '.cjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.cu', '.cuh', '.metal', '.rb', '.rake', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.gd', '.ps1', '.psm1', '.psd1', '.ex', '.exs', '.m', '.mm', '.ml', '.mli', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.svh', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.tf', '.tfvars', '.hcl', '.dm', '.dme', '.dmi', '.dmm', '.dmf', '.sln', '.slnx', '.csproj', '.fsproj', '.vbproj', '.xaml', '.razor', '.cshtml', '.cls', '.trigger', '.lisp', '.cl', '.lsp', '.asd', '.robot', '.resource'} DOC_EXTENSIONS = {'.md', '.mdx', '.qmd', '.skill', '.txt', '.rst', '.html', '.yaml', '.yml'} PAPER_EXTENSIONS = {'.pdf'} IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'} diff --git a/graphify/extract.py b/graphify/extract.py index da66f41317..4ae9283e57 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -46,6 +46,7 @@ from graphify.extractors.dm import extract_dm, extract_dmf, extract_dmi, extract_dmm # noqa: F401 from graphify.extractors.elixir import extract_elixir # noqa: F401 from graphify.extractors.fortran import _cpp_preprocess, extract_fortran # noqa: F401 +from graphify.extractors.gdscript import extract_gdscript # noqa: F401 from graphify.extractors.go import _GO_PREDECLARED_FUNCS, extract_go # noqa: F401 from graphify.extractors.json_config import extract_json # noqa: F401 from graphify.extractors.commonlisp import extract_commonlisp # noqa: F401 @@ -2769,6 +2770,7 @@ def _lang_is_case_insensitive(source_file: object) -> bool: ".cs": "dotnet", ".razor": "dotnet", ".cshtml": "dotnet", ".xaml": "dotnet", ".lua": "lua", ".luau": "lua", ".zig": "zig", + ".gd": "gdscript", ".ex": "elixir", ".exs": "elixir", ".jl": "julia", ".dart": "dart", @@ -6302,6 +6304,7 @@ def add_existing_edge(edge: dict) -> None: ".luau": extract_lua, ".toc": extract_lua, ".zig": extract_zig, + ".gd": extract_gdscript, ".ps1": extract_powershell, ".psm1": extract_powershell, ".psd1": extract_powershell_manifest, diff --git a/graphify/extractors/__init__.py b/graphify/extractors/__init__.py index 68ff3340c3..0d27605cd0 100644 --- a/graphify/extractors/__init__.py +++ b/graphify/extractors/__init__.py @@ -18,6 +18,7 @@ from graphify.extractors.dm import extract_dm, extract_dmf, extract_dmi, extract_dmm from graphify.extractors.elixir import extract_elixir from graphify.extractors.fortran import extract_fortran +from graphify.extractors.gdscript import extract_gdscript from graphify.extractors.go import extract_go from graphify.extractors.json_config import extract_json from graphify.extractors.julia import extract_julia @@ -47,6 +48,7 @@ "dmm": extract_dmm, "elixir": extract_elixir, "fortran": extract_fortran, + "gdscript": extract_gdscript, "go": extract_go, "json": extract_json, "julia": extract_julia, diff --git a/graphify/extractors/gdscript.py b/graphify/extractors/gdscript.py new file mode 100644 index 0000000000..f7422154e2 --- /dev/null +++ b/graphify/extractors/gdscript.py @@ -0,0 +1,589 @@ +"""GDScript extractor (tree-sitter-gdscript). + +Godot 4 scripts. One file is one class: the node graph is the file node plus its +``func`` / ``signal`` / ``const`` / ``enum`` / inner ``class`` members, and edges +follow how a Godot project is actually wired together: + +* ``extends`` -> ``inherits`` to the parent SCRIPT (a ``class_name`` or a + ``res://`` path). An engine class (``Node``, ``RefCounted``) has no script and + yields no edge. +* ``preload(...)`` / ``load(...)`` of a ``.gd`` -> ``imports`` to that script's + file node. This is how a script with no ``class_name`` is reached at all. +* ``sig.connect(cb)`` / ``sig.emit(...)`` / ``emit_signal("sig", ...)`` -> + ``uses`` from the enclosing function to the signal node; a connect also + ``references`` its callback when that is a function of this file. +* calls: same-file functions and functions of an ancestor script resolve + EXTRACTED; ``Autoload.method()`` resolves through the project's + ``[autoload]`` table to that script's method (the receiver names the script + in source, so it is exact); ``Alias.method()`` through a ``preload`` const + alias; ``ClassName.method()`` / ``ClassName.new()`` through the project's + ``class_name`` index. A bare call resolved by none of those is an engine + builtin or a dynamic dispatch and is NOT handed to the shared name-matching + resolver — counted in ``unresolved_calls`` instead. +* ``.md §N.N`` in a comment -> ``references`` from the enclosing function + (or the file) to a section node of that documentation page. A bare ``§N.N`` + inherits the last page a comment in the same file named (INFERRED). + +The project index (autoloads, ``class_name`` -> script, ``uid://`` -> script, +per-script ``func`` names) is built once per ``project.godot`` root and cached +per process; a file outside any Godot project still extracts, with only the +same-file resolutions. +""" +from __future__ import annotations + +import re +from pathlib import Path +from typing import Any + +from graphify.extractors.base import _file_stem, _make_id, _read_text + +_ENGINE_BUILTIN_RE = re.compile(r"^[A-Z][A-Za-z0-9]*$") +_CITATION_RE = re.compile(r"([A-Za-z0-9_./-]+\.md)`?\s*§\s*(\d+(?:\.\d+)*[a-z]?)") +_BARE_CITATION_RE = re.compile(r"§\s*(\d+(?:\.\d+)*[a-z]?)") +_CLASS_NAME_RE = re.compile(r"^class_name\s+([A-Za-z_][A-Za-z0-9_]*)", re.M) +_EXTENDS_RE = re.compile(r'^extends\s+(?:"([^"]+)"|([A-Za-z_][A-Za-z0-9_]*))', re.M) +_FUNC_RE = re.compile(r"^(?:static\s+)?func\s+([A-Za-z_][A-Za-z0-9_]*)\s*\(", re.M) +_SIGNAL_RE = re.compile(r"^signal\s+([A-Za-z_][A-Za-z0-9_]*)", re.M) +_AUTOLOAD_RE = re.compile(r'^([A-Za-z_][A-Za-z0-9_]*)="\*?((?:res|uid)://[^"]+)"', re.M) +_UID_RE = re.compile(r"^(uid://[a-z0-9]+)\s*$", re.M) + +_PROJECT_CACHE: dict[Path, "_GodotProject | None"] = {} +_FILE_INDEX_CACHE: dict[Path, tuple[frozenset[str], frozenset[str], str | None]] = {} + + +class _GodotProject: + """What the extractor knows about the Godot project a script belongs to.""" + + def __init__(self, root: Path) -> None: + self.root = root + self.autoloads: dict[str, Path] = {} + self.class_names: dict[str, Path] = {} + uids: dict[str, Path] = {} + try: + for uid_file in root.rglob("*.gd.uid"): + if ".godot" in uid_file.parts: + continue + try: + m = _UID_RE.search(uid_file.read_text(encoding="utf-8", errors="replace")) + except OSError: + continue + if m: + uids[m.group(1)] = uid_file.with_suffix("") + for gd in root.rglob("*.gd"): + if ".godot" in gd.parts: + continue + try: + head = gd.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + m = _CLASS_NAME_RE.search(head) + if m and m.group(1) not in self.class_names: + self.class_names[m.group(1)] = gd + text = (root / "project.godot").read_text(encoding="utf-8", errors="replace") + except OSError: + return + section = text.split("[autoload]", 1) + if len(section) == 2: + body = section[1].split("\n[", 1)[0] + for m in _AUTOLOAD_RE.finditer(body): + target = self.resolve(m.group(2), uids) + if target is not None: + self.autoloads[m.group(1)] = target + self._uids = uids + + def resolve(self, ref: str, uids: dict[str, Path] | None = None) -> Path | None: + """A ``res://`` or ``uid://`` reference as an absolute path, or None.""" + if ref.startswith("res://"): + return self.root / ref[len("res://"):] + if ref.startswith("uid://"): + return (uids if uids is not None else self._uids).get(ref) + return None + + +def _project_for(path: Path) -> _GodotProject | None: + """The nearest enclosing ``project.godot``, indexed once per process.""" + try: + start = path.resolve().parent + except OSError: + start = path.parent + for candidate in (start, *start.parents): + if candidate in _PROJECT_CACHE: + return _PROJECT_CACHE[candidate] + if (candidate / "project.godot").is_file(): + project = _GodotProject(candidate) + _PROJECT_CACHE[candidate] = project + return project + _PROJECT_CACHE[start] = None + return None + + +def _file_index(script: Path) -> tuple[frozenset[str], frozenset[str], str | None]: + """(func names, signal names, extends spec) of a script, by regex, cached.""" + key = script.resolve() if script.exists() else script + hit = _FILE_INDEX_CACHE.get(key) + if hit is not None: + return hit + try: + text = script.read_text(encoding="utf-8", errors="replace") + except OSError: + text = "" + m = _EXTENDS_RE.search(text) + ext = (m.group(1) or m.group(2)) if m else None + result = (frozenset(_FUNC_RE.findall(text)), frozenset(_SIGNAL_RE.findall(text)), ext) + _FILE_INDEX_CACHE[key] = result + return result + + +def _script_of(spec: str, script: Path, project: _GodotProject | None) -> Path | None: + """The script an ``extends`` / reference spec names, or None (engine class).""" + if spec.startswith("res://") or spec.startswith("uid://"): + return project.resolve(spec) if project else None + if spec.endswith(".gd"): + return (script.parent / spec).resolve() + if project and spec in project.class_names: + return project.class_names[spec] + return None + + +def _ancestors(script: Path, project: _GodotProject | None, limit: int = 16) -> list[Path]: + """Ancestor scripts of ``script``, nearest first, following ``extends``.""" + out: list[Path] = [] + seen = {script} + current = script + while len(out) < limit: + _funcs, _signals, spec = _file_index(current) + if not spec: + break + parent = _script_of(spec, current, project) + if parent is None or parent in seen: + break + seen.add(parent) + out.append(parent) + current = parent + return out + + +def extract_gdscript(path: Path) -> dict: + """Extract classes, functions, signals, constants, enums, imports, signal + wiring, calls and documentation citations from a .gd file.""" + try: + import tree_sitter_gdscript as tsgd + from tree_sitter import Language, Parser + except ImportError: + return {"nodes": [], "edges": [], "error": "tree_sitter_gdscript not installed"} + + try: + language = Language(tsgd.language()) + parser = Parser(language) + source = path.read_bytes() + tree = parser.parse(source) + root = tree.root_node + except Exception as e: + return {"nodes": [], "edges": [], "error": str(e)} + + project = _project_for(path) + stem = _file_stem(path) + str_path = str(path) + nodes: list[dict] = [] + edges: list[dict] = [] + seen_ids: set[str] = set() + seen_edges: set[tuple[str, str, str]] = set() + unresolved_calls: list[dict] = [] + + def add_node(nid: str, label: str, line: int, kind: str, **extra: Any) -> None: + if nid in seen_ids: + return + seen_ids.add(nid) + node = {"id": nid, "label": label, "file_type": "code", "type": kind, + "source_file": str_path, "source_location": f"L{line}"} + node.update(extra) + nodes.append(node) + + def add_edge(src: str, tgt: str, relation: str, line: int, + confidence: str = "EXTRACTED", context: str | None = None) -> None: + key = (src, tgt, relation) + if key in seen_edges or src == tgt: + return + seen_edges.add(key) + edge = {"source": src, "target": tgt, "relation": relation, + "confidence": confidence, "source_file": str_path, + "source_location": f"L{line}", "weight": 1.0} + if context: + edge["context"] = context + edges.append(edge) + + def other_file_nid(script: Path) -> str: + return _make_id(str(script)) + + def other_symbol_nid(script: Path, name: str) -> str: + return _make_id(_file_stem(script), name) + + def line_of(node) -> int: + return node.start_point[0] + 1 + + def string_value(node) -> str | None: + if node.type != "string": + return None + return _read_text(node, source).strip("\"'") + + file_nid = _make_id(str(path)) + class_label = path.name + if project is not None: + # An autoload with no class_name is known to every other script by its + # autoload name, so that is the label the graph carries for it. + try: + resolved = path.resolve() + except OSError: + resolved = path + for auto_name, auto_path in project.autoloads.items(): + if auto_path == resolved or auto_path == path: + class_label = auto_name + break + add_node(file_nid, class_label, 1, "class") + + # --- pass 1: declarations -------------------------------------------------- + func_nids: dict[str, str] = {} # top-level func name -> nid + signal_nids: dict[str, str] = {} # signal name -> nid + preload_alias: dict[str, Path] = {} # const NAME := preload("...gd") -> script + function_bodies: list[tuple[str, Any, Any]] = [] # (nid, body node, def node) + top_funcs: frozenset[str] = frozenset() + + def declare_member(node, owner_nid: str, owner_stem: str, inner: bool) -> None: + t = node.type + if t == "function_definition": + name_node = node.child_by_field_name("name") + if not name_node: + return + name = _read_text(name_node, source) + nid = _make_id(owner_stem, name) + add_node(nid, f".{name}()" if inner else f"{name}()", line_of(node), "function") + add_edge(owner_nid, nid, "method" if inner else "contains", line_of(node)) + if not inner: + func_nids[name] = nid + body = node.child_by_field_name("body") + if body: + function_bodies.append((nid, body, node)) + return + if t == "signal_statement": + name_node = node.child_by_field_name("name") + if not name_node: + return + name = _read_text(name_node, source) + nid = _make_id(owner_stem, name) + add_node(nid, f"signal {name}", line_of(node), "signal") + add_edge(owner_nid, nid, "contains", line_of(node)) + if not inner: + signal_nids[name] = nid + return + if t == "const_statement": + name_node = node.child_by_field_name("name") + if not name_node: + return + name = _read_text(name_node, source) + # kind-qualified: make_id case-folds, so a `const Helper` and a + # `func helper()` would otherwise claim one id + nid = _make_id(owner_stem, "const", name) + add_node(nid, name, line_of(node), "constant") + add_edge(owner_nid, nid, "contains", line_of(node)) + for child in node.children: + if child.type == "call": + target = _load_target(child) + if target is not None: + preload_alias[name] = target + return + if t == "enum_definition": + name_node = node.child_by_field_name("name") + if not name_node: + return + name = _read_text(name_node, source) + nid = _make_id(owner_stem, "enum", name) + add_node(nid, f"enum {name}", line_of(node), "enum") + add_edge(owner_nid, nid, "contains", line_of(node)) + return + if t == "class_definition": + name_node = node.child_by_field_name("name") + if not name_node: + return + name = _read_text(name_node, source) + nid = _make_id(owner_stem, name) + add_node(nid, name, line_of(node), "class") + add_edge(owner_nid, nid, "contains", line_of(node)) + for child in node.children: + if child.type == "extends_statement": + emit_extends(child, nid) + elif child.type == "class_body": + for member in child.children: + declare_member(member, nid, f"{owner_stem}/{name}", True) + return + + def _load_target(call_node) -> Path | None: + """The script a ``preload("…")`` / ``load("…")`` call names, or None.""" + fn = None + args = None + for c in call_node.children: + if c.type == "identifier" and fn is None: + fn = _read_text(c, source) + elif c.type == "arguments": + args = c + if fn not in ("preload", "load") or args is None: + return None + for a in args.children: + value = string_value(a) + if value is not None: + return _script_of(value, path, project) if value.endswith(".gd") else None + return None + + def emit_extends(node, owner_nid: str) -> None: + spec = None + for c in node.children: + if c.type == "type": + spec = _read_text(c, source) + elif c.type == "string": + spec = string_value(c) + if not spec: + return + target = _script_of(spec, path, project) + if target is not None: + add_edge(owner_nid, other_file_nid(target), "inherits", line_of(node)) + + for child in root.children: + if child.type == "extends_statement": + emit_extends(child, file_nid) + elif child.type == "class_name_statement": + name_node = child.child_by_field_name("name") + if name_node: + class_label = _read_text(name_node, source) + for n in nodes: + if n["id"] == file_nid: + n["label"] = class_label + else: + declare_member(child, file_nid, stem, False) + top_funcs = frozenset(func_nids) + + # imports: every preload/load of a .gd anywhere in the file (const aliases, + # locals, inline arguments) -> the target script's file node. + def walk_loads(node) -> None: + if node.type == "call": + target = _load_target(node) + if target is not None: + add_edge(file_nid, other_file_nid(target), "imports", line_of(node)) + for c in node.children: + walk_loads(c) + walk_loads(root) + + # --- pass 2: calls and signal wiring inside function bodies ------------------- + ancestors = _ancestors(path, project) + + def resolve_bare(name: str) -> str | None: + if name in func_nids: + return func_nids[name] + for anc in ancestors: + funcs, _signals, _ext = _file_index(anc) + if name in funcs: + return other_symbol_nid(anc, name) + return None + + def resolve_receiver(receiver: str) -> Path | None: + if receiver in preload_alias: + return preload_alias[receiver] + if project is not None: + if receiver in project.autoloads: + return project.autoloads[receiver] + if receiver in project.class_names: + return project.class_names[receiver] + return None + + def signal_nid_for(script: Path | None, name: str) -> str | None: + if script is None: + if name in signal_nids: + return signal_nids[name] + for anc in ancestors: + _funcs, signals, _ext = _file_index(anc) + if name in signals: + return other_symbol_nid(anc, name) + return None + _funcs, signals, _ext = _file_index(script) + return other_symbol_nid(script, name) if name in signals else None + + def callback_of(args_node) -> str | None: + for a in args_node.children: + if a.type == "identifier": + return _read_text(a, source) + if a.type == "attribute": + parts = [c for c in a.children if c.type == "identifier"] + if len(parts) == 2 and _read_text(parts[0], source) == "self": + return _read_text(parts[1], source) + return None + + def handle_attribute(node, caller_nid: str) -> None: + # attribute := ('.' identifier)* '.' attribute_call + parts = list(node.children) + call = next((c for c in parts if c.type == "attribute_call"), None) + if call is None: + return + idx = parts.index(call) + chain = [c for c in parts[:idx] if c.type in ("identifier", "attribute_call", "get_node", "call", "self")] + method_node = next((c for c in call.children if c.type == "identifier"), None) + args = next((c for c in call.children if c.type == "arguments"), None) + if method_node is None: + return + method = _read_text(method_node, source) + line = line_of(node) + head = _read_text(chain[0], source) if chain and chain[0].type == "identifier" else None + names = [_read_text(c, source) for c in chain if c.type == "identifier"] + + # signal wiring: .connect(cb) / .emit(...) + if method in ("connect", "emit") and names: + sig_name = names[-1] + script = None + if len(names) >= 2 and names[0] != "self": + script = resolve_receiver(names[0]) + if script is None: + # a signal of some other object (a child node, a local) — + # nothing in this project index names it + unresolved_calls.append({"caller_nid": caller_nid, "callee": f"{'.'.join(names)}.{method}", "line": line}) + return + sig = signal_nid_for(script, sig_name) + if sig is not None: + add_edge(caller_nid, sig, "uses", line, context=method) + if method == "connect" and args is not None: + cb = callback_of(args) + cb_nid = resolve_bare(cb) if cb else None + if cb_nid is not None: + add_edge(caller_nid, cb_nid, "references", line, context="connect") + return + + if head in ("self", "super") and len(names) == 1: + target = resolve_bare(method) + if target is not None: + add_edge(caller_nid, target, "calls", line, context="call") + else: + unresolved_calls.append({"caller_nid": caller_nid, "callee": method, "line": line}) + return + + if head is not None and len(names) == 1: + script = resolve_receiver(head) + if script is not None: + if method == "new": + add_edge(caller_nid, other_file_nid(script), "references", line, context="instantiates") + return + funcs, _signals, _ext = _file_index(script) + if method in funcs: + add_edge(caller_nid, other_symbol_nid(script, method), "calls", line, context="call") + return + for anc in _ancestors(script, project): + afuncs, _s, _e = _file_index(anc) + if method in afuncs: + add_edge(caller_nid, other_symbol_nid(anc, method), "calls", line, context="call") + return + unresolved_calls.append({"caller_nid": caller_nid, "callee": f"{head}.{method}", "line": line}) + return + unresolved_calls.append({"caller_nid": caller_nid, "callee": f"{'.'.join(names)}.{method}" if names else method, "line": line}) + + def handle_call(node, caller_nid: str) -> None: + fn = None + args = None + for c in node.children: + if c.type == "identifier" and fn is None: + fn = _read_text(c, source) + elif c.type == "arguments": + args = c + if fn is None: + return + line = line_of(node) + if fn in ("preload", "load"): + return + if fn == "emit_signal" and args is not None: + for a in args.children: + name = string_value(a) + if name is not None: + sig = signal_nid_for(None, name) + if sig is not None: + add_edge(caller_nid, sig, "uses", line, context="emit") + return + return + if fn in ("connect", "is_connected", "disconnect") and args is not None: + # Godot 3 style connect("sig", target, "method") is not Godot 4; skip. + return + target = resolve_bare(fn) + if target is not None: + add_edge(caller_nid, target, "calls", line, context="call") + elif not _ENGINE_BUILTIN_RE.match(fn): + unresolved_calls.append({"caller_nid": caller_nid, "callee": fn, "line": line}) + + def walk_calls(node, caller_nid: str) -> None: + t = node.type + if t == "function_definition": + return + if t == "attribute": + handle_attribute(node, caller_nid) + for c in node.children: + if c.type == "attribute_call": + for a in c.children: + if a.type == "arguments": + walk_calls(a, caller_nid) + elif c.type in ("call", "attribute"): + walk_calls(c, caller_nid) + return + if t == "call": + handle_call(node, caller_nid) + for c in node.children: + if c.type == "arguments": + walk_calls(c, caller_nid) + return + for c in node.children: + walk_calls(c, caller_nid) + + for nid, body, _def in function_bodies: + walk_calls(body, nid) + + # --- pass 3: documentation citations in comments ----------------------------- + spans = [(d.start_byte, d.end_byte, nid) for nid, _b, d in function_bodies] + + def enclosing(byte: int) -> str: + for start, end, nid in spans: + if start <= byte < end: + return nid + return file_nid + + def section_node(page: str, section: str, line: int) -> str: + page_rel = page.lstrip("./") + if project is not None: + for candidate in (project.root / page_rel, project.root / "docs" / page_rel): + if candidate.is_file(): + page_rel = candidate.relative_to(project.root).as_posix() + break + nid = _make_id("doc", page_rel, "s" + section) + if nid not in seen_ids: + seen_ids.add(nid) + nodes.append({"id": nid, "label": f"{Path(page_rel).name} §{section}", + "file_type": "doc", "type": "section", + "source_file": page_rel, "source_location": f"§{section}"}) + return nid + + last_page: str | None = None + + def walk_comments(node) -> None: + nonlocal last_page + if node.type == "comment": + text = _read_text(node, source) + line = line_of(node) + src = enclosing(node.start_byte) + explicit: list[tuple[int, int]] = [] + for m in _CITATION_RE.finditer(text): + page = m.group(1).split("/")[-1] + last_page = page + explicit.append((m.start(), m.end())) + add_edge(src, section_node(page, m.group(2), line), "references", line, context="citation") + for m in _BARE_CITATION_RE.finditer(text): + if last_page is None or any(a <= m.start() < b for a, b in explicit): + continue + add_edge(src, section_node(last_page, m.group(1), line), "references", line, + confidence="INFERRED", context="citation") + return + for c in node.children: + walk_comments(c) + walk_comments(root) + + return {"nodes": nodes, "edges": edges, "raw_calls": [], + "unresolved_calls": unresolved_calls} diff --git a/pyproject.toml b/pyproject.toml index 5e39d8c54f..baa87c616a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" -version = "0.9.65" +version = "0.9.65+gms1" description = "AI coding assistant skill (Claude Code, CodeBuddy, Codex, OpenCode, Kilo Code, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Devin CLI, Google Antigravity) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph" readme = "README.md" license = "Apache-2.0" @@ -40,6 +40,7 @@ dependencies = [ "tree-sitter-julia>=0.23,<0.25", "tree-sitter-verilog>=1.0,<2.0", "tree-sitter-fortran>=0.6,<0.8", + "tree-sitter-gdscript>=6.1,<7", "tree-sitter-bash>=0.23,<0.27", "tree-sitter-json>=0.23,<0.26", ] diff --git a/uv.lock b/uv.lock index fd04b1bea5..6cc3a6f3e3 100644 --- a/uv.lock +++ b/uv.lock @@ -1096,7 +1096,7 @@ wheels = [ [[package]] name = "graphifyy" -version = "0.9.65" +version = "0.9.65+gms1" source = { editable = "." } dependencies = [ { name = "networkx", version = "3.4.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, @@ -1112,6 +1112,7 @@ dependencies = [ { name = "tree-sitter-cpp" }, { name = "tree-sitter-elixir" }, { name = "tree-sitter-fortran" }, + { name = "tree-sitter-gdscript" }, { name = "tree-sitter-go" }, { name = "tree-sitter-groovy" }, { name = "tree-sitter-java" }, @@ -1342,6 +1343,7 @@ requires-dist = [ { name = "tree-sitter-dm", marker = "extra == 'dm'" }, { name = "tree-sitter-elixir", specifier = ">=0.3,<0.5" }, { name = "tree-sitter-fortran", specifier = ">=0.6,<0.8" }, + { name = "tree-sitter-gdscript", specifier = ">=6.1,<7" }, { name = "tree-sitter-go", specifier = ">=0.23,<0.26" }, { name = "tree-sitter-groovy", specifier = ">=0.1,<0.3" }, { name = "tree-sitter-hcl", marker = "extra == 'all'" }, @@ -4714,6 +4716,17 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/6c/e3/bb2c89f65497b3c8d43fb71fd6f47fef098dc3e3b0bf16083f6f9e4fc92d/tree_sitter_fortran-0.6.0-cp39-abi3-win_arm64.whl", hash = "sha256:45b0e226325e626101949d6aafcf0422fc210c3cf3ae9b9a2281b41f47d9cc20", size = 379749, upload-time = "2026-04-24T14:15:11.079Z" }, ] +[[package]] +name = "tree-sitter-gdscript" +version = "6.1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d5/61/d681398e5551b27b3ae32026665392f41f8ca2764847acecc727312e462e/tree_sitter_gdscript-6.1.0.tar.gz", hash = "sha256:33c4406a47ca5e3c436849d2a9b878a161b6f4fda305cf5620f60abebf5acf9e", size = 107803, upload-time = "2026-09-04T09:04:59.03Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/f4/2c/eab1b3890a99c85f97638b55d9d5745746fb3afc4b392e2363c7f8ed7d03/tree_sitter_gdscript-6.1.0-cp39-abi3-macosx_11_0_arm64.whl", hash = "sha256:c588efba0ae4e1aa385caadcc19b3854b1270a855337ba5e4df1da4bbf5c38d1", size = 52192, upload-time = "2026-09-04T09:04:55.645Z" }, + { url = "https://files.pythonhosted.org/packages/e6/96/f724d35e7dcb443f6c8b3bb73e8a211d96abed8d829078511cfa9af8b2b6/tree_sitter_gdscript-6.1.0-cp39-abi3-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:85558120e50ee87b6847bb81273f7a5c8fadcbc7d4ee531701f8ed403d8de33c", size = 76616, upload-time = "2026-09-04T09:04:56.771Z" }, + { url = "https://files.pythonhosted.org/packages/34/10/58b1ad211163473162648a96be4ed3c25d98dfc1b2f846802b99e540424d/tree_sitter_gdscript-6.1.0-cp39-abi3-win_amd64.whl", hash = "sha256:b2b87ea4cf21b3c33a00406840936f073b0db344ac46215f88edb8f72c72c761", size = 51341, upload-time = "2026-09-04T09:04:57.985Z" }, +] + [[package]] name = "tree-sitter-go" version = "0.25.0" From 846ad66b3fb7602c0ce75d3eddcb646f52046660 Mon Sep 17 00:00:00 2001 From: Anthony Date: Wed, 23 Sep 2026 00:18:41 +1000 Subject: [PATCH 2/5] test: cover the GDScript extractor with a fixture project A fixture script exercising every declaration shape, extracted inside a tmp_path Godot project (autoloads by res:// and uid://, a preload-only law script, a helper); inheritance and ancestor-call resolution across three scripts; the autoload label; a script outside any project; and the dispatch assertion that no other grammar reaches a .gd file. --- tests/fixtures/sample.gd | 58 +++++++++ tests/test_extractors_registry.py | 9 ++ tests/test_languages.py | 203 ++++++++++++++++++++++++++++++ 3 files changed, 270 insertions(+) create mode 100644 tests/fixtures/sample.gd diff --git a/tests/fixtures/sample.gd b/tests/fixtures/sample.gd new file mode 100644 index 0000000000..be9bb105c6 --- /dev/null +++ b/tests/fixtures/sample.gd @@ -0,0 +1,58 @@ +## A player character for a small platformer, exercising every shape the GDScript +## extractor reads. Movement rules are described in movement.md §2.1; the jump arc in §2.3. +extends CharacterBody2D +class_name PlayerController + +## No class_name on the movement helper: preload it (conventions.md §1.2). +const Movement := preload("res://actors/movement.gd") +const Inventory = load("res://actors/inventory.gd") +const DustPuff := preload("res://effects/dust_puff.tscn") +const MAX_JUMPS: int = 2 + +signal jumped(height: float) +signal landed + +enum State { IDLE, WALKING, JUMPING } + +class Stats extends RefCounted: + func speed_for(state: int) -> float: + return 120.0 + + +static func clamp_speed(value: float) -> float: + return minf(value, 300.0) + + +func _ready() -> void: + jumped.connect(_on_jumped) + landed.connect(self._on_landed) + GameState.paused.connect(_on_jumped) + jumped.emit(48.0) + emit_signal("landed") + GameState.add_score(10) + GameState.unknown_method() + clamp_speed(3.0) + self.clamp_speed(4.0) + apply_gravity() + var vector := Movement.walk_vector(1.0) + var bag := Inventory.new() + Stats.new() + $Sprite.play("idle") + unknown_free_call() + + +func walk(direction: float) -> void: + # The run speed follows physics.md §3.4; the slide (§3.5) belongs to the enemy script. + pass + + +func jump() -> void: + pass + + +func _on_jumped(height: float) -> void: + pass + + +func _on_landed() -> void: + pass diff --git a/tests/test_extractors_registry.py b/tests/test_extractors_registry.py index db647201ff..defed18244 100644 --- a/tests/test_extractors_registry.py +++ b/tests/test_extractors_registry.py @@ -42,3 +42,12 @@ def test_terraform_migrated(): assert facade.extract_terraform is extract_terraform assert LANGUAGE_EXTRACTORS["terraform"] is extract_terraform + + +def test_gdscript_registered(): + # GDScript was born in extractors/ (it never lived in extract.py): the facade + # re-export and the registry both point at the one module-level object. + from graphify.extractors.gdscript import extract_gdscript + + assert facade.extract_gdscript is extract_gdscript + assert LANGUAGE_EXTRACTORS["gdscript"] is extract_gdscript diff --git a/tests/test_languages.py b/tests/test_languages.py index 4508bbfdd4..23d2ca485b 100644 --- a/tests/test_languages.py +++ b/tests/test_languages.py @@ -4568,3 +4568,206 @@ def test_robot_path_variables_match_case_space_underscore_insensitively(): assert _resolve_robot_import("..${/}Resource${/}common.robot", rel_src) == P("Tests/Resource/common.robot") # any other variable, in any casing, still yields no edge assert _resolve_robot_import("${Root_Dir}/x.robot", rel_src) is None + + +# --------------------------------------------------------------------------- +# GDScript (Godot 4) +# --------------------------------------------------------------------------- +from graphify.extract import extract_gdscript + +_needs_gdscript = pytest.mark.skipif( + _ilu.find_spec("tree_sitter_gdscript") is None, + reason="tree-sitter-gdscript not installed", +) + + +def _godot_project(tmp_path: Path) -> Path: + """A minimal Godot project: two autoloads, a preload-only movement helper, an inventory.""" + (tmp_path / "project.godot").write_text( + "config_version=5\n\n[application]\n\nconfig/name=\"Probe\"\n\n[autoload]\n\n" + "GameState=\"*res://autoload/game_state.gd\"\n" + "AudioBus=\"*uid://bprobe1234\"\n\n[display]\n", + encoding="utf-8", + ) + (tmp_path / "autoload").mkdir() + (tmp_path / "actors").mkdir() + (tmp_path / "autoload" / "game_state.gd").write_text( + "extends Node\n\nsignal paused\n\n\nfunc add_score(points: int) -> void:\n\tpass\n", + encoding="utf-8", + ) + (tmp_path / "autoload" / "audio_bus.gd").write_text( + "extends Node\n\n\nfunc play_sfx(name: String) -> void:\n\tpass\n", encoding="utf-8", + ) + (tmp_path / "autoload" / "audio_bus.gd.uid").write_text("uid://bprobe1234\n", encoding="utf-8") + (tmp_path / "actors" / "movement.gd").write_text( + "## No class_name: preload this script.\nextends RefCounted\n\n\n" + "static func walk_vector(direction: float) -> Vector2:\n\treturn Vector2(direction, 0.0)\n", + encoding="utf-8", + ) + (tmp_path / "actors" / "inventory.gd").write_text( + "extends RefCounted\n\n\nfunc add_item(item: String) -> void:\n\tpass\n", encoding="utf-8", + ) + return tmp_path + + +def _by_label(result: dict, label: str) -> dict: + hits = [n for n in result["nodes"] if n["label"] == label] + assert len(hits) == 1, f"expected one node labelled {label!r}, got {hits}" + return hits[0] + + +def _edges(result: dict, relation: str, src_label: str | None = None) -> list[dict]: + src_id = _by_label(result, src_label)["id"] if src_label else None + return [e for e in result["edges"] + if e["relation"] == relation and (src_id is None or e["source"] == src_id)] + + +@_needs_gdscript +def test_gdscript_fixture_declares_every_shape(tmp_path): + root = _godot_project(tmp_path) + (root / "docs").mkdir() + (root / "docs" / "physics.md").write_text("# Physics\n", encoding="utf-8") + script = root / "actors" / "player.gd" + script.write_text((FIXTURES / "sample.gd").read_text(encoding="utf-8"), encoding="utf-8") + + r = extract_gdscript(script) + assert "error" not in r + labels = {n["label"] for n in r["nodes"]} + # the file node wears the class_name; members are typed + file_node = _by_label(r, "PlayerController") + assert file_node["type"] == "class" and file_node["source_location"] == "L1" + for want in ("clamp_speed()", "_ready()", "walk()", "jump()", "_on_jumped()", "_on_landed()", + "signal jumped", "signal landed", "enum State", "MAX_JUMPS", "Movement", + "Stats", ".speed_for()"): + assert want in labels, want + assert _by_label(r, "signal jumped")["type"] == "signal" + assert _by_label(r, "enum State")["type"] == "enum" + assert _by_label(r, "MAX_JUMPS")["type"] == "constant" + # an inner class holds its function as a method + inner_methods = _edges(r, "method", "Stats") + assert [e["target"] for e in inner_methods] == [_by_label(r, ".speed_for()")["id"]] + + # imports: the two .gd preload/load targets, never the .tscn + imports = _edges(r, "imports", "PlayerController") + assert len(imports) == 2 + assert {e["target"].endswith("_movement_gd") or e["target"].endswith("_inventory_gd") + for e in imports} == {True} + assert all(e["confidence"] == "EXTRACTED" for e in imports) + # `extends CharacterBody2D` is an engine class: no inherits edge at all + assert _edges(r, "inherits") == [] + + ready = _by_label(r, "_ready()")["id"] + use_targets = {e["target"] for e in _edges(r, "uses", "_ready()")} + jumped, landed = _by_label(r, "signal jumped")["id"], _by_label(r, "signal landed")["id"] + # connect + emit on one signal is ONE uses edge; the autoload's signal resolves too + assert jumped in use_targets and landed in use_targets + assert any(t.endswith("_game_state_paused") for t in use_targets), "autoload signal not resolved" + assert len(use_targets) == 3 + # a connect names its callback + refs = _edges(r, "references", "_ready()") + callbacks = {e["target"] for e in refs if e.get("context") == "connect"} + assert callbacks == {_by_label(r, "_on_jumped()")["id"], _by_label(r, "_on_landed()")["id"]} + # Inventory.new() references the inventory script through its load alias + assert any(e["context"] == "instantiates" and e["target"].endswith("_inventory_gd") for e in refs) + + calls = _edges(r, "calls", "_ready()") + targets = {e["target"] for e in calls} + assert _by_label(r, "clamp_speed()")["id"] in targets # clamp_speed(3.0) and self.clamp_speed(4.0), once + assert any(t.endswith("_game_state_add_score") for t in targets), "autoload call not resolved" + assert any(t.endswith("_movement_walk_vector") for t in targets), "preload-alias call not resolved" + assert all(e["confidence"] == "EXTRACTED" for e in calls) + assert all(e["source"] == ready for e in calls) + # what could not be resolved is counted, never guessed, and never handed to the + # shared name-matching resolver + assert r["raw_calls"] == [] + unresolved = {u["callee"] for u in r["unresolved_calls"]} + assert {"GameState.unknown_method", "apply_gravity", "unknown_free_call"} <= unresolved + assert "add_score" not in unresolved and "clamp_speed" not in unresolved + + # documentation citations: explicit pages EXTRACTED, a bare section INFERRED + # against the last page named in the file; the page resolves to docs/ when it exists + cites = [e for e in r["edges"] if e.get("context") == "citation"] + by_label = {} + for e in cites: + node = next(n for n in r["nodes"] if n["id"] == e["target"]) + by_label[node["label"]] = (e["source"], e["confidence"], node["source_file"]) + assert by_label["movement.md §2.1"] == (file_node["id"], "EXTRACTED", "movement.md") + assert by_label["movement.md §2.3"] == (file_node["id"], "INFERRED", "movement.md") + assert by_label["conventions.md §1.2"][1] == "EXTRACTED" + walk = _by_label(r, "walk()")["id"] + assert by_label["physics.md §3.4"] == (walk, "EXTRACTED", "docs/physics.md") + assert by_label["physics.md §3.5"] == (walk, "INFERRED", "docs/physics.md") + section = next(n for n in r["nodes"] if n["label"] == "physics.md §3.4") + assert section["file_type"] == "doc" and section["type"] == "section" + assert section["source_location"] == "§3.4" + + +@_needs_gdscript +def test_gdscript_inherits_and_ancestor_calls(tmp_path): + root = _godot_project(tmp_path) + (root / "actors" / "actor.gd").write_text( + "extends CharacterBody2D\nclass_name Actor\n\nsignal died\n\n\n" + "func take_damage(amount: int) -> void:\n\tpass\n", encoding="utf-8") + (root / "actors" / "enemy.gd").write_text( + "extends Actor\n\n\nfunc patrol() -> void:\n\tpass\n", encoding="utf-8") + grunt = root / "actors" / "grunt.gd" + grunt.write_text( + "extends \"res://actors/enemy.gd\"\n\n\nfunc _ready() -> void:\n" + "\ttake_damage(1)\n\tsuper.patrol()\n\tdied.emit()\n\tActor.take_damage(2)\n", + encoding="utf-8") + + r = extract_gdscript(grunt) + inherits = _edges(r, "inherits", "grunt.gd") + assert len(inherits) == 1 and inherits[0]["target"].endswith("_enemy_gd") + calls = {e["target"] for e in _edges(r, "calls", "_ready()")} + assert any(t.endswith("_actor_take_damage") for t in calls), "grandparent method not resolved" + assert any(t.endswith("_enemy_patrol") for t in calls), "super call not resolved" + uses = {e["target"] for e in _edges(r, "uses", "_ready()")} + assert any(t.endswith("_actor_died") for t in uses), "inherited signal not resolved" + assert r["unresolved_calls"] == [] + + enemy = extract_gdscript(root / "actors" / "enemy.gd") + assert [e["target"].endswith("_actor_gd") for e in _edges(enemy, "inherits", "enemy.gd")] == [True] + + +@_needs_gdscript +def test_gdscript_autoload_scripts_wear_their_autoload_name(tmp_path): + root = _godot_project(tmp_path) + state = extract_gdscript(root / "autoload" / "game_state.gd") + assert _by_label(state, "GameState")["type"] == "class" + # a uid:// autoload resolves through its .gd.uid sidecar + bus = extract_gdscript(root / "autoload" / "audio_bus.gd") + assert "AudioBus" in {n["label"] for n in bus["nodes"]} + caller = root / "actors" / "door.gd" + caller.write_text("extends Node\n\n\nfunc _ready() -> void:\n\tAudioBus.play_sfx(\"creak\")\n", + encoding="utf-8") + r = extract_gdscript(caller) + assert any(e["target"].endswith("_audio_bus_play_sfx") for e in _edges(r, "calls", "_ready()")) + + +@_needs_gdscript +def test_gdscript_outside_a_project_still_extracts(tmp_path): + lone = tmp_path / "lone.gd" + lone.write_text( + "extends Node\n\nsignal done\n\n\nfunc go() -> void:\n\tdone.emit()\n\tGameState.x()\n\thelp()\n\n\n" + "func help() -> void:\n\tpass\n", encoding="utf-8") + r = extract_gdscript(lone) + assert "error" not in r + assert {n["label"] for n in r["nodes"]} >= {"lone.gd", "signal done", "go()", "help()"} + assert [e["target"] for e in _edges(r, "uses", "go()")] == [_by_label(r, "signal done")["id"]] + assert [e["target"] for e in _edges(r, "calls", "go()")] == [_by_label(r, "help()")["id"]] + assert {u["callee"] for u in r["unresolved_calls"]} == {"GameState.x"} + + +def test_gdscript_is_dispatched_to_its_own_extractor_only(tmp_path): + """No other grammar reaches a .gd file: the dispatch, the family table, the + detect set and the watcher all name the GDScript extractor for it.""" + from graphify.detect import CODE_EXTENSIONS + from graphify.extract import _DISPATCH, _get_extractor, _lang_family + from graphify.watch import _WATCHED_EXTENSIONS + + assert _DISPATCH[".gd"] is extract_gdscript + assert _get_extractor(tmp_path / "thing.gd") is extract_gdscript + assert _lang_family("src/thing.gd") == "gdscript" + assert ".gd" in CODE_EXTENSIONS and ".gd" in _WATCHED_EXTENSIONS + assert sum(1 for _suffix, fn in _DISPATCH.items() if fn is extract_gdscript) == 1 From 693d46cf79b97e13aec15f79e23cc4088e7050d9 Mon Sep 17 00:00:00 2001 From: Anthony Date: Wed, 23 Sep 2026 00:18:41 +1000 Subject: [PATCH 3/5] docs: list GDScript among the supported grammars --- README.md | 2 +- graphify/extractors/MIGRATION.md | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index a49abe8638..daf30df798 100644 --- a/README.md +++ b/README.md @@ -341,7 +341,7 @@ To remove graphify from all platforms at once: `graphify uninstall` (add `--purg | Type | Extensions | |------|-----------| -| Code (37 tree-sitter grammars) | `.py .ts .mts .cts .js .jsx .tsx .mjs .go .rs .java .c .cpp .cc .cxx .h .hpp .cu .cuh .metal .rb .cs .kt .kts .scala .php .swift .lua .luau .toc .zig .ps1 .psm1 .psd1 .ex .exs .m .mm .ml .mli .jl .vue .svelte .astro .groovy .gradle .dart .v .sv .svh .sql .f .f90 .f95 .f03 .f08 .pas .pp .dpr .dpk .lpr .inc .dfm .lfm .lpk .sh .bash .json .dm .dme .dmi .dmm .dmf .sln .slnx .csproj .fsproj .vbproj .xaml .razor .cshtml` (`.dm`/`.dme` requires `uv tool install graphifyy[dm]`, `.ml`/`.mli` requires `uv tool install graphifyy[ocaml]`; `.mts`/`.cts` reuse the TypeScript grammar, `.cc`/`.cxx` and CUDA `.cu`/`.cuh` and Metal `.metal` reuse the C++ grammar) | +| Code (38 tree-sitter grammars) | `.py .ts .mts .cts .js .jsx .tsx .mjs .go .rs .java .c .cpp .cc .cxx .h .hpp .cu .cuh .metal .rb .cs .kt .kts .scala .php .swift .lua .luau .toc .zig .gd .ps1 .psm1 .psd1 .ex .exs .m .mm .ml .mli .jl .vue .svelte .astro .groovy .gradle .dart .v .sv .svh .sql .f .f90 .f95 .f03 .f08 .pas .pp .dpr .dpk .lpr .inc .dfm .lfm .lpk .sh .bash .json .dm .dme .dmi .dmm .dmf .sln .slnx .csproj .fsproj .vbproj .xaml .razor .cshtml` (`.dm`/`.dme` requires `uv tool install graphifyy[dm]`, `.ml`/`.mli` requires `uv tool install graphifyy[ocaml]`; `.mts`/`.cts` reuse the TypeScript grammar, `.cc`/`.cxx` and CUDA `.cu`/`.cuh` and Metal `.metal` reuse the C++ grammar) | | Salesforce Apex | `.cls .trigger` (regex-based; classes, interfaces, enums, methods, triggers, SOQL/DML edges) | | Terraform / HCL | `.tf .tfvars .hcl` (requires `uv tool install graphifyy[terraform]`) | | OCaml | `.ml .mli` (requires `uv tool install graphifyy[ocaml]`) | diff --git a/graphify/extractors/MIGRATION.md b/graphify/extractors/MIGRATION.md index 75e1525e2e..8b83a16391 100644 --- a/graphify/extractors/MIGRATION.md +++ b/graphify/extractors/MIGRATION.md @@ -25,6 +25,7 @@ written so an AI agent can execute it in a single session. | sln | yes | | pascal_forms (dfm + lfm) | yes | | json_config | yes | +| gdscript | yes (born here, never lived in extract.py) | | (config-driven core: python, js, java, c, cpp, csharp, kotlin, scala, php, lua, swift, groovy, vue, svelte, astro, xaml, groovy) | no — shared _extract_generic core, move as one batch | | (other bespoke: julia, verilog, markdown, objc, csproj, slnx, lazarus_package, pascal) | no | From a1561e32545d59e971d3526330c083770bddd88f Mon Sep 17 00:00:00 2001 From: Anthony Date: Tue, 22 Sep 2026 23:55:07 +1000 Subject: [PATCH 4/5] chore: strip `+gms1` from version number for PR to original repo --- pyproject.toml | 2 +- uv.lock | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index baa87c616a..3ccc579a3b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" -version = "0.9.65+gms1" +version = "0.9.65" description = "AI coding assistant skill (Claude Code, CodeBuddy, Codex, OpenCode, Kilo Code, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Devin CLI, Google Antigravity) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph" readme = "README.md" license = "Apache-2.0" diff --git a/uv.lock b/uv.lock index 6cc3a6f3e3..8df43a7716 100644 --- a/uv.lock +++ b/uv.lock @@ -1096,7 +1096,7 @@ wheels = [ [[package]] name = "graphifyy" -version = "0.9.65+gms1" +version = "0.9.65" source = { editable = "." } dependencies = [ { name = "networkx", version = "3.4.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, From b569873bbe2e020fc872c7a0af03526cace91689 Mon Sep 17 00:00:00 2001 From: Anthony Date: Wed, 23 Sep 2026 00:54:53 +1000 Subject: [PATCH 5/5] fix(extract): harden the GDScript project index and citation resolution MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the three advisory findings from the automated review: - the project index initialises its uid table before anything can fail, so a project.godot that cannot be read still leaves an index that answers None for a uid:// instead of raising AttributeError; - a cited documentation page only rewrites the recorded path when it resolves to a real file INSIDE the project root (or its docs/), and a page carrying '..' segments is never looked up at all — the caller already hands a basename, but the resolver now enforces containment itself; - citations are walked in the order they appear in a comment, so a bare section attaches to the page named before it and never to one named later on the same line. One test per finding. --- graphify/extractors/gdscript.py | 52 +++++++++++++++++++++------------ tests/test_languages.py | 47 +++++++++++++++++++++++++++++ 2 files changed, 81 insertions(+), 18 deletions(-) diff --git a/graphify/extractors/gdscript.py b/graphify/extractors/gdscript.py index f7422154e2..2ce8721107 100644 --- a/graphify/extractors/gdscript.py +++ b/graphify/extractors/gdscript.py @@ -58,7 +58,10 @@ def __init__(self, root: Path) -> None: self.root = root self.autoloads: dict[str, Path] = {} self.class_names: dict[str, Path] = {} - uids: dict[str, Path] = {} + # always present: a project.godot that cannot be read returns early below, + # and resolve() must still answer None for a uid:// rather than raise + self._uids: dict[str, Path] = {} + uids = self._uids try: for uid_file in root.rglob("*.gd.uid"): if ".godot" in uid_file.parts: @@ -89,7 +92,6 @@ def __init__(self, root: Path) -> None: target = self.resolve(m.group(2), uids) if target is not None: self.autoloads[m.group(1)] = target - self._uids = uids def resolve(self, ref: str, uids: dict[str, Path] | None = None) -> Path | None: """A ``res://`` or ``uid://`` reference as an absolute path, or None.""" @@ -547,12 +549,23 @@ def enclosing(byte: int) -> str: return file_nid def section_node(page: str, section: str, line: int) -> str: + # The caller hands a basename, but the resolver enforces containment itself: + # a page is looked up under the project root (or its docs/), and only a real + # file that RESOLVES inside the root is allowed to rewrite the recorded path. page_rel = page.lstrip("./") - if project is not None: - for candidate in (project.root / page_rel, project.root / "docs" / page_rel): - if candidate.is_file(): - page_rel = candidate.relative_to(project.root).as_posix() - break + if project is not None and ".." not in Path(page_rel).parts: + try: + root = project.root.resolve() + except OSError: + root = project.root + for candidate in (root / page_rel, root / "docs" / page_rel): + try: + resolved = candidate.resolve() + if resolved.is_file() and resolved.is_relative_to(root): + page_rel = resolved.relative_to(root).as_posix() + break + except OSError: + continue nid = _make_id("doc", page_rel, "s" + section) if nid not in seen_ids: seen_ids.add(nid) @@ -569,17 +582,20 @@ def walk_comments(node) -> None: text = _read_text(node, source) line = line_of(node) src = enclosing(node.start_byte) - explicit: list[tuple[int, int]] = [] - for m in _CITATION_RE.finditer(text): - page = m.group(1).split("/")[-1] - last_page = page - explicit.append((m.start(), m.end())) - add_edge(src, section_node(page, m.group(2), line), "references", line, context="citation") - for m in _BARE_CITATION_RE.finditer(text): - if last_page is None or any(a <= m.start() < b for a, b in explicit): - continue - add_edge(src, section_node(last_page, m.group(1), line), "references", line, - confidence="INFERRED", context="citation") + explicit = list(_CITATION_RE.finditer(text)) + spans = [(m.start(), m.end()) for m in explicit] + bare = [m for m in _BARE_CITATION_RE.finditer(text) + if not any(a <= m.start() < b for a, b in spans)] + # walk every citation in the order it appears, so a bare section takes the + # page named before it — never one named later on the same line + for m in sorted(explicit + bare, key=lambda m: m.start()): + if m.re is _CITATION_RE: + last_page = m.group(1).split("/")[-1] + add_edge(src, section_node(last_page, m.group(2), line), "references", line, + context="citation") + elif last_page is not None: + add_edge(src, section_node(last_page, m.group(1), line), "references", line, + confidence="INFERRED", context="citation") return for c in node.children: walk_comments(c) diff --git a/tests/test_languages.py b/tests/test_languages.py index 23d2ca485b..bcbca3bedd 100644 --- a/tests/test_languages.py +++ b/tests/test_languages.py @@ -4771,3 +4771,50 @@ def test_gdscript_is_dispatched_to_its_own_extractor_only(tmp_path): assert _lang_family("src/thing.gd") == "gdscript" assert ".gd" in CODE_EXTENSIONS and ".gd" in _WATCHED_EXTENSIONS assert sum(1 for _suffix, fn in _DISPATCH.items() if fn is extract_gdscript) == 1 + + +def test_gdscript_project_index_survives_an_unreadable_project_file(tmp_path): + """A project.godot that cannot be read leaves an index that still answers, never + raises: uid:// lookups return None (review finding on the early return).""" + from graphify.extractors.gdscript import _GodotProject + + (tmp_path / "autoload").mkdir() + (tmp_path / "autoload" / "audio_bus.gd.uid").write_text("uid://bprobe1234\n", encoding="utf-8") + project = _GodotProject(tmp_path) # no project.godot on disk: the read fails + assert project.autoloads == {} + assert project.resolve("uid://bprobe1234") == tmp_path / "autoload" / "audio_bus.gd" + assert project.resolve("uid://missing") is None + + +@_needs_gdscript +def test_gdscript_citation_pages_never_resolve_outside_the_project(tmp_path): + (tmp_path / "outside.md").write_text("# outside\n", encoding="utf-8") + (tmp_path / "proj").mkdir() + root = _godot_project(tmp_path / "proj") + script = root / "actors" / "cite.gd" + script.write_text("## Rules in ../../outside.md §1 and docs/../../outside.md §2.\nextends Node\n", + encoding="utf-8") + r = extract_gdscript(script) + sections = [n for n in r["nodes"] if n.get("type") == "section"] + assert {n["label"] for n in sections} == {"outside.md §1", "outside.md §2"} + for n in sections: + assert ".." not in n["source_file"] and n["source_file"] == "outside.md" + + +@_needs_gdscript +def test_gdscript_bare_citation_takes_the_page_named_before_it(tmp_path): + root = _godot_project(tmp_path) + script = root / "actors" / "order.gd" + script.write_text( + "## See §9.9 first, then guide.md §1.1 and §1.2.\n" + "## Later, §1.3 alone.\n" + "extends Node\n", encoding="utf-8") + r = extract_gdscript(script) + by_label = {} + for e in (e for e in r["edges"] if e.get("context") == "citation"): + node = next(n for n in r["nodes"] if n["id"] == e["target"]) + by_label[node["label"]] = e["confidence"] + # no page precedes §9.9, so it is not attributed to a page named after it + assert "guide.md §9.9" not in by_label + assert by_label == {"guide.md §1.1": "EXTRACTED", "guide.md §1.2": "INFERRED", + "guide.md §1.3": "INFERRED"}