From 304c82fe104a4739d84c08cf0a6958ddad78b6d9 Mon Sep 17 00:00:00 2001 From: Abdullah Date: Mon, 21 Sep 2026 16:25:48 +0530 Subject: [PATCH] feat: add VB.NET language extractor --- graphify/analyze.py | 2 +- graphify/detect.py | 2 +- graphify/extract.py | 9 +- graphify/extractors/__init__.py | 2 + graphify/extractors/vbnet.py | 409 +++++++++++++++++++++++++ pyproject.toml | 3 +- tests/fixtures/new_languages/sample.vb | 10 + tests/test_vbnet_extractor.py | 142 +++++++++ uv.lock | 19 +- 9 files changed, 593 insertions(+), 5 deletions(-) create mode 100644 graphify/extractors/vbnet.py create mode 100644 tests/fixtures/new_languages/sample.vb create mode 100644 tests/test_vbnet_extractor.py diff --git a/graphify/analyze.py b/graphify/analyze.py index df1c206610..a51fcf783d 100644 --- a/graphify/analyze.py +++ b/graphify/analyze.py @@ -38,7 +38,7 @@ **{e: "c" for e in (".c", ".h", ".cpp", ".cc", ".cxx", ".hpp")}, **{e: "ruby" for e in (".rb", ".rake")}, **{e: "swift" for e in (".swift",)}, - **{e: "dotnet" for e in (".cs",)}, + **{e: "dotnet" for e in (".cs", ".vb")}, **{e: "php" for e in (".php",)}, **{e: "r" for e in (".r",)}, } diff --git a/graphify/detect.py b/graphify/detect.py index eec8aaf1d2..e7d1202c66 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -41,7 +41,7 @@ class FileType(str, Enum): _MTIME_COARSE_S = 2.0 _MTIME_SUBSECOND_S = 0.05 -CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.mts', '.cts', '.js', '.jsx', '.mjs', '.cjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.cu', '.cuh', '.metal', '.rb', '.rake', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.psm1', '.psd1', '.ex', '.exs', '.m', '.mm', '.ml', '.mli', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.svh', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.tf', '.tfvars', '.hcl', '.dm', '.dme', '.dmi', '.dmm', '.dmf', '.sln', '.slnx', '.csproj', '.fsproj', '.vbproj', '.xaml', '.razor', '.cshtml', '.cls', '.trigger', '.lisp', '.cl', '.lsp', '.asd', '.robot', '.resource'} +CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.mts', '.cts', '.js', '.jsx', '.mjs', '.cjs', '.ejs', '.ets', '.go', '.rs', '.vb', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.cu', '.cuh', '.metal', '.rb', '.rake', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.psm1', '.psd1', '.ex', '.exs', '.m', '.mm', '.ml', '.mli', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.svh', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.tf', '.tfvars', '.hcl', '.dm', '.dme', '.dmi', '.dmm', '.dmf', '.sln', '.slnx', '.csproj', '.fsproj', '.vbproj', '.xaml', '.razor', '.cshtml', '.cls', '.trigger', '.lisp', '.cl', '.lsp', '.asd', '.robot', '.resource'} DOC_EXTENSIONS = {'.md', '.mdx', '.qmd', '.skill', '.txt', '.rst', '.html', '.yaml', '.yml'} PAPER_EXTENSIONS = {'.pdf'} IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'} diff --git a/graphify/extract.py b/graphify/extract.py index da66f41317..80d3ffd2ee 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -60,6 +60,7 @@ from graphify.extractors.sql import extract_sql # noqa: F401 from graphify.extractors.terraform import extract_terraform, prepare_terraform, resolve_terraform_modules # noqa: F401 from graphify.extractors.verilog import extract_verilog # noqa: F401 +from graphify.extractors.vbnet import extract_vbnet, resolve_vbnet_partial_calls # noqa: F401 from graphify.extractors.zig import extract_zig # noqa: F401 from graphify.security import sanitize_metadata from graphify.paths import disambiguate_ambiguous_candidates @@ -2728,6 +2729,7 @@ def _canonicalize_csharp_namespace_nodes(all_nodes: list[dict], all_edges: list[ ".php", ".phtml", ".php3", ".php4", ".php5", ".php7", ".phps", # PHP fns/classes ".sql", # SQL identifiers ".nim", ".nims", ".nimble", # Nim (style-insensitive) + ".vb", }) @@ -2766,7 +2768,7 @@ def _lang_is_case_insensitive(source_file: object) -> bool: ".rb": "ruby", ".rake": "ruby", ".php": "php", ".phtml": "php", ".php3": "php", ".php4": "php", ".php5": "php", ".php7": "php", ".phps": "php", - ".cs": "dotnet", ".razor": "dotnet", ".cshtml": "dotnet", ".xaml": "dotnet", + ".cs": "dotnet", ".vb": "dotnet", ".razor": "dotnet", ".cshtml": "dotnet", ".xaml": "dotnet", ".lua": "lua", ".luau": "lua", ".zig": "zig", ".ex": "elixir", ".exs": "elixir", @@ -5194,6 +5196,9 @@ def _resolve_kotlin_member_calls( register_language_resolver( LanguageResolver("rust_self_member_calls", frozenset({".rs"}), _resolve_rust_self_member_calls) ) +register_language_resolver( + LanguageResolver("vbnet_partial_calls", frozenset({".vb"}), resolve_vbnet_partial_calls) +) register_language_resolver( LanguageResolver( "elixir_import_targets", @@ -6293,6 +6298,7 @@ def add_existing_edge(edge: dict) -> None: ".metal": extract_cpp, ".rb": extract_ruby, ".rake": extract_ruby, ".cs": extract_csharp, + ".vb": extract_vbnet, ".kt": extract_kotlin, ".kts": extract_kotlin, ".scala": extract_scala, @@ -6378,6 +6384,7 @@ def add_existing_edge(edge: dict) -> None: # rather than falling back like Pascal does. Used by the #1745 warning in # extract() to tell the user which extra restores the language. _EXTRA_FOR_EXTENSION = { + ".vb": "vbnet", ".sql": "sql", ".tf": "terraform", ".tfvars": "terraform", diff --git a/graphify/extractors/__init__.py b/graphify/extractors/__init__.py index 68ff3340c3..bd5172663f 100644 --- a/graphify/extractors/__init__.py +++ b/graphify/extractors/__init__.py @@ -32,6 +32,7 @@ from graphify.extractors.sql import extract_sql from graphify.extractors.terraform import extract_terraform from graphify.extractors.verilog import extract_verilog +from graphify.extractors.vbnet import extract_vbnet from graphify.extractors.zig import extract_zig LANGUAGE_EXTRACTORS: dict[str, Callable[[Path], dict]] = { @@ -62,5 +63,6 @@ "sql": extract_sql, "terraform": extract_terraform, "verilog": extract_verilog, + "vbnet": extract_vbnet, "zig": extract_zig, } diff --git a/graphify/extractors/vbnet.py b/graphify/extractors/vbnet.py new file mode 100644 index 0000000000..a9c4759ec7 --- /dev/null +++ b/graphify/extractors/vbnet.py @@ -0,0 +1,409 @@ +"""Structural extraction for Visual Basic .NET source files.""" +from __future__ import annotations + +from pathlib import Path +from typing import Any + +from tree_sitter import Node + +from graphify.extractors.base import _file_stem, _make_id, _read_text + + +_TYPE_BLOCKS = { + "class_block": "class", + "module_block": "module", + "interface_block": "interface", + "structure_block": "structure", + "enum_block": "enum", +} + + +def resolve_vbnet_partial_calls( + per_file: list[dict], all_nodes: list[dict], all_edges: list[dict] +) -> None: + """Resolve calls across files that declare the same partial VB type.""" + methods: dict[tuple[str, str, int], list[str]] = {} + for node in all_nodes: + metadata = node.get("metadata") + if not isinstance(metadata, dict) or metadata.get("language") != "vbnet": + continue + if metadata.get("kind") not in {"method", "constructor"}: + continue + owner = metadata.get("owner") + name = metadata.get("name") + accepted_arities = metadata.get("accepted_arities") + if ( + isinstance(owner, str) + and isinstance(name, str) + and isinstance(accepted_arities, list) + ): + for arity in accepted_arities: + if isinstance(arity, int): + methods.setdefault( + (owner.casefold(), name.casefold(), arity), [] + ).append(node["id"]) + + existing = { + (edge.get("source"), edge.get("target")) + for edge in all_edges + if edge.get("relation") == "calls" + } + for result in per_file: + for call in result.get("raw_calls", []): + if call.get("language") != "vbnet" or not call.get("owner"): + continue + key = ( + str(call["owner"]).casefold(), + str(call.get("callee", "")).casefold(), + int(call.get("arity", -1)), + ) + candidates = methods.get(key, []) + caller = call.get("caller_nid") + if len(candidates) != 1 or candidates[0] == caller: + continue + pair = (caller, candidates[0]) + if pair in existing: + continue + existing.add(pair) + all_edges.append({ + "source": caller, + "target": candidates[0], + "relation": "calls", + "context": "partial_type_call", + "confidence": "EXTRACTED", + "confidence_score": 1.0, + "source_file": call.get("source_file", ""), + "source_location": call.get("source_location"), + "weight": 1.0, + }) + + +def extract_vbnet(path: Path) -> dict: + try: + import tree_sitter_vb_dotnet + from tree_sitter import Language, Parser + except ImportError: + return {"nodes": [], "edges": [], "error": "tree-sitter-vb-dotnet not installed"} + + try: + source = path.read_bytes() + root = Parser(Language(tree_sitter_vb_dotnet.language())).parse(source).root_node + except Exception as exc: + return {"nodes": [], "edges": [], "error": f"VB.NET grammar failed to load: {exc}"} + + source_file = str(path) + stem = _file_stem(path) + file_id = _make_id(source_file) + nodes: list[dict[str, Any]] = [] + edges: list[dict[str, Any]] = [] + raw_calls: list[dict[str, Any]] = [] + seen_ids: set[str] = set() + seen_edges: set[tuple[str, str, str]] = set() + types_by_name: dict[str, str] = {} + methods: dict[tuple[str, str, int], list[str]] = {} + events: dict[tuple[str, str], str] = {} + bodies: list[tuple[Node, str, str, str]] = [] + pending_handles: list[tuple[str, str, str, Node]] = [] + + def add_node( + nid: str, + label: str, + node: Node, + *, + kind: str, + source_backed: bool = True, + callable_node: bool = False, + metadata: dict[str, Any] | None = None, + ) -> str: + if nid not in seen_ids: + seen_ids.add(nid) + details: dict[str, Any] = {"language": "vbnet", "kind": kind} + if metadata: + details.update(metadata) + item: dict[str, Any] = { + "id": nid, + "label": label, + "file_type": "code", + "source_location": f"L{node.start_point[0] + 1}", + "metadata": details, + } + if source_backed: + item["source_file"] = source_file + if callable_node: + item["_callable"] = True + nodes.append(item) + return nid + + def add_edge(source_id: str, target_id: str, relation: str, node: Node) -> None: + key = (source_id, target_id, relation) + if not source_id or not target_id or source_id == target_id or key in seen_edges: + return + seen_edges.add(key) + edges.append({ + "source": source_id, + "target": target_id, + "relation": relation, + "confidence": "EXTRACTED", + "source_file": source_file, + "source_location": f"L{node.start_point[0] + 1}", + "weight": 1.0, + }) + + add_node(file_id, path.name, root, kind="file") + + for statement in (node for node in root.named_children if node.type == "imports_statement"): + namespace_node = next( + (child for child in statement.named_children if child.type == "namespace_name"), + None, + ) + if namespace_node is None: + continue + namespace = _read_text(namespace_node, source) + target = add_node( + _make_id("vbnet", "namespace", namespace.casefold()), + namespace, + namespace_node, + kind="external_namespace", + source_backed=False, + ) + add_edge(file_id, target, "imports", statement) + + def type_reference(name: str, node: Node) -> str: + short_name = name.split(".")[-1] + target = types_by_name.get(short_name.casefold()) + if target is not None: + return target + return add_node( + _make_id("vbnet", "type", name.casefold()), + short_name, + node, + kind="external_type", + source_backed=False, + ) + + def add_data_member(type_id: str, member: Node, name: str, kind: str) -> str: + member_id = add_node( + _make_id(type_id, kind, name.casefold(), str(member.start_point[0])), + name, + member, + kind=kind, + ) + add_edge(type_id, member_id, "contains", member) + return member_id + + def process_type(block: Node, parent_id: str, namespace: str) -> None: + kind = _TYPE_BLOCKS[block.type] + name_node = block.child_by_field_name("name") + if name_node is None: + return + name = _read_text(name_node, source) + full_name = f"{namespace}.{name}" if namespace else name + owner_key = full_name.casefold() + type_id = add_node( + _make_id(stem, "type", owner_key), + name, + block, + kind=kind, + callable_node=kind in {"class", "structure"}, + metadata={"name": name, "full_name": full_name}, + ) + add_edge(parent_id, type_id, "contains", block) + types_by_name[name.casefold()] = type_id + + for clause in block.named_children: + if clause.type not in {"inherits_clause", "implements_clause"}: + continue + relation = "inherits" if clause.type == "inherits_clause" else "implements" + for type_node in (child for child in clause.named_children if child.type == "type"): + referenced = _read_text(type_node, source) + add_edge(type_id, type_reference(referenced, type_node), relation, clause) + + for member in block.named_children: + if member.type == "enum_member": + member_name = member.child_by_field_name("name") + if member_name is not None: + add_data_member( + type_id, member, _read_text(member_name, source), "enum_member" + ) + continue + if member.type == "field_declaration": + for declarator in ( + child for child in member.named_children + if child.type == "variable_declarator" + ): + member_name = declarator.child_by_field_name("name") + if member_name is not None: + add_data_member( + type_id, + declarator, + _read_text(member_name, source), + "field", + ) + continue + if member.type in {"property_declaration", "event_declaration"}: + member_name = member.child_by_field_name("name") + if member_name is not None: + data_name = _read_text(member_name, source) + member_id = add_data_member( + type_id, + member, + data_name, + "property" if member.type == "property_declaration" else "event", + ) + if member.type == "event_declaration": + events[(owner_key, data_name.casefold())] = member_id + continue + if member.type not in {"method_declaration", "constructor_declaration"}: + continue + + if member.type == "constructor_declaration": + method_name = "New" + method_kind = "constructor" + else: + method_name_node = member.child_by_field_name("name") + if method_name_node is None: + continue + method_name = _read_text(method_name_node, source) + method_kind = "method" + parameters = member.child_by_field_name("parameters") + parameter_nodes = ( + [child for child in parameters.named_children if child.type == "parameter"] + if parameters is not None + else [] + ) + arity = len(parameter_nodes) + required_arity = sum( + not _read_text(parameter, source).lstrip().casefold().startswith("optional ") + for parameter in parameter_nodes + ) + accepted_arities = list(range(required_arity, arity + 1)) + method_id = add_node( + _make_id( + type_id, + method_kind, + method_name.casefold(), + str(arity), + str(member.start_point[0]), + ), + f"{method_name}()", + member, + kind=method_kind, + callable_node=True, + metadata={ + "owner": owner_key, + "name": method_name, + "arity": arity, + "accepted_arities": accepted_arities, + }, + ) + add_edge(type_id, method_id, "method", member) + for accepted_arity in accepted_arities: + methods.setdefault( + (owner_key, method_name.casefold(), accepted_arity), [] + ).append(method_id) + for clause in ( + child for child in member.named_children if child.type == "handles_clause" + ): + for handled in ( + child for child in clause.named_children if child.type == "namespace_name" + ): + pending_handles.append(( + method_id, + owner_key, + _read_text(handled, source), + handled, + )) + bodies.append((member, method_id, owner_key, name)) + + def scan(node: Node, parent_id: str, namespace: str) -> None: + if node.type == "imports_statement": + return + if node.type == "namespace_block": + name_node = node.child_by_field_name("name") + if name_node is None: + return + local_name = _read_text(name_node, source) + full_namespace = f"{namespace}.{local_name}" if namespace else local_name + namespace_id = add_node( + _make_id(stem, "namespace", full_namespace.casefold()), + local_name, + node, + kind="namespace", + metadata={"full_name": full_namespace}, + ) + add_edge(parent_id, namespace_id, "contains", node) + for child in node.named_children: + if child != name_node: + scan(child, namespace_id, full_namespace) + return + if node.type in _TYPE_BLOCKS: + process_type(node, parent_id, namespace) + # Nested type declarations are the only child containers that still + # need recursion after processing this type's own members. + for child in node.named_children: + if child.type == "type_declaration": + scan(child, parent_id, namespace) + return + for child in node.named_children: + scan(child, parent_id, namespace) + + scan(root, file_id, "") + + for method_id, owner_key, written, handled_node in pending_handles: + event_name = written.split(".")[-1] + target = events.get((owner_key, event_name.casefold())) + if target is None: + target = add_node( + _make_id("vbnet", "event", written.casefold()), + written, + handled_node, + kind="external_event", + source_backed=False, + ) + add_edge(method_id, target, "handles", handled_node) + + for body, caller_id, owner_key, type_name in bodies: + stack = [body] + while stack: + node = stack.pop() + if node != body and node.type in { + "method_declaration", "constructor_declaration", "property_declaration" + }: + continue + if node.type == "invocation": + target_node = node.child_by_field_name("target") + arguments = node.child_by_field_name("arguments") + if target_node is not None: + written = _read_text(target_node, source) + parts = written.split(".") + receiver = ".".join(parts[:-1]).casefold() + callee = parts[-1] + known_receiver = ( + not receiver + or receiver in {"me", "myclass", type_name.casefold()} + ) + if known_receiver: + arity = len(arguments.named_children) if arguments is not None else 0 + candidates = methods.get( + (owner_key, callee.casefold(), arity), [] + ) + if len(candidates) == 1: + add_edge(caller_id, candidates[0], "calls", node) + else: + raw_calls.append({ + "caller_nid": caller_id, + "callee": callee, + "arity": arity, + "owner": owner_key, + "is_member_call": True, + "language": "vbnet", + "source_file": source_file, + "source_location": f"L{node.start_point[0] + 1}", + }) + stack.extend(reversed(node.named_children)) + + clean_edges = [ + edge for edge in edges + if edge["source"] in seen_ids and edge["target"] in seen_ids + ] + return {"nodes": nodes, "edges": clean_edges, "raw_calls": raw_calls} diff --git a/pyproject.toml b/pyproject.toml index 5e39d8c54f..27c5054bd5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -105,7 +105,8 @@ commonlisp = ["tree-sitter-commonlisp"] # every platform, no C toolchain; optional because Robot Framework corpora are # QA-automation specific. robot = ["robotframework>=4.0"] -all = ["mcp>=1,<3", "starlette>=1.3.1,<2", "neo4j", "falkordb", "pypdf>=6.16.1", "markdownify", "watchdog", "graspologic; python_version < '3.13'", "graspologic-native>=1.3.1,<2; python_version >= '3.13'", "python-docx", "openpyxl", "psycopg[binary]", "faster-whisper; python_version >= '3.11'", "yt-dlp>=2026.7.4", "matplotlib", "pillow>=12.3.0", "numpy>=2.0; python_version >= '3.13'", "openai", "tiktoken", "boto3", "anthropic", "tree-sitter-sql", "jieba; python_version < '3.12'", "jieba-py>=0.46.12,<1; python_version >= '3.12'", "tree-sitter-dm", "tree-sitter-hcl", "tree-sitter-pascal", "tree-sitter-ocaml", "tree-sitter-commonlisp", "robotframework>=4.0"] +vbnet = ["tree-sitter-vb-dotnet==0.3.0"] +all = ["mcp>=1,<3", "starlette>=1.3.1,<2", "neo4j", "falkordb", "pypdf>=6.16.1", "markdownify", "watchdog", "graspologic; python_version < '3.13'", "graspologic-native>=1.3.1,<2; python_version >= '3.13'", "python-docx", "openpyxl", "psycopg[binary]", "faster-whisper; python_version >= '3.11'", "yt-dlp>=2026.7.4", "matplotlib", "pillow>=12.3.0", "numpy>=2.0; python_version >= '3.13'", "openai", "tiktoken", "boto3", "anthropic", "tree-sitter-sql", "jieba; python_version < '3.12'", "jieba-py>=0.46.12,<1; python_version >= '3.12'", "tree-sitter-dm", "tree-sitter-hcl", "tree-sitter-pascal", "tree-sitter-ocaml", "tree-sitter-commonlisp", "robotframework>=4.0", "tree-sitter-vb-dotnet==0.3.0"] [project.scripts] graphify = "graphify.__main__:main" diff --git a/tests/fixtures/new_languages/sample.vb b/tests/fixtures/new_languages/sample.vb new file mode 100644 index 0000000000..ee67dcd74a --- /dev/null +++ b/tests/fixtures/new_languages/sample.vb @@ -0,0 +1,10 @@ +Namespace Fixtures + Public Class Sample + Public Sub Run() + Helper() + End Sub + + Private Sub Helper() + End Sub + End Class +End Namespace diff --git a/tests/test_vbnet_extractor.py b/tests/test_vbnet_extractor.py new file mode 100644 index 0000000000..ee3f024b75 --- /dev/null +++ b/tests/test_vbnet_extractor.py @@ -0,0 +1,142 @@ +"""Extraction coverage for vbnet.""" + + +from __future__ import annotations + + + + + +import sys + + +from pathlib import Path + + + + + +from graphify.extract import extract + + + + + +FIXTURE = Path(__file__).parent / "fixtures" / "new_languages" / "sample.vb" + + + + + +def _edge_labels(result: dict, relation: str) -> set[tuple[str, str]]: + labels = {node["id"]: node["label"] for node in result["nodes"]} + return { + (labels.get(edge["source"], edge["source"]), labels.get(edge["target"], edge["target"])) + for edge in result["edges"] + if edge["relation"] == relation + } + + +def test_vbnet_class_methods_and_case_insensitive_calls(tmp_path): + source = tmp_path / "Counter.vb" + source.write_text( + "Namespace Demo\n" + " Public Class Counter\n" + " Public Sub Run()\n" + " helper()\n" + " End Sub\n" + " Private Sub Helper()\n" + " End Sub\n" + " End Class\n" + "End Namespace\n", + encoding="utf-8", + ) + + result = extract([source], cache_root=tmp_path) + + labels = {node["label"] for node in result["nodes"]} + assert {"Demo", "Counter", "Run()", "Helper()"} <= labels + assert ("Run()", "Helper()") in _edge_labels(result, "calls") + + +def test_vbnet_types_members_relationships_and_partial_calls(tmp_path): + first = tmp_path / "Counter.vb" + first.write_text( + "Imports System.Text\n" + "Namespace Demo\n" + " Public Interface IWorker\n Sub Run()\n End Interface\n" + " Public Class BaseCounter\n End Class\n" + " Public Partial Class Counter\n" + " Inherits BaseCounter\n Implements IWorker\n" + " Private count As Integer\n" + " Public Event Changed As EventHandler\n" + " Public Property Value As Integer\n" + " Public Sub New()\n End Sub\n" + " Public Sub Run() Implements IWorker.Run Handles Me.Changed\n" + " hElPeR()\n End Sub\n" + " End Class\n" + " Public Structure Pair\n End Structure\n" + " Public Enum State\n OnValue\n OffValue\n End Enum\n" + "End Namespace\n", + encoding="utf-8", + ) + second = tmp_path / "Counter.Designer.vb" + second.write_text( + "Namespace Demo\n" + " Partial Public Class Counter\n" + " Private Sub Helper(Of T)(\n" + " Optional value As T = Nothing)\n" + " End Sub\n" + " End Class\n" + "End Namespace\n", + encoding="utf-8", + ) + project = tmp_path / "Demo.vbproj" + project.write_text( + '' + 'net8.0', + encoding="utf-8", + ) + + result = extract([first, second, project], cache_root=tmp_path) + + labels = {node["label"] for node in result["nodes"]} + assert { + "Demo", "IWorker", "BaseCounter", "Counter", "Pair", "State", + "OnValue", "OffValue", "count", "Changed", "Value", "New()", + "Run()", "Helper()", "System.Text", "net8.0", + } <= labels + assert ("Counter", "BaseCounter") in _edge_labels(result, "inherits") + assert ("Counter", "IWorker") in _edge_labels(result, "implements") + assert ("Run()", "Helper()") in _edge_labels(result, "calls") + assert ("Run()", "Changed") in _edge_labels(result, "handles") + + +def test_vbnet_fixture_uses_normal_extract_path(tmp_path): + result = extract([FIXTURE], cache_root=tmp_path) + + labels = {node["label"] for node in result["nodes"]} + assert {'Helper()', 'Fixtures', 'Run()', 'Sample'} <= labels + assert ('Run()', 'Helper()') in _edge_labels(result, "calls") + + +def test_vbnet_malformed_tail_comments_and_strings_do_not_create_phantoms(tmp_path): + source = tmp_path / 'Broken.vb' + source.write_text("Class Broken\n Sub Valid()\n End Sub\nEnd Class\n???\n' Sub Ghost()\n", encoding="utf-8") + + result = extract([source], cache_root=tmp_path) + + labels = {node["label"].casefold() for node in result["nodes"]} + assert 'valid()' in labels + assert labels.isdisjoint({'ghost()'}) + + +def test_vbnet_missing_parser_reports_install_hint(tmp_path, monkeypatch, capsys): + source = tmp_path / "missing.vb" + source.write_text('Class Missing\nEnd Class\n', encoding="utf-8") + monkeypatch.setitem(sys.modules, 'tree_sitter_vb_dotnet', None) + + result = extract([source], cache_root=tmp_path) + + assert result["nodes"] == [] + assert 'pip install "graphifyy[vbnet]"' in capsys.readouterr().err diff --git a/uv.lock b/uv.lock index fd04b1bea5..d42f94b286 100644 --- a/uv.lock +++ b/uv.lock @@ -1163,6 +1163,7 @@ all = [ { name = "tree-sitter-ocaml" }, { name = "tree-sitter-pascal" }, { name = "tree-sitter-sql" }, + { name = "tree-sitter-vb-dotnet" }, { name = "watchdog" }, { name = "yt-dlp" }, ] @@ -1246,6 +1247,9 @@ svg = [ terraform = [ { name = "tree-sitter-hcl" }, ] +vbnet = [ + { name = "tree-sitter-vb-dotnet" }, +] video = [ { name = "faster-whisper", marker = "python_full_version >= '3.11'" }, { name = "yt-dlp" }, @@ -1368,6 +1372,8 @@ requires-dist = [ { name = "tree-sitter-sql", marker = "extra == 'sql'" }, { name = "tree-sitter-swift", specifier = ">=0.7,<0.9" }, { name = "tree-sitter-typescript", specifier = ">=0.23,<0.25" }, + { name = "tree-sitter-vb-dotnet", marker = "extra == 'all'", specifier = "==0.3.0" }, + { name = "tree-sitter-vb-dotnet", marker = "extra == 'vbnet'", specifier = "==0.3.0" }, { name = "tree-sitter-verilog", specifier = ">=1.0,<2.0" }, { name = "tree-sitter-zig", specifier = ">=1.0,<2.0" }, { name = "watchdog", marker = "extra == 'all'" }, @@ -1375,7 +1381,7 @@ requires-dist = [ { name = "yt-dlp", marker = "extra == 'all'", specifier = ">=2026.7.4" }, { name = "yt-dlp", marker = "extra == 'video'", specifier = ">=2026.7.4" }, ] -provides-extras = ["mcp", "neo4j", "falkordb", "pdf", "watch", "svg", "leiden", "office", "google", "postgres", "video", "kimi", "ollama", "bedrock", "anthropic", "gemini", "openai", "chinese", "sql", "pascal", "dm", "terraform", "ocaml", "commonlisp", "robot", "all"] +provides-extras = ["mcp", "neo4j", "falkordb", "pdf", "watch", "svg", "leiden", "office", "google", "postgres", "video", "kimi", "ollama", "bedrock", "anthropic", "gemini", "openai", "chinese", "sql", "pascal", "dm", "terraform", "ocaml", "commonlisp", "robot", "vbnet", "all"] [package.metadata.requires-dev] dev = [ @@ -5042,6 +5048,17 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/9f/e4/81f9a935789233cf412a0ed5fe04c883841d2c8fb0b7e075958a35c65032/tree_sitter_typescript-0.23.2-cp39-abi3-win_arm64.whl", hash = "sha256:05db58f70b95ef0ea126db5560f3775692f609589ed6f8dd0af84b7f19f1cbb7", size = 274052, upload-time = "2024-11-11T02:36:09.514Z" }, ] +[[package]] +name = "tree-sitter-vb-dotnet" +version = "0.3.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/04/a1/95f99d291163cab128f86e6303268f50c37856bcf7dc361435f752549a48/tree_sitter_vb_dotnet-0.3.0.tar.gz", hash = "sha256:6e7178a8cf7c34b8ba07aa3914bef5475f4eafd348cdd393ff3ebfc7cb41ae61", size = 1049419, upload-time = "2026-09-04T10:33:34.645Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/68/22/d209d8bce9148d50e17190b8c3fb7909fa51d65ed41a83efb7d99f4aa180/tree_sitter_vb_dotnet-0.3.0-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:ad7dcf05170b429a07f05f78a2f0f5d3d46fc6174b8a7c90eec7cbfb980fca76", size = 326738, upload-time = "2026-09-04T10:33:30.602Z" }, + { url = "https://files.pythonhosted.org/packages/cd/69/d089c1f8e12ce743d41546650ca9a479bf99ea0eb74652ccd8939b198a4b/tree_sitter_vb_dotnet-0.3.0-cp310-abi3-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:7d2a2c6670c9f5917b3606e8a44045d1b508fc3258ede9f31a9b3a75f704dc99", size = 325760, upload-time = "2026-09-04T10:33:31.896Z" }, + { url = "https://files.pythonhosted.org/packages/c2/19/4edabceed5aaa9f0390bae115bd7a68f8e3b4fbe3c893c1e79bf060ec0a1/tree_sitter_vb_dotnet-0.3.0-cp310-abi3-win_amd64.whl", hash = "sha256:f5615797aa1c5d98f1906eeda5ad3a78667069bd51b78ddf143118aaff57801e", size = 307980, upload-time = "2026-09-04T10:33:33.195Z" }, +] + [[package]] name = "tree-sitter-verilog" version = "1.0.3"