diff --git a/.changeset/grep-exclude.md b/.changeset/grep-exclude.md new file mode 100644 index 00000000..38a3ab4c --- /dev/null +++ b/.changeset/grep-exclude.md @@ -0,0 +1,6 @@ +--- +"@cloudflare/dofs": patch +"@cloudflare/computer": patch +--- + +Add an `exclude` option to `fs.grep`, matching `fs.find`. diff --git a/docs/04_filesystem_interface.md b/docs/04_filesystem_interface.md index d5f6eaa5..4563fc16 100644 --- a/docs/04_filesystem_interface.md +++ b/docs/04_filesystem_interface.md @@ -392,6 +392,7 @@ interface GrepOptions { limit?: number; offset?: number; include?: string; + exclude?: string[]; } interface WorkspaceGrepContextLine { @@ -420,14 +421,23 @@ letter case. `context` adds that many lines before and after each match. `include` is a glob relative to a searched directory. `limit` and `offset` paginate matching lines. +`exclude` takes globs of the same shape as `find`'s, matched against the same +directory-relative path and applied before `include`, so an exclusion always +wins. An excluded directory is pruned before its children are queried, so the +subtree costs nothing rather than being read and filtered. As with `find`, name +both the directory and its contents to skip a whole subtree: `node_modules/**` +matches what is below `node_modules`, not `node_modules` itself. + `path` may be a directory or a single file. Directory searches return matches in deterministic depth-first discovery order, then line order within each -file. Results are not globally sorted by full path. +file. Results are not globally sorted by full path. A single-file search has no +traversal to prune, so `exclude` does not apply to it. ```ts const hits = await fs.grep("TODO", "/workspace/src", { ignoreCase: true, include: "**/*.ts", + exclude: ["node_modules", "node_modules/**"], }); for (const hit of hits) { console.log(`${hit.path}:${hit.line}: ${hit.text}`); @@ -517,7 +527,7 @@ maps to `Workspace.fs`: | `watch` | — | Low-level primitive in `fs/watch.ts` (`createWatcher`, `createWatchAsyncIterable`, `WatchHandle`, `WatchOptions`); not exposed on the `WorkspaceFilesystem` class. | | `open` / `FileHandle` | — | Use streams instead. | | `glob` | `find` | Limited glob support (`*`, `**`, `**/`, and `?`), plus `exclude` for pruning subtrees. | -| — | `grep` | Not in `node:fs`; literal by default, with optional regular expressions. | +| — | `grep` | Not in `node:fs`; literal by default, with optional regular expressions. Shares `find`'s `include`/`exclude` globs. | | — | `find` | Recursive directory walk with an optional glob, relative-rooted. | | — | `ls` | Flat list of file paths under a directory (segment-aware). | diff --git a/docs/09_tool_interface.md b/docs/09_tool_interface.md index 93bfb800..a8629a1c 100644 --- a/docs/09_tool_interface.md +++ b/docs/09_tool_interface.md @@ -185,6 +185,7 @@ The pattern is relative to `path`. `*` stays within one path segment, `**` cross path?: string; // default /workspace query: string; include?: string; // glob relative to path + exclude?: string[]; // globs pruned from the walk regex?: boolean; // default false ignoreCase?: boolean; // default false context?: number; // 0 through 10 @@ -195,9 +196,11 @@ The pattern is relative to `path`. `*` stays within one path segment, `**` cross The AI tool defaults to literal, case-sensitive matching. Set `regex: true` to interpret `query` as a regular expression and `ignoreCase: true` to ignore letter case. Matches include path, line number, text, and optional numbered context. Invalid regular expressions return a structured error. A non-final page includes `nextOffset`. -The tool passes `include`, `limit`, and `offset` through one `workspace.fs.grep` call. The storage search pages matching files and stops after the requested matches, so an included search does not build the full file or match list in the tool layer. Directory searches return matches in deterministic depth-first discovery order, then line order within each file. They are not globally sorted by full path. +`exclude` works exactly as it does on `find`: globs of the same shape, matched against the same relative path, applied before `include` so an exclusion always wins. An excluded directory is pruned before its children are queried rather than being read and filtered, so `exclude: ["node_modules", "node_modules/**"]` keeps the search out of a package tree. Name both forms, since `node_modules/**` matches what is below the directory rather than the directory itself. A single-file search has no traversal to prune, so `exclude` does not apply to it. -The lower-level `workspace.fs.grep` uses the same literal, case-sensitive defaults. Its options also accept `limit`, `offset`, `include`, `context`, `regex`, and `ignoreCase`. +The tool passes `include`, `exclude`, `limit`, and `offset` through one `workspace.fs.grep` call. The storage search pages matching files and stops after the requested matches, so an included search does not build the full file or match list in the tool layer. Directory searches return matches in deterministic depth-first discovery order, then line order within each file. They are not globally sorted by full path. + +The lower-level `workspace.fs.grep` uses the same literal, case-sensitive defaults. Its options also accept `limit`, `offset`, `include`, `exclude`, `context`, `regex`, and `ignoreCase`. ## `write` diff --git a/packages/computer/src/tools/fs/grep.ts b/packages/computer/src/tools/fs/grep.ts index 43f35cf8..430df6f2 100644 --- a/packages/computer/src/tools/fs/grep.ts +++ b/packages/computer/src/tools/fs/grep.ts @@ -21,6 +21,7 @@ interface GrepOptions { limit?: number; offset?: number; include?: string; + exclude?: string[]; } export interface GrepWorkspaceLike { @@ -43,6 +44,12 @@ const inputSchema = z.object({ .string() .optional() .describe('Glob relative to path that limits searched files, for example "**/*.ts".'), + exclude: z + .array(z.string()) + .optional() + .describe( + 'Glob patterns to leave out, for example ["node_modules/**", "**/.git/**"]. An excluded directory is skipped along with everything below it.', + ), regex: z.boolean().optional().describe("Interpret query as a regular expression."), ignoreCase: z.boolean().optional().describe("Ignore letter case."), context: z.number().int().min(0).max(10).optional(), @@ -55,7 +62,17 @@ export function createGrepTool(options: GrepToolOptions): Tool { + execute: async ({ + path, + query, + include, + exclude, + regex, + ignoreCase, + context, + limit, + offset, + }) => { try { const pageSize = limit ?? DEFAULT_LIMIT; const pageOffset = offset ?? 0; @@ -67,6 +84,7 @@ export function createGrepTool(options: GrepToolOptions): Tool { }); }); + describe("exclude", () => { + it("leaves an excluded file out of the results", async () => { + await withDB(async (db) => { + await writeFile(db, "/keep.ts", "TODO keep\n", {}, () => 0); + await writeFile(db, "/skip.ts", "TODO skip\n", {}, () => 0); + + expect( + (await grep(db, "TODO", "/", { exclude: ["skip.ts"] })).map((match) => match.path), + ).toEqual(["/keep.ts"]); + }); + }); + + it("drops an excluded directory along with everything below it", async () => { + await withDB(async (db) => { + mkdir(db, "/src", { recursive: true }, () => 0); + mkdir(db, "/node_modules/dep", { recursive: true }, () => 0); + await writeFile(db, "/src/index.ts", "TODO mine\n", {}, () => 0); + await writeFile(db, "/node_modules/dep/index.ts", "TODO theirs\n", {}, () => 0); + + // Both forms, as the find tests do: `node_modules` prunes the + // directory and `node_modules/**` covers anything below it. + expect( + ( + await grep(db, "TODO", "/", { + exclude: ["node_modules", "node_modules/**"], + }) + ).map((match) => match.path), + ).toEqual(["/src/index.ts"]); + }); + }); + + it("never reads a file below an excluded directory", async () => { + await withDB(async (db) => { + mkdir(db, "/src", { recursive: true }, () => 0); + mkdir(db, "/vendor", { recursive: true }, () => 0); + await writeFile(db, "/src/a.ts", "TODO mine\n", {}, () => 0); + await writeFile(db, "/vendor/b.ts", "TODO theirs\n", {}, () => 0); + + // Pruning has to happen during the walk, not as a filter over + // results: the whole point is that the excluded subtree costs + // nothing. If the walker descended and grep then discarded the + // matches, the blob for b.ts would still be queried. + // + // The directory itself must be named to be pruned -- `vendor/**` + // matches what is *below* `vendor`, not `vendor` -- which is why the + // existing find tests pass both forms. With only `vendor/**` the + // walker still descends and excludes each child, costing a query. + const seen: string[] = []; + const all = db.all.bind(db); + vi.spyOn(db, "all").mockImplementation((query: unknown, ...args: unknown[]) => { + seen.push(String(query)); + return all(query as never, ...(args as never[])); + }); + + await grep(db, "TODO", "/", { exclude: ["vendor", "vendor/**"] }); + vi.restoreAllMocks(); + + // The walker reads children of / and of /src, but never of /vendor. + const childQueries = seen.filter((query) => query.includes("d.name > ?")); + expect(childQueries.length).toBe(2); + }); + }); + + it("applies exclusion before the inclusion glob", async () => { + await withDB(async (db) => { + mkdir(db, "/pkg", { recursive: true }, () => 0); + await writeFile(db, "/pkg/keep.ts", "TODO keep\n", {}, () => 0); + await writeFile(db, "/pkg/skip.ts", "TODO skip\n", {}, () => 0); + + expect( + ( + await grep(db, "TODO", "/", { + include: "**/*.ts", + exclude: ["pkg/skip.ts"], + }) + ).map((match) => match.path), + ).toEqual(["/pkg/keep.ts"]); + }); + }); + + it("ignores exclusion when the path names a single file", async () => { + await withDB(async (db) => { + await writeFile(db, "/only.ts", "TODO here\n", {}, () => 0); + + // The caller named the file, so there is no traversal to prune and + // nothing to second-guess. + expect( + (await grep(db, "TODO", "/only.ts", { exclude: ["only.ts"] })).map((match) => match.path), + ).toEqual(["/only.ts"]); + }); + }); + }); + it("walks each directory page once during a search", async () => { await withDB(async (db) => { for (let index = 0; index < 260; index += 1) { diff --git a/packages/dofs/src/fs/grep.ts b/packages/dofs/src/fs/grep.ts index 2eae9392..b2132cf0 100644 --- a/packages/dofs/src/fs/grep.ts +++ b/packages/dofs/src/fs/grep.ts @@ -31,6 +31,13 @@ export interface GrepOptions { offset?: number; /** Glob relative to a searched directory that limits files. */ include?: string; + /** + * Glob patterns whose matches are not searched. Matched against the + * same directory-relative path as the inclusion glob and applied + * first, so an exclusion always wins. An excluded directory is + * pruned: neither it nor anything below it is read. + */ + exclude?: string[]; } interface ScanState { @@ -68,7 +75,12 @@ export async function grep( }); const matches: WorkspaceGrepMatch[] = []; const state: ScanState = { seen: 0, accepted: 0 }; - const filePaths = node.type === "file" ? [canonical] : filesUnder(db, canonical, options.include); + // Grepping a single file has no traversal to prune, so `exclude` does not + // apply to it: the caller named the file explicitly. + const filePaths = + node.type === "file" + ? [canonical] + : filesUnder(db, canonical, options.include, options.exclude); for (const filePath of filePaths) { const complete = await scanFile( db, @@ -117,8 +129,12 @@ function* filesUnder( db: Database, directory: string, include: string | undefined, + exclude: string[] | undefined, ): Iterable { - for (const entry of iterateFoundEntries(db, directory, include)) { + // The find walker already prunes excluded directories before querying their + // children, so an excluded subtree costs nothing here rather than being + // walked and filtered. + for (const entry of iterateFoundEntries(db, directory, include, exclude)) { if (entry.type === "file") yield entry.path; } }