diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index 2a30b5b..8f6dd3c 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -3,7 +3,7 @@ "name": "codebase-index", "displayName": "Codebase Index", "description": "Give Claude a precise local map of your codebase: find implementations, trace behavior, and predict change impact with file-line evidence.", - "version": "2.0.0", + "version": "2.1.0", "author": { "name": "codebase-index contributors" }, diff --git a/.claude/skills/codebase-index/.skill_version b/.claude/skills/codebase-index/.skill_version index 227cea2..7ec1d6d 100644 --- a/.claude/skills/codebase-index/.skill_version +++ b/.claude/skills/codebase-index/.skill_version @@ -1 +1 @@ -2.0.0 +2.1.0 diff --git a/.claude/skills/codebase-index/SKILL.md b/.claude/skills/codebase-index/SKILL.md index 432005b..ba39234 100644 --- a/.claude/skills/codebase-index/SKILL.md +++ b/.claude/skills/codebase-index/SKILL.md @@ -6,106 +6,60 @@ allowed-tools: Bash(codebase-index search *), Bash(codebase-index explain *), Ba # Codebase Index -Use the local index before reading repository files. +Use the local index before reading repository files. It answers with ranked, +numbered lines, so most answers need no Read and no Grep at all. -The operating principle is **Find → Trace → Verify → Predict**: - -- **Find** the implementation with ranked retrieval. -- **Trace** behavior through definitions, callers, dependencies, and paths. -- **Verify** that evidence you already hold is still true before relying on it. -- **Predict** change impact while preserving an explicit evidence trail. - -## Route the question +## Route the question — always `--compact` | Intent | Command | |---|---| -| Where is X implemented? | `codebase-index search "X" --session --json` | -| How does X work? | `codebase-index explain "X" --session --json` | -| What is this codebase? | `codebase-index architecture --json` | -| Find a named symbol | `codebase-index symbol "X" --json` | -| Who calls or references X? | `codebase-index refs "X" --json` | -| What changes if X changes? | `codebase-index impact "X" --json` | -| What does my current diff affect? | `codebase-index diff-impact --json` | -| How are X and Y connected? | `codebase-index path "X" "Y" --json` | -| Describe X and its neighborhood | `codebase-index describe "X" --json` | +| Where / how does X work? | `codebase-index search "X" --compact --session ` | +| Overview of a feature | `codebase-index explain "X" --compact --session ` | +| Every caller / call site of X | `codebase-index refs "Owner.member" --compact` | +| What breaks if X changes (incl. other modules) | `codebase-index impact "Owner.member" --compact` | +| Find a definition | `codebase-index symbol "X"` | +| A class: members, who uses it | `codebase-index describe "Class" --json` | +| What does my diff affect? | `codebase-index diff-impact --json` | | Is what I read earlier still true? | `codebase-index verify --session --json` | -| Produce a human graph | `codebase-index graph "X" --output ` | - -Use `search --mode symbol` for exact symbol work, `--mode fts` for text and -error messages, and the default `hybrid` mode for mixed questions. Use pure -`vector` mode only when embeddings are enabled and exact vocabulary is unknown. - -Read [references/commands.md](references/commands.md) only when command options -or routing remain unclear. - -## Evidence protocol - -1. Pick one session tag for this conversation (for example `auth-fix-1`) and - pass `--session ` to every `search` and `explain`. -2. Run the best-matching command with `--json`. -3. Check `index` before trusting the payload: - - missing → run `codebase-index index`, then repeat; - - stale with fewer than 20 changed files → run `codebase-index update`; - - stale with 20 or more changed files → run `codebase-index index`; - - fresh → continue. -4. Start with ranks 1–3. Read only `recommended_reads` line ranges. -5. Trace one additional hop only when the question requires behavior, - ownership, or impact. -6. Before answering or editing from evidence gathered earlier in the task, run - `codebase-index verify --session --json` and reread anything whose - state is not `valid` or `relocated`. -7. Answer with `file:line` evidence and state uncertainty explicitly. - -Do not open whole files when a line range is available. A snippet may already -be sufficient. `skeletonized: true` means the response intentionally folded -unrelated body lines; read the supplied range when the missing body matters. - -## Evidence memory - -- `reused: true` with `snippet: null` — this session already received that - exact text and its source is unchanged. Use your earlier copy; if you can no - longer see it, Read the range. -- `memory.invalidated` — evidence this session received has changed since. - Treat your earlier copy as wrong and reread before relying on it. -- `stale: true` — the index is older than the file. Run `codebase-index update` - or Read the range. -- A tag belongs to one context. Never give it to a subagent or another - conversation. Start a new tag after the context is cleared or compacted, or - whenever earlier snippets are no longer visible to you. - -Verdict states and citing evidence in notes: [references/memory.md](references/memory.md). - -## Confidence contract - -- **high** — answer from the indexed evidence. -- **medium** — read the recommended ranges and confirm the key claim with one - targeted lookup if necessary. -- **low** or no results — follow `fallback_suggestions`, then use a narrow - Grep/Glob fallback. - -On `refs` and `impact`, inspect `coverage`. If `coverage.partial` is true, an -empty result is inconclusive; confirm with targeted Grep before saying that -nothing references the target. - -Edges carry `confidence`: - -- `extracted` — exact parser evidence; -- `inferred` — heuristic resolution; -- `ambiguous` — unresolved or non-unique. - -Never present an inferred or ambiguous chain as certain. - -## Answer contract - -Structure repository answers around: - -1. **Answer** — the direct conclusion. -2. **Evidence** — the minimum supporting `file:line` references. -3. **Confidence** — only when evidence is partial, inferred, stale, or missing. -4. **Next check** — only when another check would materially reduce uncertainty. - -Do not narrate every search step. Do not claim absence from a partial graph. -Do not replace evidence with a generated HTML graph. -For payload fields and failure handling, read -[references/response-contract.md](references/response-contract.md). +`--compact` prints `path:start-end symbol` per result with the matching lines +numbered underneath (` 24| public static final String FILE_ID = ...`). Cite +those lines as `path:24` directly. Read a range only when the lines shown do not +answer the question — and then only that range, never the whole file. + +Name members as `Owner.member` (`TownService.refresh`, `activation::place`, +`CoreError::ObjectDamaged`) so same-named methods of other types are excluded. +`refs --compact` lines are `path:line kind caller -> target`: the list of call +sites with the calling function is usually the whole answer. Add +`--exclude-tests` for production-only lists and `--path ` for one module. +`kind reference` is a non-call use, e.g. an enum variant matched in a `match` arm. + +`impact --compact` starts with `# module : build files depending on it: ...` +— the answer to "does another module depend on this", with no build-file grep. + +## Protocol + +1. Pick one session tag per conversation (e.g. `auth-fix-1`); pass it to + `search`/`explain`. Results already sent in this session print as + `(already sent)` — use your earlier copy. +2. A header saying `index stale` → run `codebase-index update` once; `NO INDEX` + → `codebase-index index`. +3. Batch independent questions into one Bash call (`cmd1; echo ---; cmd2`). +4. `# partial:` on refs/impact means the list may be incomplete. For + `Owner.member` it lists `possible_call` sites (calls on a variable the index + cannot type), nearest first: check those few lines, do not grep the repo. +5. Before relying on evidence from earlier in a long task, run + `codebase-index verify --session --json`. +6. Grep only when the index returns nothing relevant or for non-code text. + +Edge confidence: `extracted` exact, `inferred` heuristic (receiver matched a +type), `ambiguous` unresolved. Never present an inferred chain as certain. + +## Answer + +Lead with the answer, then the minimum `file:line` evidence. State uncertainty +only when evidence is partial, inferred or stale. + +Options and JSON fields: [references/commands.md](references/commands.md), +[references/response-contract.md](references/response-contract.md), +session memory: [references/memory.md](references/memory.md). diff --git a/.claude/skills/codebase-index/references/commands.md b/.claude/skills/codebase-index/references/commands.md index 79c91a6..197ffe7 100644 --- a/.claude/skills/codebase-index/references/commands.md +++ b/.claude/skills/codebase-index/references/commands.md @@ -37,6 +37,24 @@ codebase-index describe "" --json - `architecture` reads module analysis cached at index time. - `refs` finds definitions, calls, and graph-backed references. + For a method whose name several types share, pass `Owner.member` + (`TownService.refresh`, or `module::fn` for a top-level function); each site + names its `caller` and the `target` it resolved to. Calls the index cannot type + (`chest.holdings().take(..)`) come back as `possible_call`, nearest first, with + `coverage.partial: true`: read those before claiming a complete list. + `impact` and `symbol` accept the same form. + Filters: `--exclude-tests`, `--path ` (repeatable). `--compact` prints + `path:line kind caller -> target [confidence]`, one site per line. +- `search`, `explain` and `impact` also take `--compact`: agent text instead of + JSON. For search/explain, each result is `path:start-end symbols` followed by up + to eight numbered lines that carry the match (rarer query words first); + evidence already sent to the session prints as `(already sent)`. For impact, + one `d path:line name via edge` line per node, after a + `# module ...` line naming the build files that depend on the target's module. +- `symbol` returns exact matches alone when there are any and reports how many + prefix matches it left out (`more_prefix_matches`); `--exact` drops them. +- `describe` on a class, enum, struct or trait also lists its `members` and + folds their edges into `used_by` / `uses` (code outside the type). - `impact` walks dependents (`up`), dependencies (`down`), or both. - `diff-impact` aggregates impact for tracked changes relative to a verified Git commit; new or excluded files are reported as unresolved. diff --git a/.codex/skills/codebase-index/.skill_version b/.codex/skills/codebase-index/.skill_version index 227cea2..7ec1d6d 100644 --- a/.codex/skills/codebase-index/.skill_version +++ b/.codex/skills/codebase-index/.skill_version @@ -1 +1 @@ -2.0.0 +2.1.0 diff --git a/.codex/skills/codebase-index/SKILL.md b/.codex/skills/codebase-index/SKILL.md index 432005b..ba39234 100644 --- a/.codex/skills/codebase-index/SKILL.md +++ b/.codex/skills/codebase-index/SKILL.md @@ -6,106 +6,60 @@ allowed-tools: Bash(codebase-index search *), Bash(codebase-index explain *), Ba # Codebase Index -Use the local index before reading repository files. +Use the local index before reading repository files. It answers with ranked, +numbered lines, so most answers need no Read and no Grep at all. -The operating principle is **Find → Trace → Verify → Predict**: - -- **Find** the implementation with ranked retrieval. -- **Trace** behavior through definitions, callers, dependencies, and paths. -- **Verify** that evidence you already hold is still true before relying on it. -- **Predict** change impact while preserving an explicit evidence trail. - -## Route the question +## Route the question — always `--compact` | Intent | Command | |---|---| -| Where is X implemented? | `codebase-index search "X" --session --json` | -| How does X work? | `codebase-index explain "X" --session --json` | -| What is this codebase? | `codebase-index architecture --json` | -| Find a named symbol | `codebase-index symbol "X" --json` | -| Who calls or references X? | `codebase-index refs "X" --json` | -| What changes if X changes? | `codebase-index impact "X" --json` | -| What does my current diff affect? | `codebase-index diff-impact --json` | -| How are X and Y connected? | `codebase-index path "X" "Y" --json` | -| Describe X and its neighborhood | `codebase-index describe "X" --json` | +| Where / how does X work? | `codebase-index search "X" --compact --session ` | +| Overview of a feature | `codebase-index explain "X" --compact --session ` | +| Every caller / call site of X | `codebase-index refs "Owner.member" --compact` | +| What breaks if X changes (incl. other modules) | `codebase-index impact "Owner.member" --compact` | +| Find a definition | `codebase-index symbol "X"` | +| A class: members, who uses it | `codebase-index describe "Class" --json` | +| What does my diff affect? | `codebase-index diff-impact --json` | | Is what I read earlier still true? | `codebase-index verify --session --json` | -| Produce a human graph | `codebase-index graph "X" --output ` | - -Use `search --mode symbol` for exact symbol work, `--mode fts` for text and -error messages, and the default `hybrid` mode for mixed questions. Use pure -`vector` mode only when embeddings are enabled and exact vocabulary is unknown. - -Read [references/commands.md](references/commands.md) only when command options -or routing remain unclear. - -## Evidence protocol - -1. Pick one session tag for this conversation (for example `auth-fix-1`) and - pass `--session ` to every `search` and `explain`. -2. Run the best-matching command with `--json`. -3. Check `index` before trusting the payload: - - missing → run `codebase-index index`, then repeat; - - stale with fewer than 20 changed files → run `codebase-index update`; - - stale with 20 or more changed files → run `codebase-index index`; - - fresh → continue. -4. Start with ranks 1–3. Read only `recommended_reads` line ranges. -5. Trace one additional hop only when the question requires behavior, - ownership, or impact. -6. Before answering or editing from evidence gathered earlier in the task, run - `codebase-index verify --session --json` and reread anything whose - state is not `valid` or `relocated`. -7. Answer with `file:line` evidence and state uncertainty explicitly. - -Do not open whole files when a line range is available. A snippet may already -be sufficient. `skeletonized: true` means the response intentionally folded -unrelated body lines; read the supplied range when the missing body matters. - -## Evidence memory - -- `reused: true` with `snippet: null` — this session already received that - exact text and its source is unchanged. Use your earlier copy; if you can no - longer see it, Read the range. -- `memory.invalidated` — evidence this session received has changed since. - Treat your earlier copy as wrong and reread before relying on it. -- `stale: true` — the index is older than the file. Run `codebase-index update` - or Read the range. -- A tag belongs to one context. Never give it to a subagent or another - conversation. Start a new tag after the context is cleared or compacted, or - whenever earlier snippets are no longer visible to you. - -Verdict states and citing evidence in notes: [references/memory.md](references/memory.md). - -## Confidence contract - -- **high** — answer from the indexed evidence. -- **medium** — read the recommended ranges and confirm the key claim with one - targeted lookup if necessary. -- **low** or no results — follow `fallback_suggestions`, then use a narrow - Grep/Glob fallback. - -On `refs` and `impact`, inspect `coverage`. If `coverage.partial` is true, an -empty result is inconclusive; confirm with targeted Grep before saying that -nothing references the target. - -Edges carry `confidence`: - -- `extracted` — exact parser evidence; -- `inferred` — heuristic resolution; -- `ambiguous` — unresolved or non-unique. - -Never present an inferred or ambiguous chain as certain. - -## Answer contract - -Structure repository answers around: - -1. **Answer** — the direct conclusion. -2. **Evidence** — the minimum supporting `file:line` references. -3. **Confidence** — only when evidence is partial, inferred, stale, or missing. -4. **Next check** — only when another check would materially reduce uncertainty. - -Do not narrate every search step. Do not claim absence from a partial graph. -Do not replace evidence with a generated HTML graph. -For payload fields and failure handling, read -[references/response-contract.md](references/response-contract.md). +`--compact` prints `path:start-end symbol` per result with the matching lines +numbered underneath (` 24| public static final String FILE_ID = ...`). Cite +those lines as `path:24` directly. Read a range only when the lines shown do not +answer the question — and then only that range, never the whole file. + +Name members as `Owner.member` (`TownService.refresh`, `activation::place`, +`CoreError::ObjectDamaged`) so same-named methods of other types are excluded. +`refs --compact` lines are `path:line kind caller -> target`: the list of call +sites with the calling function is usually the whole answer. Add +`--exclude-tests` for production-only lists and `--path ` for one module. +`kind reference` is a non-call use, e.g. an enum variant matched in a `match` arm. + +`impact --compact` starts with `# module : build files depending on it: ...` +— the answer to "does another module depend on this", with no build-file grep. + +## Protocol + +1. Pick one session tag per conversation (e.g. `auth-fix-1`); pass it to + `search`/`explain`. Results already sent in this session print as + `(already sent)` — use your earlier copy. +2. A header saying `index stale` → run `codebase-index update` once; `NO INDEX` + → `codebase-index index`. +3. Batch independent questions into one Bash call (`cmd1; echo ---; cmd2`). +4. `# partial:` on refs/impact means the list may be incomplete. For + `Owner.member` it lists `possible_call` sites (calls on a variable the index + cannot type), nearest first: check those few lines, do not grep the repo. +5. Before relying on evidence from earlier in a long task, run + `codebase-index verify --session --json`. +6. Grep only when the index returns nothing relevant or for non-code text. + +Edge confidence: `extracted` exact, `inferred` heuristic (receiver matched a +type), `ambiguous` unresolved. Never present an inferred chain as certain. + +## Answer + +Lead with the answer, then the minimum `file:line` evidence. State uncertainty +only when evidence is partial, inferred or stale. + +Options and JSON fields: [references/commands.md](references/commands.md), +[references/response-contract.md](references/response-contract.md), +session memory: [references/memory.md](references/memory.md). diff --git a/.codex/skills/codebase-index/references/commands.md b/.codex/skills/codebase-index/references/commands.md index 79c91a6..197ffe7 100644 --- a/.codex/skills/codebase-index/references/commands.md +++ b/.codex/skills/codebase-index/references/commands.md @@ -37,6 +37,24 @@ codebase-index describe "" --json - `architecture` reads module analysis cached at index time. - `refs` finds definitions, calls, and graph-backed references. + For a method whose name several types share, pass `Owner.member` + (`TownService.refresh`, or `module::fn` for a top-level function); each site + names its `caller` and the `target` it resolved to. Calls the index cannot type + (`chest.holdings().take(..)`) come back as `possible_call`, nearest first, with + `coverage.partial: true`: read those before claiming a complete list. + `impact` and `symbol` accept the same form. + Filters: `--exclude-tests`, `--path ` (repeatable). `--compact` prints + `path:line kind caller -> target [confidence]`, one site per line. +- `search`, `explain` and `impact` also take `--compact`: agent text instead of + JSON. For search/explain, each result is `path:start-end symbols` followed by up + to eight numbered lines that carry the match (rarer query words first); + evidence already sent to the session prints as `(already sent)`. For impact, + one `d path:line name via edge` line per node, after a + `# module ...` line naming the build files that depend on the target's module. +- `symbol` returns exact matches alone when there are any and reports how many + prefix matches it left out (`more_prefix_matches`); `--exact` drops them. +- `describe` on a class, enum, struct or trait also lists its `members` and + folds their edges into `used_by` / `uses` (code outside the type). - `impact` walks dependents (`up`), dependencies (`down`), or both. - `diff-impact` aggregates impact for tracked changes relative to a verified Git commit; new or excluded files are reported as unresolved. diff --git a/.opencode/skills/codebase-index/.skill_version b/.opencode/skills/codebase-index/.skill_version index 227cea2..7ec1d6d 100644 --- a/.opencode/skills/codebase-index/.skill_version +++ b/.opencode/skills/codebase-index/.skill_version @@ -1 +1 @@ -2.0.0 +2.1.0 diff --git a/.opencode/skills/codebase-index/SKILL.md b/.opencode/skills/codebase-index/SKILL.md index 432005b..ba39234 100644 --- a/.opencode/skills/codebase-index/SKILL.md +++ b/.opencode/skills/codebase-index/SKILL.md @@ -6,106 +6,60 @@ allowed-tools: Bash(codebase-index search *), Bash(codebase-index explain *), Ba # Codebase Index -Use the local index before reading repository files. +Use the local index before reading repository files. It answers with ranked, +numbered lines, so most answers need no Read and no Grep at all. -The operating principle is **Find → Trace → Verify → Predict**: - -- **Find** the implementation with ranked retrieval. -- **Trace** behavior through definitions, callers, dependencies, and paths. -- **Verify** that evidence you already hold is still true before relying on it. -- **Predict** change impact while preserving an explicit evidence trail. - -## Route the question +## Route the question — always `--compact` | Intent | Command | |---|---| -| Where is X implemented? | `codebase-index search "X" --session --json` | -| How does X work? | `codebase-index explain "X" --session --json` | -| What is this codebase? | `codebase-index architecture --json` | -| Find a named symbol | `codebase-index symbol "X" --json` | -| Who calls or references X? | `codebase-index refs "X" --json` | -| What changes if X changes? | `codebase-index impact "X" --json` | -| What does my current diff affect? | `codebase-index diff-impact --json` | -| How are X and Y connected? | `codebase-index path "X" "Y" --json` | -| Describe X and its neighborhood | `codebase-index describe "X" --json` | +| Where / how does X work? | `codebase-index search "X" --compact --session ` | +| Overview of a feature | `codebase-index explain "X" --compact --session ` | +| Every caller / call site of X | `codebase-index refs "Owner.member" --compact` | +| What breaks if X changes (incl. other modules) | `codebase-index impact "Owner.member" --compact` | +| Find a definition | `codebase-index symbol "X"` | +| A class: members, who uses it | `codebase-index describe "Class" --json` | +| What does my diff affect? | `codebase-index diff-impact --json` | | Is what I read earlier still true? | `codebase-index verify --session --json` | -| Produce a human graph | `codebase-index graph "X" --output ` | - -Use `search --mode symbol` for exact symbol work, `--mode fts` for text and -error messages, and the default `hybrid` mode for mixed questions. Use pure -`vector` mode only when embeddings are enabled and exact vocabulary is unknown. - -Read [references/commands.md](references/commands.md) only when command options -or routing remain unclear. - -## Evidence protocol - -1. Pick one session tag for this conversation (for example `auth-fix-1`) and - pass `--session ` to every `search` and `explain`. -2. Run the best-matching command with `--json`. -3. Check `index` before trusting the payload: - - missing → run `codebase-index index`, then repeat; - - stale with fewer than 20 changed files → run `codebase-index update`; - - stale with 20 or more changed files → run `codebase-index index`; - - fresh → continue. -4. Start with ranks 1–3. Read only `recommended_reads` line ranges. -5. Trace one additional hop only when the question requires behavior, - ownership, or impact. -6. Before answering or editing from evidence gathered earlier in the task, run - `codebase-index verify --session --json` and reread anything whose - state is not `valid` or `relocated`. -7. Answer with `file:line` evidence and state uncertainty explicitly. - -Do not open whole files when a line range is available. A snippet may already -be sufficient. `skeletonized: true` means the response intentionally folded -unrelated body lines; read the supplied range when the missing body matters. - -## Evidence memory - -- `reused: true` with `snippet: null` — this session already received that - exact text and its source is unchanged. Use your earlier copy; if you can no - longer see it, Read the range. -- `memory.invalidated` — evidence this session received has changed since. - Treat your earlier copy as wrong and reread before relying on it. -- `stale: true` — the index is older than the file. Run `codebase-index update` - or Read the range. -- A tag belongs to one context. Never give it to a subagent or another - conversation. Start a new tag after the context is cleared or compacted, or - whenever earlier snippets are no longer visible to you. - -Verdict states and citing evidence in notes: [references/memory.md](references/memory.md). - -## Confidence contract - -- **high** — answer from the indexed evidence. -- **medium** — read the recommended ranges and confirm the key claim with one - targeted lookup if necessary. -- **low** or no results — follow `fallback_suggestions`, then use a narrow - Grep/Glob fallback. - -On `refs` and `impact`, inspect `coverage`. If `coverage.partial` is true, an -empty result is inconclusive; confirm with targeted Grep before saying that -nothing references the target. - -Edges carry `confidence`: - -- `extracted` — exact parser evidence; -- `inferred` — heuristic resolution; -- `ambiguous` — unresolved or non-unique. - -Never present an inferred or ambiguous chain as certain. - -## Answer contract - -Structure repository answers around: - -1. **Answer** — the direct conclusion. -2. **Evidence** — the minimum supporting `file:line` references. -3. **Confidence** — only when evidence is partial, inferred, stale, or missing. -4. **Next check** — only when another check would materially reduce uncertainty. - -Do not narrate every search step. Do not claim absence from a partial graph. -Do not replace evidence with a generated HTML graph. -For payload fields and failure handling, read -[references/response-contract.md](references/response-contract.md). +`--compact` prints `path:start-end symbol` per result with the matching lines +numbered underneath (` 24| public static final String FILE_ID = ...`). Cite +those lines as `path:24` directly. Read a range only when the lines shown do not +answer the question — and then only that range, never the whole file. + +Name members as `Owner.member` (`TownService.refresh`, `activation::place`, +`CoreError::ObjectDamaged`) so same-named methods of other types are excluded. +`refs --compact` lines are `path:line kind caller -> target`: the list of call +sites with the calling function is usually the whole answer. Add +`--exclude-tests` for production-only lists and `--path ` for one module. +`kind reference` is a non-call use, e.g. an enum variant matched in a `match` arm. + +`impact --compact` starts with `# module : build files depending on it: ...` +— the answer to "does another module depend on this", with no build-file grep. + +## Protocol + +1. Pick one session tag per conversation (e.g. `auth-fix-1`); pass it to + `search`/`explain`. Results already sent in this session print as + `(already sent)` — use your earlier copy. +2. A header saying `index stale` → run `codebase-index update` once; `NO INDEX` + → `codebase-index index`. +3. Batch independent questions into one Bash call (`cmd1; echo ---; cmd2`). +4. `# partial:` on refs/impact means the list may be incomplete. For + `Owner.member` it lists `possible_call` sites (calls on a variable the index + cannot type), nearest first: check those few lines, do not grep the repo. +5. Before relying on evidence from earlier in a long task, run + `codebase-index verify --session --json`. +6. Grep only when the index returns nothing relevant or for non-code text. + +Edge confidence: `extracted` exact, `inferred` heuristic (receiver matched a +type), `ambiguous` unresolved. Never present an inferred chain as certain. + +## Answer + +Lead with the answer, then the minimum `file:line` evidence. State uncertainty +only when evidence is partial, inferred or stale. + +Options and JSON fields: [references/commands.md](references/commands.md), +[references/response-contract.md](references/response-contract.md), +session memory: [references/memory.md](references/memory.md). diff --git a/.opencode/skills/codebase-index/references/commands.md b/.opencode/skills/codebase-index/references/commands.md index 79c91a6..197ffe7 100644 --- a/.opencode/skills/codebase-index/references/commands.md +++ b/.opencode/skills/codebase-index/references/commands.md @@ -37,6 +37,24 @@ codebase-index describe "" --json - `architecture` reads module analysis cached at index time. - `refs` finds definitions, calls, and graph-backed references. + For a method whose name several types share, pass `Owner.member` + (`TownService.refresh`, or `module::fn` for a top-level function); each site + names its `caller` and the `target` it resolved to. Calls the index cannot type + (`chest.holdings().take(..)`) come back as `possible_call`, nearest first, with + `coverage.partial: true`: read those before claiming a complete list. + `impact` and `symbol` accept the same form. + Filters: `--exclude-tests`, `--path ` (repeatable). `--compact` prints + `path:line kind caller -> target [confidence]`, one site per line. +- `search`, `explain` and `impact` also take `--compact`: agent text instead of + JSON. For search/explain, each result is `path:start-end symbols` followed by up + to eight numbered lines that carry the match (rarer query words first); + evidence already sent to the session prints as `(already sent)`. For impact, + one `d path:line name via edge` line per node, after a + `# module ...` line naming the build files that depend on the target's module. +- `symbol` returns exact matches alone when there are any and reports how many + prefix matches it left out (`more_prefix_matches`); `--exact` drops them. +- `describe` on a class, enum, struct or trait also lists its `members` and + folds their edges into `used_by` / `uses` (code outside the type). - `impact` walks dependents (`up`), dependencies (`down`), or both. - `diff-impact` aggregates impact for tracked changes relative to a verified Git commit; new or excluded files are reported as unresolved. diff --git a/CHANGELOG.md b/CHANGELOG.md index 1e6a1f1..442ed6c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,101 @@ All notable changes to this project are documented here. The format is based on ## [Unreleased] +## [2.1.0] - 2026-09-24 + +Agent-efficiency release. An agent answering five code questions on a 5.9k-file +Java/Rust monorepo spent **19% fewer tokens (83.5k vs 103.0k) and 25% fewer tool calls +(9 vs 12)** with the skill than with Grep/Read alone, three runs per arm, with no +overlap between the arms' ranges; before this release the two were at parity. The +gain comes from answer-ready output (`--compact`: ranked files with numbered matching +lines), `refs`/`impact` that resolve `Owner.member` targets exactly and say when a list +may be incomplete, build-module awareness in `impact`, and ranking measured over 342 +queries (MRR +0.010, p=0.031). Index schema 4: the first `update` rebuilds the index +once. + +### Fixed + +- `refs`, `impact` and `symbol` accept a qualified `Owner.member` target + (`TownService.refresh`, also `pkg.Owner.member`, `Owner::member`, `Owner#member`), + where the owner is a type or a module (`activation::place`). Before, a qualified name + matched nothing and a bare name that several types share (`refresh`) returned every + same-named method's callers mixed together. +- Calls written against a type or module (`TownService.refresh(server)`, `Foo::new()`, + `activation::place(..)`) now resolve across files when exactly one type or module of + that name defines the callee, so `impact` walks them. These edges are marked + `inferred`. +- Call resolution no longer invents edges: a call never binds across languages (a + Python `queue.take()` to a Java `take`), a call on another type + (`ApprenticeService.take`) no longer binds to the repo's only `take` when that one + belongs to a different type, and a call on a type the file does not define no + longer binds to a same-named method in the calling file. Name uniqueness is now + judged per language family (JVM, JS/TS, C/C++), so a Rust `place` resolves even + when Java also defines one. +- `refs "Owner.member"` no longer reports a confident empty answer when calls it + cannot type exist (`chest.holdings().take(..)`): they are listed as + `possible_call`, nearest to the definition first, and `coverage.partial` is true. +- `refs` and `impact` report `coverage.partial: true` with a reason when the target + matches no indexed definition or call site, instead of an empty answer that looked + like "no callers". +- `update` parses changed files in a process pool, like a full build. It used to parse + them one at a time, so a large change set was slower to update than to rebuild + (2009 changed files: 149 s, against 13 s for a full build of all 5904). +- A class is returned instead of its same-named constructor. The constructor's + `new X()` callers gave it the higher in-degree, so it took the file's one slot on + the page with a one-line body while the class never appeared. +- Kotlin call sites are extracted (with their receiver). Kotlin's `call_expression` + has no named fields, so no Kotlin call edge was ever recorded. + +### Added + +- Rust enum variants, Java enum constants and C# enum members are indexed as + `variant` symbols (`CoreError.ObjectDamaged`), and a Rust `Owner::name` path used in a + pattern or as a value (`Err(CoreError::ObjectDamaged(_)) =>`, `if let Kind::A = ..`) + is recorded as a `reference` edge. `refs` reports these as `kind: "reference"`, so it + finds where a variant is handled, not only where it is constructed. +- `refs --exclude-tests`, `refs --path ` (repeatable; also on the MCP + `find_refs` tool) and `refs --compact`, which prints one line per site: + `path:line kind caller -> target [confidence]`. +- `describe` on a class, enum, struct or trait lists its `members` and folds their + edges from and to code outside the type into `used_by` / `uses`. It used to report + zero callers and callees for a class, whose methods, not the class, are called. +- `--session` is accepted by `symbol`, `refs` and `impact`, so one tag can be passed to + every read command. +- `search --compact` / `explain --compact`: agent text instead of JSON. Each result is + `path:start-end symbols` with up to eight numbered lines underneath that carry the + match (rarer query words first), so an agent can cite `file:line` without a Read or + a `grep -n`. On a benchmark query the packet went from 9.6 KB of JSON to 3.7 KB. + Evidence already sent to the session prints as `(already sent)`. +- `impact --compact`, and a `modules` field on `impact`: the build-system module the + target lives in (Gradle, Maven, Cargo, npm, Go, Python) and the lines of other + build files that name it, split into workspace membership and dependencies. "Does + another module depend on this?" no longer needs a grep over build files. +- The skill routes every question through `--compact` and is 40% shorter. + +### Changed + +- Ranking, measured over 342 queries on six query sets (five repositories, including + 24 natural-language "how does X work" questions): MRR +0.010 (p=0.036), recall@10 + +0.015 (p=0.038), nDCG@10 +0.009 (p=0.018), useful@budget +0.020 (p=0.015), 15 fewer + snippet tokens per query; no corpus regressed. On the natural-language questions MRR + went 0.428 → 0.475 and recall@10 0.708 → 0.833. Three signals, each ablatable: + - `resource_priors`: localisation catalogues (`lang/en_us.json`, `i18n/*.json`, + `.po`) and workflow artifacts (review `.diff`/`.patch` files, agent scratch + directories such as `.superpowers/`) are demoted like generated code; + - `stem_match`: a file named after a query term (`Treasury.java` for a question + about the treasury) gets a small bonus; + - `question_pool_floor`: a question-form query ("how is a crop loaded") gets a + 40-deep candidate pool, so the class it paraphrases reaches the reranker. +- `symbol` returns exact matches alone when there are any, with a count of the prefix + matches it left out (`more_prefix_matches`), instead of burying the one symbol asked + for among twenty `TreasuryX` prefix matches. + +- Every `refs` site carries `caller` (the function it sits in), `target` (the + definition it resolved to) and `receiver` (what the call was made on), so callers can + be named and same-named methods told apart without reading each file. +- Index schema 4: call edges store their receiver. The first `update` after upgrading + rebuilds the index once; read commands keep working on an older index until then. + ## [2.0.0] - 2026-09-14 Evidence release. Everything an agent reads through `codebase-index` is now identified @@ -773,7 +868,8 @@ Pooled over 305 queries (Python, Java, TypeScript), v1.8.0 → 1.9.0: - Hooks example + `watch` mode for keeping the index fresh without blocking the edit loop (M8). - `doctor`, `stats`, `clean` diagnostics/maintenance commands. -[Unreleased]: https://github.com/denfry/codebase-index/compare/v2.0.0...HEAD +[Unreleased]: https://github.com/denfry/codebase-index/compare/v2.1.0...HEAD +[2.1.0]: https://github.com/denfry/codebase-index/compare/v2.0.0...v2.1.0 [2.0.0]: https://github.com/denfry/codebase-index/compare/v1.9.0...v2.0.0 [1.10.0]: https://github.com/denfry/codebase-index/compare/v1.9.0...abb67df [1.9.0]: https://github.com/denfry/codebase-index/compare/v1.8.0...v1.9.0 diff --git a/README.md b/README.md index 5271de0..ba3aa4d 100644 --- a/README.md +++ b/README.md @@ -130,13 +130,16 @@ affected files (8): src/flask/app.py, src/flask/ctx.py, src/flask/globals.py, .. More: `explain "how are blueprints registered"`, `describe dispatch_request`, `architecture` (modules, god nodes, surprising links), `graph User --output graph.html`. -Add `--json` to any command for the machine-readable packet. +Name a method as `Owner.member` (`SessionInterface.open_session`) to leave out +same-named methods of other types. Add `--json` to any command for the +machine-readable packet, or `--compact` (search, explain, refs, impact) for the +agent format: `path:start-end symbol` with the matching lines numbered underneath. ## Agent integrations | Agent | Setup | What it gets | |---|---|---| -| **Claude Code** | `codebase-index init --target claude`, or the plugin: `/plugin marketplace add denfry/codebase-index` then `/plugin install codebase-index@codebase-index` | A skill that routes repository questions to the index and reads only `recommended_reads` ranges; optional PostToolUse hook keeps the index fresh | +| **Claude Code** | `codebase-index init --target claude`, or the plugin: `/plugin marketplace add denfry/codebase-index` then `/plugin install codebase-index@codebase-index` | A skill that routes repository questions to the index and answers from `--compact` output (ranked files with numbered matching lines, `refs` with the calling function per site); optional PostToolUse hook keeps the index fresh | | **Codex CLI** | `codebase-index init --target codex` | A managed block in `AGENTS.md` plus the skill resources | | **OpenCode** | `codebase-index init --target opencode` | `/codebase-index` command, agent file, skill resources | | **Any MCP client** (Claude Desktop, Cursor, VS Code, Zed, Windsurf, ...) | `pip install "codebase-index[mcp]"` then `codebase-index mcp --root /path/to/repo` | 11 tools (`search_code`, `find_refs`, `impact_of`, `impact_of_diff`, `path_between`, ...) with a versioned JSON envelope | @@ -172,10 +175,14 @@ significant on a pooled multi-language query set ([tests/eval](tests/eval/README.md)). 1.9.0 removed two signals that could not show a benefit and rejected five plausible ones. -**What is not measured yet**: whether an *agent* completes tasks better with the -index. That needs model calls and a rubric and is the top item in -[BENCHMARKS.md](docs/BENCHMARKS.md#future-work-in-priority-order). Please do not -quote task-success numbers for this project; there are none. +**Agent pilot (2.1.0).** The same five code questions on a 5.9k-file Java/Rust +monorepo, answered by a coding agent three times with the skill and three times with +Grep/Read only: **19% fewer tokens and 25% fewer tool calls** with the skill, all +answers correct in both arms, no overlap between the arms' token ranges. It is one +private repository and five questions, so read it as a direction: +[raw runs and caveats](tests/eval/results/2026-09-24-agent-pilot.md). A broader +task-level evaluation is still the top item in +[BENCHMARKS.md](docs/BENCHMARKS.md#future-work-in-priority-order). ## How it works diff --git a/docs/BENCHMARKS.md b/docs/BENCHMARKS.md index 4b3c80f..2cfa975 100644 --- a/docs/BENCHMARKS.md +++ b/docs/BENCHMARKS.md @@ -178,6 +178,25 @@ Every ranking signal that ships has an ablation row. 1.9.0 removed two signals that could not demonstrate a benefit and rejected several plausible ones (IDF-weighted coverage, stemming, graph propagation, MMR, a file-length prior). +## Agent pilot (2.1.0) + +The first task-level measurement: five code questions on one private 5.9k-file +Java/Rust monorepo, a Claude Code subagent answering each set three times with the +skill and three times with Grep/Read only +([raw runs, questions, ground truth, caveats](../tests/eval/results/2026-09-24-agent-pilot.md)). + +| arm | mean tokens | range | mean tool calls | correct | +|---|---:|---|---:|---| +| skill 2.1.0 | **83.5k** | 82.1k–85.0k | **9.0** | 15/15 | +| Grep/Read only | 103.0k | 98.7k–107.1k | 12.0 | 15/15 | +| skill, pre-2.1.0 template | 100.6k | 96.1k–103.5k | 16.7 | 15/15 | + +The 2.1.0 ranking changes were gated separately over 342 queries (six query sets, +five repositories, 24 of them natural-language questions): MRR +0.010 (p=0.031), +recall@10 +0.019 (p=0.017), nDCG@10 +0.011 (p=0.005), useful@budget +0.022 +(p=0.008), no corpus regressing +([log](../tests/eval/results/2026-09-24-retrieval-2.1.0.txt)). + ## Evidence memory `tests/eval/memory_eval.py` replays a repository's own history: for every git-derived @@ -223,16 +242,15 @@ Do not write, imply, or ship any of these until a run with published logs exists - Any *token savings* multiplier without naming the read model. Under symmetric accounting the index costs about the same as disciplined grep. - Latency comparisons against external tools. -- Any statement about *LLM agent task success*. The baselines here are - retrieval metrics; nobody has measured whether an agent with the index - finishes tasks faster or better. That is the most important open item. +- Any general statement about *LLM agent task success* or savings beyond the + pilot above (one private repository, five questions, one model). Quote the + pilot with its scope or not at all. ## Future work, in priority order -1. **Agent task-level evaluation**: the same questions, an actual coding agent - (Claude Code or Codex CLI) with and without the skill, measuring answer - correctness, files read, tokens, and wall time. Needs model calls and a - grading rubric; not started. +1. **Agent task-level evaluation**: the pilot above covers one repository and + five questions. Next: public repositories, natural-language and debugging + tasks, more runs per arm, and a second agent (Codex CLI). 2. **Large repository run**: a 500k–1M LOC monorepo, reporting index build time, incremental update latency, memory, and the same retrieval metrics. 3. **Graph task benchmark**: hand-labelled `refs`, `impact`, and diff --git a/docs/installer.md b/docs/installer.md index abf2c1f..458c3b5 100644 --- a/docs/installer.md +++ b/docs/installer.md @@ -110,7 +110,7 @@ pwsh ./install.ps1 -Target claude -InstallDir "D:\skills\codebase-index" **Pin to a branch or tag** (reproducibility and safety): ```sh -sh install.sh --branch v2.0.0 +sh install.sh --branch v2.1.0 ``` --- diff --git a/requirements.lock b/requirements.lock index eb8619a..084a720 100644 --- a/requirements.lock +++ b/requirements.lock @@ -1,3 +1,3 @@ -codebase-index @ https://github.com/denfry/codebase-index/archive/refs/tags/v2.0.0.tar.gz +codebase-index @ https://github.com/denfry/codebase-index/archive/refs/tags/v2.1.0.tar.gz tree-sitter==0.25.2 tree-sitter-language-pack==1.8.1 diff --git a/skill/SKILL.md b/skill/SKILL.md index 432005b..ba39234 100644 --- a/skill/SKILL.md +++ b/skill/SKILL.md @@ -6,106 +6,60 @@ allowed-tools: Bash(codebase-index search *), Bash(codebase-index explain *), Ba # Codebase Index -Use the local index before reading repository files. +Use the local index before reading repository files. It answers with ranked, +numbered lines, so most answers need no Read and no Grep at all. -The operating principle is **Find → Trace → Verify → Predict**: - -- **Find** the implementation with ranked retrieval. -- **Trace** behavior through definitions, callers, dependencies, and paths. -- **Verify** that evidence you already hold is still true before relying on it. -- **Predict** change impact while preserving an explicit evidence trail. - -## Route the question +## Route the question — always `--compact` | Intent | Command | |---|---| -| Where is X implemented? | `codebase-index search "X" --session --json` | -| How does X work? | `codebase-index explain "X" --session --json` | -| What is this codebase? | `codebase-index architecture --json` | -| Find a named symbol | `codebase-index symbol "X" --json` | -| Who calls or references X? | `codebase-index refs "X" --json` | -| What changes if X changes? | `codebase-index impact "X" --json` | -| What does my current diff affect? | `codebase-index diff-impact --json` | -| How are X and Y connected? | `codebase-index path "X" "Y" --json` | -| Describe X and its neighborhood | `codebase-index describe "X" --json` | +| Where / how does X work? | `codebase-index search "X" --compact --session ` | +| Overview of a feature | `codebase-index explain "X" --compact --session ` | +| Every caller / call site of X | `codebase-index refs "Owner.member" --compact` | +| What breaks if X changes (incl. other modules) | `codebase-index impact "Owner.member" --compact` | +| Find a definition | `codebase-index symbol "X"` | +| A class: members, who uses it | `codebase-index describe "Class" --json` | +| What does my diff affect? | `codebase-index diff-impact --json` | | Is what I read earlier still true? | `codebase-index verify --session --json` | -| Produce a human graph | `codebase-index graph "X" --output ` | - -Use `search --mode symbol` for exact symbol work, `--mode fts` for text and -error messages, and the default `hybrid` mode for mixed questions. Use pure -`vector` mode only when embeddings are enabled and exact vocabulary is unknown. - -Read [references/commands.md](references/commands.md) only when command options -or routing remain unclear. - -## Evidence protocol - -1. Pick one session tag for this conversation (for example `auth-fix-1`) and - pass `--session ` to every `search` and `explain`. -2. Run the best-matching command with `--json`. -3. Check `index` before trusting the payload: - - missing → run `codebase-index index`, then repeat; - - stale with fewer than 20 changed files → run `codebase-index update`; - - stale with 20 or more changed files → run `codebase-index index`; - - fresh → continue. -4. Start with ranks 1–3. Read only `recommended_reads` line ranges. -5. Trace one additional hop only when the question requires behavior, - ownership, or impact. -6. Before answering or editing from evidence gathered earlier in the task, run - `codebase-index verify --session --json` and reread anything whose - state is not `valid` or `relocated`. -7. Answer with `file:line` evidence and state uncertainty explicitly. - -Do not open whole files when a line range is available. A snippet may already -be sufficient. `skeletonized: true` means the response intentionally folded -unrelated body lines; read the supplied range when the missing body matters. - -## Evidence memory - -- `reused: true` with `snippet: null` — this session already received that - exact text and its source is unchanged. Use your earlier copy; if you can no - longer see it, Read the range. -- `memory.invalidated` — evidence this session received has changed since. - Treat your earlier copy as wrong and reread before relying on it. -- `stale: true` — the index is older than the file. Run `codebase-index update` - or Read the range. -- A tag belongs to one context. Never give it to a subagent or another - conversation. Start a new tag after the context is cleared or compacted, or - whenever earlier snippets are no longer visible to you. - -Verdict states and citing evidence in notes: [references/memory.md](references/memory.md). - -## Confidence contract - -- **high** — answer from the indexed evidence. -- **medium** — read the recommended ranges and confirm the key claim with one - targeted lookup if necessary. -- **low** or no results — follow `fallback_suggestions`, then use a narrow - Grep/Glob fallback. - -On `refs` and `impact`, inspect `coverage`. If `coverage.partial` is true, an -empty result is inconclusive; confirm with targeted Grep before saying that -nothing references the target. - -Edges carry `confidence`: - -- `extracted` — exact parser evidence; -- `inferred` — heuristic resolution; -- `ambiguous` — unresolved or non-unique. - -Never present an inferred or ambiguous chain as certain. - -## Answer contract - -Structure repository answers around: - -1. **Answer** — the direct conclusion. -2. **Evidence** — the minimum supporting `file:line` references. -3. **Confidence** — only when evidence is partial, inferred, stale, or missing. -4. **Next check** — only when another check would materially reduce uncertainty. - -Do not narrate every search step. Do not claim absence from a partial graph. -Do not replace evidence with a generated HTML graph. -For payload fields and failure handling, read -[references/response-contract.md](references/response-contract.md). +`--compact` prints `path:start-end symbol` per result with the matching lines +numbered underneath (` 24| public static final String FILE_ID = ...`). Cite +those lines as `path:24` directly. Read a range only when the lines shown do not +answer the question — and then only that range, never the whole file. + +Name members as `Owner.member` (`TownService.refresh`, `activation::place`, +`CoreError::ObjectDamaged`) so same-named methods of other types are excluded. +`refs --compact` lines are `path:line kind caller -> target`: the list of call +sites with the calling function is usually the whole answer. Add +`--exclude-tests` for production-only lists and `--path ` for one module. +`kind reference` is a non-call use, e.g. an enum variant matched in a `match` arm. + +`impact --compact` starts with `# module : build files depending on it: ...` +— the answer to "does another module depend on this", with no build-file grep. + +## Protocol + +1. Pick one session tag per conversation (e.g. `auth-fix-1`); pass it to + `search`/`explain`. Results already sent in this session print as + `(already sent)` — use your earlier copy. +2. A header saying `index stale` → run `codebase-index update` once; `NO INDEX` + → `codebase-index index`. +3. Batch independent questions into one Bash call (`cmd1; echo ---; cmd2`). +4. `# partial:` on refs/impact means the list may be incomplete. For + `Owner.member` it lists `possible_call` sites (calls on a variable the index + cannot type), nearest first: check those few lines, do not grep the repo. +5. Before relying on evidence from earlier in a long task, run + `codebase-index verify --session --json`. +6. Grep only when the index returns nothing relevant or for non-code text. + +Edge confidence: `extracted` exact, `inferred` heuristic (receiver matched a +type), `ambiguous` unresolved. Never present an inferred chain as certain. + +## Answer + +Lead with the answer, then the minimum `file:line` evidence. State uncertainty +only when evidence is partial, inferred or stale. + +Options and JSON fields: [references/commands.md](references/commands.md), +[references/response-contract.md](references/response-contract.md), +session memory: [references/memory.md](references/memory.md). diff --git a/skill/references/commands.md b/skill/references/commands.md index 79c91a6..197ffe7 100644 --- a/skill/references/commands.md +++ b/skill/references/commands.md @@ -37,6 +37,24 @@ codebase-index describe "" --json - `architecture` reads module analysis cached at index time. - `refs` finds definitions, calls, and graph-backed references. + For a method whose name several types share, pass `Owner.member` + (`TownService.refresh`, or `module::fn` for a top-level function); each site + names its `caller` and the `target` it resolved to. Calls the index cannot type + (`chest.holdings().take(..)`) come back as `possible_call`, nearest first, with + `coverage.partial: true`: read those before claiming a complete list. + `impact` and `symbol` accept the same form. + Filters: `--exclude-tests`, `--path ` (repeatable). `--compact` prints + `path:line kind caller -> target [confidence]`, one site per line. +- `search`, `explain` and `impact` also take `--compact`: agent text instead of + JSON. For search/explain, each result is `path:start-end symbols` followed by up + to eight numbered lines that carry the match (rarer query words first); + evidence already sent to the session prints as `(already sent)`. For impact, + one `d path:line name via edge` line per node, after a + `# module ...` line naming the build files that depend on the target's module. +- `symbol` returns exact matches alone when there are any and reports how many + prefix matches it left out (`more_prefix_matches`); `--exact` drops them. +- `describe` on a class, enum, struct or trait also lists its `members` and + folds their edges into `used_by` / `uses` (code outside the type). - `impact` walks dependents (`up`), dependencies (`down`), or both. - `diff-impact` aggregates impact for tracked changes relative to a verified Git commit; new or excluded files are reported as unresolved. diff --git a/skills/codebase-index/SKILL.md b/skills/codebase-index/SKILL.md index 432005b..ba39234 100644 --- a/skills/codebase-index/SKILL.md +++ b/skills/codebase-index/SKILL.md @@ -6,106 +6,60 @@ allowed-tools: Bash(codebase-index search *), Bash(codebase-index explain *), Ba # Codebase Index -Use the local index before reading repository files. +Use the local index before reading repository files. It answers with ranked, +numbered lines, so most answers need no Read and no Grep at all. -The operating principle is **Find → Trace → Verify → Predict**: - -- **Find** the implementation with ranked retrieval. -- **Trace** behavior through definitions, callers, dependencies, and paths. -- **Verify** that evidence you already hold is still true before relying on it. -- **Predict** change impact while preserving an explicit evidence trail. - -## Route the question +## Route the question — always `--compact` | Intent | Command | |---|---| -| Where is X implemented? | `codebase-index search "X" --session --json` | -| How does X work? | `codebase-index explain "X" --session --json` | -| What is this codebase? | `codebase-index architecture --json` | -| Find a named symbol | `codebase-index symbol "X" --json` | -| Who calls or references X? | `codebase-index refs "X" --json` | -| What changes if X changes? | `codebase-index impact "X" --json` | -| What does my current diff affect? | `codebase-index diff-impact --json` | -| How are X and Y connected? | `codebase-index path "X" "Y" --json` | -| Describe X and its neighborhood | `codebase-index describe "X" --json` | +| Where / how does X work? | `codebase-index search "X" --compact --session ` | +| Overview of a feature | `codebase-index explain "X" --compact --session ` | +| Every caller / call site of X | `codebase-index refs "Owner.member" --compact` | +| What breaks if X changes (incl. other modules) | `codebase-index impact "Owner.member" --compact` | +| Find a definition | `codebase-index symbol "X"` | +| A class: members, who uses it | `codebase-index describe "Class" --json` | +| What does my diff affect? | `codebase-index diff-impact --json` | | Is what I read earlier still true? | `codebase-index verify --session --json` | -| Produce a human graph | `codebase-index graph "X" --output ` | - -Use `search --mode symbol` for exact symbol work, `--mode fts` for text and -error messages, and the default `hybrid` mode for mixed questions. Use pure -`vector` mode only when embeddings are enabled and exact vocabulary is unknown. - -Read [references/commands.md](references/commands.md) only when command options -or routing remain unclear. - -## Evidence protocol - -1. Pick one session tag for this conversation (for example `auth-fix-1`) and - pass `--session ` to every `search` and `explain`. -2. Run the best-matching command with `--json`. -3. Check `index` before trusting the payload: - - missing → run `codebase-index index`, then repeat; - - stale with fewer than 20 changed files → run `codebase-index update`; - - stale with 20 or more changed files → run `codebase-index index`; - - fresh → continue. -4. Start with ranks 1–3. Read only `recommended_reads` line ranges. -5. Trace one additional hop only when the question requires behavior, - ownership, or impact. -6. Before answering or editing from evidence gathered earlier in the task, run - `codebase-index verify --session --json` and reread anything whose - state is not `valid` or `relocated`. -7. Answer with `file:line` evidence and state uncertainty explicitly. - -Do not open whole files when a line range is available. A snippet may already -be sufficient. `skeletonized: true` means the response intentionally folded -unrelated body lines; read the supplied range when the missing body matters. - -## Evidence memory - -- `reused: true` with `snippet: null` — this session already received that - exact text and its source is unchanged. Use your earlier copy; if you can no - longer see it, Read the range. -- `memory.invalidated` — evidence this session received has changed since. - Treat your earlier copy as wrong and reread before relying on it. -- `stale: true` — the index is older than the file. Run `codebase-index update` - or Read the range. -- A tag belongs to one context. Never give it to a subagent or another - conversation. Start a new tag after the context is cleared or compacted, or - whenever earlier snippets are no longer visible to you. - -Verdict states and citing evidence in notes: [references/memory.md](references/memory.md). - -## Confidence contract - -- **high** — answer from the indexed evidence. -- **medium** — read the recommended ranges and confirm the key claim with one - targeted lookup if necessary. -- **low** or no results — follow `fallback_suggestions`, then use a narrow - Grep/Glob fallback. - -On `refs` and `impact`, inspect `coverage`. If `coverage.partial` is true, an -empty result is inconclusive; confirm with targeted Grep before saying that -nothing references the target. - -Edges carry `confidence`: - -- `extracted` — exact parser evidence; -- `inferred` — heuristic resolution; -- `ambiguous` — unresolved or non-unique. - -Never present an inferred or ambiguous chain as certain. - -## Answer contract - -Structure repository answers around: - -1. **Answer** — the direct conclusion. -2. **Evidence** — the minimum supporting `file:line` references. -3. **Confidence** — only when evidence is partial, inferred, stale, or missing. -4. **Next check** — only when another check would materially reduce uncertainty. - -Do not narrate every search step. Do not claim absence from a partial graph. -Do not replace evidence with a generated HTML graph. -For payload fields and failure handling, read -[references/response-contract.md](references/response-contract.md). +`--compact` prints `path:start-end symbol` per result with the matching lines +numbered underneath (` 24| public static final String FILE_ID = ...`). Cite +those lines as `path:24` directly. Read a range only when the lines shown do not +answer the question — and then only that range, never the whole file. + +Name members as `Owner.member` (`TownService.refresh`, `activation::place`, +`CoreError::ObjectDamaged`) so same-named methods of other types are excluded. +`refs --compact` lines are `path:line kind caller -> target`: the list of call +sites with the calling function is usually the whole answer. Add +`--exclude-tests` for production-only lists and `--path ` for one module. +`kind reference` is a non-call use, e.g. an enum variant matched in a `match` arm. + +`impact --compact` starts with `# module : build files depending on it: ...` +— the answer to "does another module depend on this", with no build-file grep. + +## Protocol + +1. Pick one session tag per conversation (e.g. `auth-fix-1`); pass it to + `search`/`explain`. Results already sent in this session print as + `(already sent)` — use your earlier copy. +2. A header saying `index stale` → run `codebase-index update` once; `NO INDEX` + → `codebase-index index`. +3. Batch independent questions into one Bash call (`cmd1; echo ---; cmd2`). +4. `# partial:` on refs/impact means the list may be incomplete. For + `Owner.member` it lists `possible_call` sites (calls on a variable the index + cannot type), nearest first: check those few lines, do not grep the repo. +5. Before relying on evidence from earlier in a long task, run + `codebase-index verify --session --json`. +6. Grep only when the index returns nothing relevant or for non-code text. + +Edge confidence: `extracted` exact, `inferred` heuristic (receiver matched a +type), `ambiguous` unresolved. Never present an inferred chain as certain. + +## Answer + +Lead with the answer, then the minimum `file:line` evidence. State uncertainty +only when evidence is partial, inferred or stale. + +Options and JSON fields: [references/commands.md](references/commands.md), +[references/response-contract.md](references/response-contract.md), +session memory: [references/memory.md](references/memory.md). diff --git a/skills/codebase-index/references/commands.md b/skills/codebase-index/references/commands.md index 79c91a6..197ffe7 100644 --- a/skills/codebase-index/references/commands.md +++ b/skills/codebase-index/references/commands.md @@ -37,6 +37,24 @@ codebase-index describe "" --json - `architecture` reads module analysis cached at index time. - `refs` finds definitions, calls, and graph-backed references. + For a method whose name several types share, pass `Owner.member` + (`TownService.refresh`, or `module::fn` for a top-level function); each site + names its `caller` and the `target` it resolved to. Calls the index cannot type + (`chest.holdings().take(..)`) come back as `possible_call`, nearest first, with + `coverage.partial: true`: read those before claiming a complete list. + `impact` and `symbol` accept the same form. + Filters: `--exclude-tests`, `--path ` (repeatable). `--compact` prints + `path:line kind caller -> target [confidence]`, one site per line. +- `search`, `explain` and `impact` also take `--compact`: agent text instead of + JSON. For search/explain, each result is `path:start-end symbols` followed by up + to eight numbered lines that carry the match (rarer query words first); + evidence already sent to the session prints as `(already sent)`. For impact, + one `d path:line name via edge` line per node, after a + `# module ...` line naming the build files that depend on the target's module. +- `symbol` returns exact matches alone when there are any and reports how many + prefix matches it left out (`more_prefix_matches`); `--exact` drops them. +- `describe` on a class, enum, struct or trait also lists its `members` and + folds their edges into `used_by` / `uses` (code outside the type). - `impact` walks dependents (`up`), dependencies (`down`), or both. - `diff-impact` aggregates impact for tracked changes relative to a verified Git commit; new or excluded files are reported as unresolved. diff --git a/src/codebase_index/__init__.py b/src/codebase_index/__init__.py index 61ed905..40ce013 100644 --- a/src/codebase_index/__init__.py +++ b/src/codebase_index/__init__.py @@ -4,4 +4,4 @@ See docs/ARCHITECTURE.md for the module map. """ -__version__ = "2.0.0" +__version__ = "2.1.0" diff --git a/src/codebase_index/cli.py b/src/codebase_index/cli.py index 71e6f99..9d61b2c 100644 --- a/src/codebase_index/cli.py +++ b/src/codebase_index/cli.py @@ -413,6 +413,10 @@ def search( help="Tag naming ONE agent context: evidence it already received and that is " "unchanged is not resent; changes to it are reported (docs/MEMORY.md).", ), + compact: bool = typer.Option( + False, "--compact", + help="Agent text: ranked locations with numbered matching lines, no JSON.", + ), json_out: bool = typer.Option(False, "--json", help="Emit machine-readable JSON."), ) -> None: """Hybrid ranked search; returns compact results + recommended_reads.""" @@ -439,9 +443,12 @@ def search( payload = search_payload( db_path, cfg, query, mode=mode, limit=limit, offset=offset, token_budget=token_budget, no_fallback=no_fallback, backend=backend, raw=raw, - session=session, + session=session, hit_lines=compact, ) + if compact: + typer.echo(md_renderer.render_search_compact(payload)) + return want_json = json_out or (ctx.obj and ctx.obj.get("json")) typer.echo(json_renderer.render(payload) if want_json else md_renderer.render(payload)) @@ -451,15 +458,25 @@ def symbol( ctx: typer.Context, name: str = typer.Argument(...), kind: Optional[str] = typer.Option(None, "--kind", help="Filter by symbol kind."), - exact: bool = typer.Option(False, "--exact"), + exact: bool = typer.Option( + False, "--exact", help="Only exact matches, even when there are none." + ), + session: Optional[str] = typer.Option( + None, "--session", + help="Accepted so one tag can go to every read command; this command returns " + "locations, not snippets, so there is nothing to withhold.", + ), json_flag: bool = typer.Option(False, "--json", help="Emit machine-readable JSON."), ) -> None: - """Locate a symbol definition by name.""" + """Locate a symbol definition by name (or `Owner.member`). + + Exact matches come alone when there are any; otherwise prefix matches.""" from .output import json as json_out from .output import markdown as md_out from .retrieval.searchers import symbol_lookup from .storage.db import Database + _checked_session(session) is_json = json_flag or bool(ctx.obj and ctx.obj.get("json")) db_path, _cfg = _ensure_index(ctx) @@ -473,20 +490,41 @@ def refs( ctx: typer.Context, symbol_name: str = typer.Argument(...), kind: str = typer.Option("all", "--kind", help="callers|all"), + exclude_tests: bool = typer.Option( + False, "--exclude-tests", help="Drop sites in test files." + ), + path: Optional[list[str]] = typer.Option( + None, "--path", help="Keep only sites under this path prefix (repeatable)." + ), + compact: bool = typer.Option( + False, "--compact", + help="One line per site (path:line kind caller -> target), no JSON envelope.", + ), + session: Optional[str] = typer.Option( + None, "--session", + help="Accepted so one tag can go to every read command; this command returns " + "locations, not snippets, so there is nothing to withhold.", + ), json_flag: bool = typer.Option(False, "--json", help="Emit machine-readable JSON."), ) -> None: - """Find references / callers of a symbol.""" + """Find references / callers of a symbol (a bare name or `Owner.member`).""" from .output import json as json_out from .output import markdown as md_out from .retrieval.searchers import refs_lookup from .storage.db import Database + _checked_session(session) is_json = json_flag or bool(ctx.obj and ctx.obj.get("json")) db_path, _cfg = _ensure_index(ctx) with Database(db_path) as db: - resp = refs_lookup(db.conn, symbol_name, kind=kind) - typer.echo(json_out.render(resp) if is_json else md_out.render_refs(resp)) + resp = refs_lookup( + db.conn, symbol_name, kind=kind, exclude_tests=exclude_tests, paths=path or () + ) + if compact: + typer.echo(md_out.render_refs_compact(resp)) + else: + typer.echo(json_out.render(resp) if is_json else md_out.render_refs(resp)) @app.command() @@ -495,6 +533,14 @@ def impact( target: str = typer.Argument(..., help="File path or symbol name."), depth: int = typer.Option(2, "--depth"), direction: str = typer.Option("up", "--direction", help="up|down|both"), + compact: bool = typer.Option( + False, "--compact", help="One line per affected node, no JSON." + ), + session: Optional[str] = typer.Option( + None, "--session", + help="Accepted so one tag can go to every read command; this command returns " + "locations, not snippets, so there is nothing to withhold.", + ), json_flag: bool = typer.Option(False, "--json", help="Emit machine-readable JSON."), ) -> None: """Blast radius: what is affected if `target` changes (graph walk).""" @@ -506,8 +552,16 @@ def impact( is_json = json_flag or bool(ctx.obj and ctx.obj.get("json")) db_path, _cfg = _ensure_index(ctx) + _checked_session(session) + from .graph.expand import target_paths + from .graph.modules import modules_for_paths + with Database(db_path) as db: resp = impact_lookup(db.conn, target, depth=depth, direction=direction) + resp.modules = modules_for_paths(Path(_cfg.root), target_paths(db.conn, target)) + if compact: + typer.echo(md_out.render_impact_compact(resp)) + return typer.echo(json_out.render(resp) if is_json else md_out.render_impact(resp)) @@ -560,6 +614,10 @@ def explain( help="Tag naming ONE agent context: evidence it already received and that is " "unchanged is not resent; changes to it are reported (docs/MEMORY.md).", ), + compact: bool = typer.Option( + False, "--compact", + help="Agent text: ranked locations with numbered matching lines, no JSON.", + ), json_out: bool = typer.Option(False, "--json", help="Emit machine-readable JSON."), ) -> None: """Intent-aware bundle for 'how does X work' / overview questions.""" @@ -574,9 +632,12 @@ def explain( payload = search_payload( db_path, cfg, normalize_explain_query(query), mode="hybrid", limit=10, token_budget=token_budget, no_fallback=False, backend=backend, raw=raw, - session=session, + session=session, hit_lines=compact, ) + if compact: + typer.echo(md_renderer.render_search_compact(payload)) + return want_json = json_out or (ctx.obj and ctx.obj.get("json")) typer.echo(json_renderer.render(payload) if want_json else md_renderer.render(payload)) diff --git a/src/codebase_index/graph/builder.py b/src/codebase_index/graph/builder.py index 4547883..b0df391 100644 --- a/src/codebase_index/graph/builder.py +++ b/src/codebase_index/graph/builder.py @@ -3,12 +3,15 @@ Runs once after all files are indexed (it needs the complete symbol/file tables). Symbol-target edges (call/reference/extends/implements) resolve only on an -UNAMBIGUOUS name match — if two definitions share a name, the edge is left -unresolved rather than guessed. Import edges resolve their module path to a file -by POSIX path-suffix match (e.g. 'auth.token' -> '%/auth/token.py'). - -The pass is batched: one query for globally-unique symbol names, one for file -paths (expanded into an in-memory suffix map), one executemany for the updates. +UNAMBIGUOUS match within the caller's language family — if two definitions share a +name, the edge is left unresolved rather than guessed. A call written against an +owner (`TownService.refresh(x)`, `activation::place(..)`) also resolves when exactly +one type or module of that name defines it; that match is a heuristic, recorded as +'inferred'. Import edges resolve their module path to a file by POSIX path-suffix +match (e.g. 'auth.token' -> '%/auth/token.py'). + +The pass is batched: one query for all symbols, one for file paths (expanded into +an in-memory suffix map), one executemany for the updates. The per-edge variant did an indexed lookup per symbol edge and up to ~20 full-table LIKE scans per import edge, which dominated large builds. """ @@ -18,6 +21,7 @@ import sqlite3 from typing import Optional +from ..parsers.base import module_name, names_a_type from ..storage import repo _SYMBOL_EDGE_TYPES = {"call", "reference", "extends", "implements"} @@ -48,12 +52,12 @@ def resolve_edges(conn: sqlite3.Connection) -> int: if not edges: return 0 - unique_symbols = repo.unique_symbol_ids_by_name(conn) + targets = _SymbolTargets(repo.symbols_for_resolution(conn)) suffix_map = _path_suffix_map(repo.all_file_ids_with_paths(conn)) - # (dst_kind, dst_id, edge_id, confidence). A repo-unique symbol name is an exact - # hit -> 'extracted'; an import resolved only by path-suffix matching is a best- - # effort heuristic -> 'inferred'. + # (dst_kind, dst_id, edge_id, confidence). A name unique within its language + # family is an exact hit -> 'extracted'; an owner/module match and an import + # resolved by path-suffix matching are heuristics -> 'inferred'. resolutions: list[tuple[str, int, int, str]] = [] for edge in edges: name = edge["dst_name"] @@ -62,14 +66,89 @@ def resolve_edges(conn: sqlite3.Connection) -> int: if file_id is not None: resolutions.append(("file", file_id, edge["id"], "inferred")) elif edge["edge_type"] in _SYMBOL_EDGE_TYPES: - sym_id = unique_symbols.get(name) - if sym_id is not None: - resolutions.append(("symbol", sym_id, edge["id"], "extracted")) + hit = targets.resolve(language_family(edge["lang"]), edge["dst_qualifier"], name) + if hit is not None: + resolutions.append(("symbol", hit[0], edge["id"], hit[1])) repo.resolve_edges_bulk(conn, resolutions) return len(resolutions) +# Languages that call each other directly; any other pair cannot share a call edge. +_LANGUAGE_FAMILIES = { + "kotlin": "jvm", "java": "jvm", "scala": "jvm", + "typescript": "js", "javascript": "js", + "cpp": "c", +} + + +def language_family(lang: Optional[str]) -> Optional[str]: + return _LANGUAGE_FAMILIES.get(lang or "", lang) + + +_Key = tuple[Optional[str], str, str] + + +class _SymbolTargets: + """Where a symbol edge may point, indexed three ways within a language family. + + - by name, for names defined once in the family; + - by (owner type, name), for `TownService.refresh(x)`; + - by (module, name), for a top-level definition called through its module + (`activation::place(..)`, `token.refresh()`); see `module_name`. + A key that two files claim maps to nothing: that is ambiguity, not a guess. + Overloads (one owner, one file, several rows) map to the first of them. + """ + + def __init__(self, rows: list[sqlite3.Row]) -> None: + by_name: dict[tuple[Optional[str], str], list[tuple[int, Optional[str]]]] = {} + by_owner: dict[_Key, tuple[int, int]] = {} + by_module: dict[_Key, tuple[int, int]] = {} + clashes: set[_Key] = set() + for row in rows: + family = language_family(row["lang"]) + name, sym_id, file_id = row["name"], int(row["id"]), int(row["file_id"]) + parts = (row["qualified"] or name).split(".") + owner = parts[-2] if len(parts) >= 2 and parts[-1] == name else None + by_name.setdefault((family, name), []).append((sym_id, owner)) + if owner is not None: + _claim(by_owner, clashes, (family, owner, name), sym_id, file_id) + elif len(parts) == 1: + module = module_name(row["path"]) + _claim(by_module, clashes, (family, module, name), sym_id, file_id) + self._unique = {key: ids[0] for key, ids in by_name.items() if len(ids) == 1} + self._owned = {k: v[0] for k, v in by_owner.items() if k not in clashes} + self._module = {k: v[0] for k, v in by_module.items() if k not in clashes} + + def resolve( + self, family: Optional[str], receiver: Optional[str], name: str + ) -> Optional[tuple[int, str]]: + unique = self._unique.get((family, name)) + # `ApprenticeService.take(x)` is not a call to the family's only `take` when + # that one belongs to Holdings. A top-level definition keeps matching any + # receiver (`Utils.fn()` may be a namespace import). + if unique is not None and not ( + unique[1] is not None and names_a_type(receiver) and receiver != unique[1] + ): + return unique[0], "extracted" + if receiver is None: + return None + sym_id = self._owned.get((family, receiver, name)) + if sym_id is None: + sym_id = self._module.get((family, receiver, name)) + return (sym_id, "inferred") if sym_id is not None else None + + +def _claim( + table: dict[_Key, tuple[int, int]], clashes: set[_Key], key: _Key, sym_id: int, file_id: int +) -> None: + first = table.get(key) + if first is None: + table[key] = (sym_id, file_id) + elif first[1] != file_id: + clashes.add(key) + + def _path_suffix_map(rows: list[sqlite3.Row]) -> dict[str, Optional[int]]: """Map every '/'-aligned path suffix to its file id, or None when ambiguous. diff --git a/src/codebase_index/graph/expand.py b/src/codebase_index/graph/expand.py index ba75a0f..81d1794 100644 --- a/src/codebase_index/graph/expand.py +++ b/src/codebase_index/graph/expand.py @@ -7,7 +7,8 @@ Target resolution: an exact file path -> a file node (seeded together with all symbols defined in that file, so importers AND subclassers surface). Otherwise a -symbol name -> all symbol nodes with that name. A path suffix is the last resort. +symbol name -> all symbol nodes with that name, or a qualified `Owner.member` -> +that type's members. A path suffix is the last resort. """ from __future__ import annotations @@ -17,7 +18,13 @@ from collections import deque from typing import Optional -from ..models import GraphCoverage, ImpactNode, ImpactResponse, IndexFreshness +from ..models import ( + GraphCoverage, + ImpactNode, + ImpactResponse, + IndexFreshness, + unmatched_coverage, +) from ..storage import repo @@ -76,7 +83,7 @@ def _seed_nodes(conn: sqlite3.Connection, target: str) -> list[tuple[str, int]]: seeds += [("symbol", int(s["id"])) for s in repo.symbols_in_file(conn, int(frow["id"]))] return seeds - sym_rows = repo.symbols_by_name(conn, target, exact=True) + sym_rows, _member = repo.symbols_for_target(conn, target) if sym_rows: return [("symbol", int(r["id"])) for r in sym_rows] @@ -211,11 +218,16 @@ def walk_impact( return sorted(nodes.values(), key=_impact_sort_key) +def target_paths(conn: sqlite3.Connection, target: str) -> list[str]: + """The file path(s) a refs/impact target resolves to.""" + return _target_paths(conn, target) + + def _target_paths(conn: sqlite3.Connection, target: str) -> list[str]: """The file path(s) the target resolves to, for coverage classification.""" if repo.file_by_path(conn, target) is not None: return [target] - sym_rows = repo.symbols_by_name(conn, target, exact=True) + sym_rows, _member = repo.symbols_for_target(conn, target) if sym_rows: return [r["path"] for r in sym_rows] suffix = repo.files_with_suffix(conn, target) @@ -242,5 +254,9 @@ def impact_lookup( return ImpactResponse( target=target, direction=direction, depth=depth, index=_freshness(conn), nodes=nodes, files=files, - coverage=GraphCoverage.for_paths(_target_paths(conn, target)), + coverage=( + GraphCoverage.for_paths(paths) + if (paths := _target_paths(conn, target)) + else unmatched_coverage(target) + ), ) diff --git a/src/codebase_index/graph/modules.py b/src/codebase_index/graph/modules.py new file mode 100644 index 0000000..53ab7fe --- /dev/null +++ b/src/codebase_index/graph/modules.py @@ -0,0 +1,112 @@ +"""Build-system modules: which module a file belongs to, and who names it. + +"Would a change here break another module?" is a build-graph question the code +graph cannot answer: a Gradle subproject that nothing depends on has no incoming +edges whether or not other projects exist. Agents answered it by grepping every +build file. This reads the build files instead — cheaply, without evaluating +them — and reports the lines of *other* build files that mention the module, so +the answer comes with its evidence. +""" + +from __future__ import annotations + +import os +import re +from pathlib import Path, PurePosixPath +from typing import Optional + +BUILD_FILES = ( + "build.gradle.kts", "build.gradle", "settings.gradle.kts", "settings.gradle", + "pom.xml", "Cargo.toml", "package.json", "go.mod", "pyproject.toml", +) +_SKIP_DIRS = frozenset({ + ".git", "build", "target", "node_modules", "dist", "out", ".gradle", ".venv", + "venv", "__pycache__", ".idea", ".claude", "run", +}) +_MAX_DEPTH = 5 +_COMMENT_PREFIXES = ("//", "#", "