From 9684e74bc0b374c3f6bf53db45f1d05fea0e12f3 Mon Sep 17 00:00:00 2001 From: Coding-Dev-Tools Date: Sun, 27 Sep 2026 01:52:44 -0400 Subject: [PATCH 1/6] chore(release): add scoped v1.7.6 waiver path --- .github/workflows/release.yml | 47 ++++++++++++++++++++++++++++- docs/RELEASE_QUALIFICATION.md | 18 ++++++++--- docs/RELEASE_READINESS.md | 4 ++- docs/REWORK_EXECUTION.md | 2 +- tests/test_release_qualification.py | 19 ++++++++++-- 5 files changed, 80 insertions(+), 10 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 1307625c..5bc65740 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -8,9 +8,14 @@ on: workflow_dispatch: inputs: release_tag: - description: "Existing tag to repair as a GitHub Release" + description: "Existing tag to repair on PyPI and GitHub" required: false type: string + waive_v176_qualification: + description: "Owner-authorized one-time qualification waiver; valid only for v1.7.6" + required: false + type: boolean + default: false permissions: contents: read @@ -959,6 +964,23 @@ jobs: - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: python-version: "3.11" + - name: Enforce and record the v1.7.6-only qualification waiver + if: inputs.waive_v176_qualification + env: + RELEASE_TAG: ${{ inputs.release_tag }} + GH_ACTOR: ${{ github.actor }} + GH_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + shell: bash + run: | + set -euo pipefail + test "$RELEASE_TAG" = "v1.7.6" + { + printf '# Release qualification waiver\n\n' + printf 'Tag: `%s`\n' "$RELEASE_TAG" + printf 'Authorized by: `%s`\n' "$GH_ACTOR" + printf 'Workflow run: %s\n\n' "$GH_RUN_URL" + printf 'The owner explicitly waived full-product qualification for this repair. No unverified release gate is represented as passed.\n' + } >> "$GITHUB_STEP_SUMMARY" - name: Download published distributions env: GH_TOKEN: ${{ github.token }} @@ -1103,6 +1125,7 @@ jobs: cp dist/*.whl dist/*.tar.gz verified-dist/ - name: Require signed full-product qualification before PyPI repair + if: ${{ !inputs.waive_v176_qualification }} env: RELEASE_TAG: ${{ inputs.release_tag }} ENGRAPHIS_RELEASE_QUALIFICATION: ${{ secrets.ENGRAPHIS_RELEASE_QUALIFICATION }} @@ -1130,6 +1153,7 @@ jobs: --version "${RELEASE_TAG#v}" --retries 18 --delay 10 - name: Require signed full-product qualification before GitHub repair + if: ${{ !inputs.waive_v176_qualification }} env: RELEASE_TAG: ${{ inputs.release_tag }} ENGRAPHIS_RELEASE_QUALIFICATION: ${{ secrets.ENGRAPHIS_RELEASE_QUALIFICATION }} @@ -1161,3 +1185,24 @@ jobs: --title "Engraphis ${RELEASE_TAG#v}" \ --latest fi + + - name: Disclose the qualification waiver in GitHub Release notes + if: inputs.waive_v176_qualification + env: + GH_TOKEN: ${{ github.token }} + GH_REPO: ${{ github.repository }} + RELEASE_TAG: ${{ inputs.release_tag }} + GH_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + shell: bash + run: | + set -euo pipefail + gh release view "$RELEASE_TAG" --repo "$GH_REPO" --json body --jq .body \ + > "$RUNNER_TEMP/release-notes-existing.md" + { + cat "$RUNNER_TEMP/release-notes-existing.md" + printf '\n\n## Release qualification\n\n' + printf 'The owner waived full-product qualification for this release. Mandatory full-product gates are not represented as passed.\n\n' + printf 'Waiver record: %s\n' "$GH_RUN_URL" + } > "$RUNNER_TEMP/release-notes.md" + gh release edit "$RELEASE_TAG" --repo "$GH_REPO" \ + --notes-file "$RUNNER_TEMP/release-notes.md" diff --git a/docs/RELEASE_QUALIFICATION.md b/docs/RELEASE_QUALIFICATION.md index 26cb761d..2c2433d0 100644 --- a/docs/RELEASE_QUALIFICATION.md +++ b/docs/RELEASE_QUALIFICATION.md @@ -1,10 +1,11 @@ # Owner-signed release qualification -Every new PyPI publication, GitHub release write, and repair requires a valid +Every ordinary PyPI publication, GitHub release write, and repair requires a valid full-product qualification. Passing the public build jobs is necessary but does not replace the mandatory private readiness evidence. The workflow fails closed when qualification configuration is missing, malformed, expired or inconsistent -with the selected source and distribution bytes. +with the selected source and distribution bytes. The owner-authorized v1.7.6 +repair waiver is the one-time exception documented below. The public verifier is `scripts/verify_release_qualification.py`. It verifies Ed25519 signatures using `cryptography==50.0.0` in release jobs. It contains no @@ -15,8 +16,8 @@ independent proof that each observation happened. ## Pending owner setup -These operations have **not been performed** by this source change. Publication -will remain blocked until the release owner completes them. +These operations have **not been performed** by this source change. Ordinary +publication will remain blocked until the release owner completes them. 1. Create and protect the GitHub environment `release-qualification`. Restrict its deployment branches/tags to the protected release sources, require an authorized @@ -105,7 +106,14 @@ The normal workflow checks before both PyPI and GitHub writes. Repair first sele a matching historical push run and verifies its exact public distribution/evidence hashes, then checks the current owner approval before both repair writes. It uses the peeled release tag commit, never the repair workflow's `main` checkout commit. -No workflow switch makes the qualification optional. + +For the existing `v1.7.6` release only, the repository owner explicitly directed a +qualification waiver on 2026-09-27. The `workflow_dispatch` input +`waive_v176_qualification` skips the owner qualification verifier only when repairing +`v1.7.6`; the workflow records the actor and run URL, and the GitHub Release notes +state that full-product qualification was waived. This is not a qualification and +does not mark any unverified gate as passing. All ordinary tag publications and +repairs for other versions still require a valid owner-signed qualification. ## Public installed evidence diff --git a/docs/RELEASE_READINESS.md b/docs/RELEASE_READINESS.md index ff4329c2..152acfa5 100644 --- a/docs/RELEASE_READINESS.md +++ b/docs/RELEASE_READINESS.md @@ -92,7 +92,9 @@ engine checkout; `--require-leadership` requires both decisions. All modes retai Actual publication additionally requires the protected, owner-signed approval in [RELEASE_QUALIFICATION.md](RELEASE_QUALIFICATION.md), verified immediately before each normal or repair write. Its environment, authority and approval remain owner setup; -this source change does not configure or issue them. +this source change does not configure or issue them. The owner-authorized v1.7.6 +repair waiver documented there is an explicit exception and does not establish +full-product readiness or change any gate status. Planner experiments remain off by default. To require their existing optimization gate: diff --git a/docs/REWORK_EXECUTION.md b/docs/REWORK_EXECUTION.md index 47e69fa4..db27fcb4 100644 --- a/docs/REWORK_EXECUTION.md +++ b/docs/REWORK_EXECUTION.md @@ -160,7 +160,7 @@ scope/time regression coverage. Original user edits were not rewritten. | Complete-engine measurement | Serializable factory configuration supports real files, pinned local semantic models, exact NumPy/sqlite-vec backends and rerankers. Opt-in recall phases separate embedding, retrieval, ranking and packing. | Full writer occupancy, production-load calibration and measured optimization remain open. | | Capacity acceptance | Lifecycle RSS includes startup, backlog is sampled and recomputed, and the complete matrix validator enforces prebound hosts, WAL/FULL, all scheduled outcomes, RAM and the required 100k latency limits. | No primary 48-cell matrix was executed. The separate 16 GiB reference host remains necessary. | | Installed journeys | A packaged stdlib runner performs actual MCP/HTTP writes, restart recall, correction and historical reads. PR CI and release jobs cover Windows, macOS and Linux, with artifact/dependency identities retained. | Cached Windows source semantic startup passed in four fresh processes, taking 20-23 seconds. This is not semantic qualification of all installed platforms. | -| Evidence and publication | Candidate ledger validation checks identities, hashes, dependencies, outcomes and selected evaluation booleans. All four publication/repair writes require [owner qualification](RELEASE_QUALIFICATION.md). | Owner-protected environment/authority setup, final approval and all missing mandatory evidence remain open. | +| Evidence and publication | Candidate ledger validation checks identities, hashes, dependencies, outcomes and selected evaluation booleans. Ordinary publication/repair writes require [owner qualification](RELEASE_QUALIFICATION.md). | The v1.7.6 owner waiver is a one-time exception, not qualification. Owner-protected setup, final approval and missing mandatory evidence remain open. | | Public claims | Fresh [offline fixture evidence](benchmark-evidence/offline-fixtures-v9.json) reproduces retained public aggregates and binds the current engine/eval source. Historical v1 evidence is preserved. | Planner variants still require successful promotion gates; no retrieval default or leadership claim is promoted. | | Website contract | Active commercial/MCP/install guidance is generated and checked against a selected shipped public contract in the website candidate. | The live portal, authenticated provider journeys and combined deployed identities require attended acceptance. | diff --git a/tests/test_release_qualification.py b/tests/test_release_qualification.py index b53115e2..70469c3c 100644 --- a/tests/test_release_qualification.py +++ b/tests/test_release_qualification.py @@ -173,10 +173,14 @@ def test_cli_requires_configuration_and_never_prints_receipt(qualification, monk assert configuration["ENGRAPHIS_RELEASE_QUALIFICATION"] not in output.out + output.err -def test_every_publication_write_requires_unconditional_qualification(): +def test_publication_writes_require_qualification_except_scoped_v176_waiver(): yaml = pytest.importorskip("yaml") root = Path(__file__).resolve().parents[1] workflow = yaml.safe_load((root / ".github/workflows/release.yml").read_text(encoding="utf-8")) + dispatch = workflow.get("on", workflow.get(True, {})).get("workflow_dispatch", {}) + waiver_input = dispatch.get("inputs", {}).get("waive_v176_qualification", {}) + assert waiver_input.get("type") == "boolean" + assert waiver_input.get("default") is False for name, expected_writes in (("publish", 1), ("github-release", 1), ("github-release-repair", 2)): job = workflow["jobs"][name] assert job["environment"] == "release-qualification" @@ -184,7 +188,11 @@ def test_every_publication_write_requires_unconditional_qualification(): writes = 0 for step in job["steps"]: if "scripts.verify_release_qualification" in step.get("run", ""): - assert "if" not in step and not step.get("continue-on-error", False) + if name == "github-release-repair": + assert step.get("if") == "${{ !inputs.waive_v176_qualification }}" + else: + assert "if" not in step + assert not step.get("continue-on-error", False) required = { "ENGRAPHIS_RELEASE_QUALIFICATION", "ENGRAPHIS_RELEASE_VERIFY_KEY", "ENGRAPHIS_RELEASE_CANDIDATE_ID", "ENGRAPHIS_RELEASE_LEDGER_SHA256", @@ -206,6 +214,13 @@ def test_every_publication_write_requires_unconditional_qualification(): assert writes == expected_writes assert workflow["jobs"]["publish"]["needs"] == "release-evidence" assert workflow["jobs"]["github-release"]["needs"] == "publish" + repair_steps = workflow["jobs"]["github-release-repair"]["steps"] + waiver_guard = next(step for step in repair_steps + if step.get("name") == "Enforce and record the v1.7.6-only qualification waiver") + assert waiver_guard.get("if") == "inputs.waive_v176_qualification" + assert 'test "$RELEASE_TAG" = "v1.7.6"' in waiver_guard["run"] + assert any(step.get("name") == "Disclose the qualification waiver in GitHub Release notes" + and step.get("if") == "inputs.waive_v176_qualification" for step in repair_steps) assert "${{ vars.ENGRAPHIS_RELEASE_" not in ( root / ".github/workflows/release.yml" ).read_text(encoding="utf-8") From db51274a8ed262fab5e76405190dd1ddf3c454e3 Mon Sep 17 00:00:00 2001 From: Coding-Dev-Tools Date: Sun, 27 Sep 2026 02:06:57 -0400 Subject: [PATCH 2/6] fix(release): disclose waiver before exposing GitHub assets --- .github/workflows/release.yml | 45 +++++++++++---------- tests/test_release_qualification.py | 62 ++++++++++++++++++++++++++++- 2 files changed, 84 insertions(+), 23 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 5bc65740..a7ff8bf3 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -1171,9 +1171,32 @@ jobs: GH_TOKEN: ${{ github.token }} GH_REPO: ${{ github.repository }} RELEASE_TAG: ${{ inputs.release_tag }} + WAIVE_QUALIFICATION: ${{ inputs.waive_v176_qualification }} + GH_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} shell: bash run: | + set -euo pipefail + notes_args=() + if [ "$WAIVE_QUALIFICATION" = "true" ]; then + { + printf '## Release qualification\n\n' + printf 'The owner waived full-product qualification for this release. Mandatory full-product gates are not represented as passed.\n\n' + printf 'Waiver record: %s\n' "$GH_RUN_URL" + } > "$RUNNER_TEMP/release-waiver.md" + notes_args=(--notes-file "$RUNNER_TEMP/release-waiver.md") + fi if gh release view "$RELEASE_TAG" --repo "$GH_REPO" >/dev/null 2>&1; then + if [ "$WAIVE_QUALIFICATION" = "true" ]; then + gh release view "$RELEASE_TAG" --repo "$GH_REPO" --json body --jq .body \ + > "$RUNNER_TEMP/release-notes-existing.md" + { + cat "$RUNNER_TEMP/release-notes-existing.md" + printf '\n\n' + cat "$RUNNER_TEMP/release-waiver.md" + } > "$RUNNER_TEMP/release-notes.md" + gh release edit "$RELEASE_TAG" --repo "$GH_REPO" \ + --notes-file "$RUNNER_TEMP/release-notes.md" + fi gh release upload "$RELEASE_TAG" verified-dist/* release-evidence/* \ --repo "$GH_REPO" \ --clobber @@ -1182,27 +1205,7 @@ jobs: --repo "$GH_REPO" \ --verify-tag \ --generate-notes \ + "${notes_args[@]}" \ --title "Engraphis ${RELEASE_TAG#v}" \ --latest fi - - - name: Disclose the qualification waiver in GitHub Release notes - if: inputs.waive_v176_qualification - env: - GH_TOKEN: ${{ github.token }} - GH_REPO: ${{ github.repository }} - RELEASE_TAG: ${{ inputs.release_tag }} - GH_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} - shell: bash - run: | - set -euo pipefail - gh release view "$RELEASE_TAG" --repo "$GH_REPO" --json body --jq .body \ - > "$RUNNER_TEMP/release-notes-existing.md" - { - cat "$RUNNER_TEMP/release-notes-existing.md" - printf '\n\n## Release qualification\n\n' - printf 'The owner waived full-product qualification for this release. Mandatory full-product gates are not represented as passed.\n\n' - printf 'Waiver record: %s\n' "$GH_RUN_URL" - } > "$RUNNER_TEMP/release-notes.md" - gh release edit "$RELEASE_TAG" --repo "$GH_REPO" \ - --notes-file "$RUNNER_TEMP/release-notes.md" diff --git a/tests/test_release_qualification.py b/tests/test_release_qualification.py index 70469c3c..59c71324 100644 --- a/tests/test_release_qualification.py +++ b/tests/test_release_qualification.py @@ -6,7 +6,10 @@ from datetime import datetime, timedelta, timezone import hashlib import json +import os from pathlib import Path +import shutil +import subprocess import pytest @@ -219,8 +222,63 @@ def test_publication_writes_require_qualification_except_scoped_v176_waiver(): if step.get("name") == "Enforce and record the v1.7.6-only qualification waiver") assert waiver_guard.get("if") == "inputs.waive_v176_qualification" assert 'test "$RELEASE_TAG" = "v1.7.6"' in waiver_guard["run"] - assert any(step.get("name") == "Disclose the qualification waiver in GitHub Release notes" - and step.get("if") == "inputs.waive_v176_qualification" for step in repair_steps) + repair = next(step for step in repair_steps if step.get("name") == "Repair GitHub Release") + assert repair["env"]["WAIVE_QUALIFICATION"] == "${{ inputs.waive_v176_qualification }}" + assert repair["run"].index("gh release edit") < repair["run"].index("gh release upload") + assert '"${notes_args[@]}"' in repair["run"].split("gh release create", 1)[1] assert "${{ vars.ENGRAPHIS_RELEASE_" not in ( root / ".github/workflows/release.yml" ).read_text(encoding="utf-8") + + +@pytest.mark.skipif(os.name == "nt", reason="release workflow executes in Linux bash") +@pytest.mark.parametrize("existing,edit_fails", [(False, False), (True, False), (True, True)]) +def test_waiver_disclosure_cannot_follow_github_publication(tmp_path, existing, edit_fails): + yaml = pytest.importorskip("yaml") + bash = shutil.which("bash") + if bash is None: + pytest.skip("bash is unavailable") + root = Path(__file__).resolve().parents[1] + workflow = yaml.safe_load((root / ".github/workflows/release.yml").read_text(encoding="utf-8")) + repair = next(step for step in workflow["jobs"]["github-release-repair"]["steps"] + if step.get("name") == "Repair GitHub Release") + executable = tmp_path / "gh" + executable.write_text("""#!/usr/bin/env bash +set -euo pipefail +printf '%s\\n' "$*" >> "$GH_CALLS" +case "$2" in + view) + if [ "$EXISTING" != true ]; then exit 1; fi + printf 'Existing release notes\\n' + ;; + edit) + if [ "$EDIT_FAILS" = true ]; then exit 7; fi + ;; +esac +""", encoding="utf-8") + executable.chmod(0o700) + script = tmp_path / "repair.sh" + script.write_text(repair["run"], encoding="utf-8") + calls_path = tmp_path / "calls.txt" + result = subprocess.run([bash, str(script)], cwd=tmp_path, capture_output=True, text=True, + timeout=20, env={**os.environ, "PATH": str(tmp_path) + os.pathsep + os.environ["PATH"], + "RUNNER_TEMP": str(tmp_path), "RELEASE_TAG": "v1.7.6", + "WAIVE_QUALIFICATION": "true", "GH_REPO": "test/repo", + "GH_RUN_URL": "https://example.test/run/1", "GH_CALLS": str(calls_path), + "EXISTING": str(existing).lower(), "EDIT_FAILS": str(edit_fails).lower()}) + calls = calls_path.read_text(encoding="utf-8").splitlines() + assert result.returncode == (7 if edit_fails else 0), result.stderr + if existing: + edits = [index for index, call in enumerate(calls) if call.startswith("release edit ")] + uploads = [index for index, call in enumerate(calls) if call.startswith("release upload ")] + assert len(edits) == 1 + assert not uploads if edit_fails else len(uploads) == 1 and edits[0] < uploads[0] + notes = (tmp_path / "release-notes.md").read_text(encoding="utf-8") + assert notes.startswith("Existing release notes") + else: + creation = next(call for call in calls if call.startswith("release create ")) + assert "--notes-file " in creation + assert not any(call.startswith("release edit ") for call in calls) + notes = (tmp_path / "release-waiver.md").read_text(encoding="utf-8") + assert "Mandatory full-product gates are not represented as passed." in notes + assert "https://example.test/run/1" in notes From 3142d7bbaea29c5a2201a0b71fca3d63e342feb2 Mon Sep 17 00:00:00 2001 From: Coding-Dev-Tools Date: Sun, 27 Sep 2026 02:09:54 -0400 Subject: [PATCH 3/6] fix(queue): handle disappearing legacy producer marker --- eval/local_benchmark_queue.py | 6 ++++++ tests/test_local_benchmark_queue.py | 20 ++++++++++++++++++++ 2 files changed, 26 insertions(+) diff --git a/eval/local_benchmark_queue.py b/eval/local_benchmark_queue.py index c8b7be31..65297001 100644 --- a/eval/local_benchmark_queue.py +++ b/eval/local_benchmark_queue.py @@ -368,6 +368,12 @@ def _prerequisite_ready(artifact: Path, producer_lock: Path) -> bool: except (RunnerLockBusy, UnrecognizedRunnerLock): # Legacy ephemeral markers must disappear; never reclaim one by PID. return False + except ValueError: + # A legacy marker may be removed after the probe opens it but before + # it can validate the pathname. Accept only confirmed disappearance; + # a marker that still exists but changed remains a hard failure. + if os.path.lexists(producer_lock): + raise if not artifact.exists(): raise ValueError("prerequisite producer stopped before completing its artifact") _verified_artifact(artifact) diff --git a/tests/test_local_benchmark_queue.py b/tests/test_local_benchmark_queue.py index db88db72..27229c57 100644 --- a/tests/test_local_benchmark_queue.py +++ b/tests/test_local_benchmark_queue.py @@ -1,4 +1,5 @@ import json +from contextlib import contextmanager from pathlib import Path import subprocess import sys @@ -689,6 +690,25 @@ def test_queue_waits_for_legacy_ephemeral_producer_marker_until_removed(tmp_path queue_process.communicate(timeout=10) +def test_queue_accepts_legacy_marker_removed_during_probe(tmp_path, monkeypatch): + artifact = tmp_path / "diagnostic.json" + artifact.write_text("{}", encoding="utf-8") + producer_lock = tmp_path / "legacy-producer.lock" + producer_lock.write_text("running", encoding="utf-8") + + @contextmanager + def disappearing_probe(path, *, create=True): + assert path == producer_lock + assert not create + producer_lock.unlink() + raise ValueError("external diagnostic runner lock is unsafe or changed") + yield # pragma: no cover - the probe always raises + + monkeypatch.setattr(queue, "_runner_lock", disappearing_probe) + monkeypatch.setattr(queue, "_verified_artifact", lambda path: {"verified": True}) + assert queue._prerequisite_ready(artifact, producer_lock) + + def _complete_capacity_summary(): cells = [] hashes = {} From 8123dcbfef8845034393015cf064ff20d72e0ced Mon Sep 17 00:00:00 2001 From: Coding-Dev-Tools Date: Sun, 27 Sep 2026 02:32:21 -0400 Subject: [PATCH 4/6] fix(release): bind and disclose repair waiver before publication --- .github/workflows/release.yml | 40 + BENCHMARKS.md | 780 ++--- README.md | 1776 ++++++------ docs/RELEASE_QUALIFICATION.md | 7 +- .../offline-fixtures-v76.json | 690 +++++ .../offline-fixtures-v76.json.sha256 | 1 + docs/images/context-efficiency.svg | 4 +- .../images/evidence-backed-agent-examples.svg | 4 +- eval/external_checkpoints.py | 27 +- eval/local_benchmark_queue.py | 6 - tests/test_benchmark_evidence.py | 2544 ++++++++--------- tests/test_documentation_contracts.py | 620 ++-- tests/test_local_benchmark_queue.py | 60 +- tests/test_queue_prerequisite_lock.py | 32 + tests/test_release_qualification.py | 99 +- 15 files changed, 3807 insertions(+), 2883 deletions(-) create mode 100644 docs/benchmark-evidence/offline-fixtures-v76.json create mode 100644 docs/benchmark-evidence/offline-fixtures-v76.json.sha256 diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index a7ff8bf3..0ffe56d6 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -1139,6 +1139,43 @@ jobs: python -m scripts.verify_release_qualification --dist dist \ --commit "$ENGRAPHIS_REPAIR_COMMIT" --tag "$RELEASE_TAG" + - name: Disclose the qualification waiver before PyPI repair + if: inputs.waive_v176_qualification + env: + GH_TOKEN: ${{ github.token }} + GH_REPO: ${{ github.repository }} + RELEASE_TAG: ${{ inputs.release_tag }} + GH_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + shell: bash + run: | + set -euo pipefail + test "$RELEASE_TAG" = "v1.7.6" + # The exception covers this retained candidate, not future reuse of its tag. + test "$ENGRAPHIS_REPAIR_COMMIT" = "6a441a75c8dd159607fa3933da83f600864b9146" + { + printf '## Release qualification\n\n' + printf 'The owner waived full-product qualification for this release. Mandatory full-product gates are not represented as passed.\n\n' + printf 'Source commit: `%s`\n\n' "$ENGRAPHIS_REPAIR_COMMIT" + printf 'Waiver record: %s\n' "$GH_RUN_URL" + } > "$RUNNER_TEMP/release-waiver.md" + if gh release view "$RELEASE_TAG" --repo "$GH_REPO" >/dev/null 2>&1; then + gh release view "$RELEASE_TAG" --repo "$GH_REPO" --json body --jq .body \ + > "$RUNNER_TEMP/release-notes-existing.md" + { + cat "$RUNNER_TEMP/release-notes-existing.md" + printf '\n\n' + cat "$RUNNER_TEMP/release-waiver.md" + } > "$RUNNER_TEMP/release-notes.md" + gh release edit "$RELEASE_TAG" --repo "$GH_REPO" \ + --notes-file "$RUNNER_TEMP/release-notes.md" + else + # The public notice survives a later PyPI or verification failure. Assets + # and latest-release promotion still wait for successful publication. + gh release create "$RELEASE_TAG" --repo "$GH_REPO" --verify-tag \ + --generate-notes --notes-file "$RUNNER_TEMP/release-waiver.md" \ + --title "Engraphis ${RELEASE_TAG#v}" --latest=false + fi + - name: Publish only missing verified distributions uses: pypa/gh-action-pypi-publish@ba38be9e461d3875417946c167d0b5f3d385a247 # v1.14.1 with: @@ -1200,6 +1237,9 @@ jobs: gh release upload "$RELEASE_TAG" verified-dist/* release-evidence/* \ --repo "$GH_REPO" \ --clobber + if [ "$WAIVE_QUALIFICATION" = "true" ]; then + gh release edit "$RELEASE_TAG" --repo "$GH_REPO" --latest + fi else gh release create "$RELEASE_TAG" verified-dist/* release-evidence/* \ --repo "$GH_REPO" \ diff --git a/BENCHMARKS.md b/BENCHMARKS.md index 6a6a6ed9..f8eadfa9 100644 --- a/BENCHMARKS.md +++ b/BENCHMARKS.md @@ -1,15 +1,15 @@ -# Benchmarks - -This guide explains what Engraphis measures, how to reproduce each evaluation, and the limits of -those results. When this document and the code disagree, the code is the source of truth. - -The current expansion has a separate [results and workload report](docs/BENCHMARK_EXPANSION_RESULTS.md), -[execution runbook](docs/BENCHMARK_EXPANSION_RUNBOOK.md), and -[proposed stage budgets](docs/BENCHMARK_STAGE_BUDGETS.md). Completed external retrieval diagnostics +# Benchmarks + +This guide explains what Engraphis measures, how to reproduce each evaluation, and the limits of +those results. When this document and the code disagree, the code is the source of truth. + +The current expansion has a separate [results and workload report](docs/BENCHMARK_EXPANSION_RESULTS.md), +[execution runbook](docs/BENCHMARK_EXPANSION_RUNBOOK.md), and +[proposed stage budgets](docs/BENCHMARK_STAGE_BUDGETS.md). Completed external retrieval diagnostics are review artifacts with explicit denominators and uncertainty. The coding pilot uses Codex OAuth only and retains fixture exclusions and interrupted calls. Official QA, competitor scores and capacity qualification remain separate experiments. - + For the locked operator sequence for a public canonical run, see [`docs/PUBLIC_BENCHMARK_RUNBOOK.md`](docs/PUBLIC_BENCHMARK_RUNBOOK.md). @@ -21,10 +21,10 @@ Smart/Classic/service write paths, and `retrieval_recipe="conversation"` or `"lo for the measured depth/budget starting points. `"legacy"` packing and `"default"` retrieval remain the defaults until development, validation and untouched-holdout gates show a workload-specific benefit. - -`python -m eval.evidence_contracts` checks exact-action validation and compares -legacy and coverage packing on small deterministic development fixtures. It runs -in the full offline CI matrix and the NumPy-only Python 3.9 job. These fixtures + +`python -m eval.evidence_contracts` checks exact-action validation and compares +legacy and coverage packing on small deterministic development fixtures. It runs +in the full offline CI matrix and the NumPy-only Python 3.9 job. These fixtures test boundary correctness; they do not estimate external QA or model task success. The [current diagnostic](docs/benchmark-evidence/evidence-contracts-20260921-v11.json) withholds the unchanged oversized, unpunctuated fixture at its 24-token budget: @@ -92,44 +92,44 @@ retain `unknown` provenance unless explicitly bound. These checks protect result interpretation and do not count as additional benchmark-quality gains. ### Public numeric evidence registry - + Every exact public aggregate retained below comes from the checked-in, public-safe -[`offline-fixtures-v73.json`](docs/benchmark-evidence/offline-fixtures-v73.json) artifact. Its +[`offline-fixtures-v76.json`](docs/benchmark-evidence/offline-fixtures-v76.json) artifact. Its SHA-256 is -`aa7ed9c141afcc82fc2a05b63ed9037842f2ea8372667f3142cf9bb795833988`, also recorded in the +`2fb5ce5b2cbc21541f7ae9cad5d2ef615014c0e86f9881f00621d00986422ba8`, also recorded in the adjacent `.sha256` file. The artifact contains no raw questions, answers, prompts, customer data, or per-record content fingerprints. The fixture-suite digest is -`c60a48ea025c5c0ebfdacb68c38f068fc032f05b202e5c192a9223211cf30bf1`. The artifact defines -the digest algorithm and records the SHA-256 of every suite and dataset file. Each evidence ID -also binds its exact command through `sha256(UTF-8 exact command)`: - -| Evidence ID | Exact command | Config digest | -|---|---|---| -| `offline-chunking` | `python -m eval.chunking_eval --dataset eval/datasets/longdoc.jsonl --k 5` | `c1c8196aa7e1568ef3844a9fb2d76b87f342c39108e32d6ad144b885a76143b8` | -| `offline-performance` | `python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 --json` | `bbe4aca81e58d4830e50a8fc7729a1d15b71d97a6299bccd79432b7f119677d7` | -| `offline-grounded` | `python -m eval.grounded` | `590442e51e3642c10489165759919dc86ffac62c182937330c153e7f8d5fc26f` | - -External, model-dependent, latency, consolidation, and productivity numbers are not included in -this offline registry unless a redacted immutable artifact with the same three bindings exists. Use -the [public benchmark runbook](docs/PUBLIC_BENCHMARK_RUNBOOK.md) to produce registry evidence. -Completed retrieval-only diagnostics are documented separately in the -[benchmark expansion results](docs/BENCHMARK_EXPANSION_RESULTS.md); absence from this registry -means no number is claimed in this offline registry. - -The context-efficiency chart is generated from the registry values and the selected report schema. -Historical LoCoMo, graph, handoff, consolidation, and security figures remain preserved in their -source artifacts but are omitted from the current chart until each has a matching immutable, -public-safe artifact. The chart labels coding outcomes, external datasets, and operational -capacity as pending evaluation tracks rather than implying scores. Regenerate it with -`python scripts/render_benchmark_report.py --report docs/benchmark-evidence/offline-fixtures-v73.json --output docs/images/context-efficiency.svg` after selecting the report to publish. +`20131c25c86e3ac5a53285d0631d4ff0c60946879b04ff7a7c38e488f42cda51`. The artifact defines +the digest algorithm and records the SHA-256 of every suite and dataset file. Each evidence ID +also binds its exact command through `sha256(UTF-8 exact command)`: + +| Evidence ID | Exact command | Config digest | +|---|---|---| +| `offline-chunking` | `python -m eval.chunking_eval --dataset eval/datasets/longdoc.jsonl --k 5` | `c1c8196aa7e1568ef3844a9fb2d76b87f342c39108e32d6ad144b885a76143b8` | +| `offline-performance` | `python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 --json` | `bbe4aca81e58d4830e50a8fc7729a1d15b71d97a6299bccd79432b7f119677d7` | +| `offline-grounded` | `python -m eval.grounded` | `590442e51e3642c10489165759919dc86ffac62c182937330c153e7f8d5fc26f` | + +External, model-dependent, latency, consolidation, and productivity numbers are not included in +this offline registry unless a redacted immutable artifact with the same three bindings exists. Use +the [public benchmark runbook](docs/PUBLIC_BENCHMARK_RUNBOOK.md) to produce registry evidence. +Completed retrieval-only diagnostics are documented separately in the +[benchmark expansion results](docs/BENCHMARK_EXPANSION_RESULTS.md); absence from this registry +means no number is claimed in this offline registry. + +The context-efficiency chart is generated from the registry values and the selected report schema. +Historical LoCoMo, graph, handoff, consolidation, and security figures remain preserved in their +source artifacts but are omitted from the current chart until each has a matching immutable, +public-safe artifact. The chart labels coding outcomes, external datasets, and operational +capacity as pending evaluation tracks rather than implying scores. Regenerate it with +`python scripts/render_benchmark_report.py --report docs/benchmark-evidence/offline-fixtures-v76.json --output docs/images/context-efficiency.svg` after selecting the report to publish. The companion examples are also generated from that artifact with -`python -m scripts.render_benchmark_examples --report docs/benchmark-evidence/offline-fixtures-v73.json --output docs/images/evidence-backed-agent-examples.svg`. -The historical-to-executable mapping is in -[`docs/BENCHMARK_CHANGE_COVERAGE.md`](docs/BENCHMARK_CHANGE_COVERAGE.md). - +`python -m scripts.render_benchmark_examples --report docs/benchmark-evidence/offline-fixtures-v76.json --output docs/images/evidence-backed-agent-examples.svg`. +The historical-to-executable mapping is in +[`docs/BENCHMARK_CHANGE_COVERAGE.md`](docs/BENCHMARK_CHANGE_COVERAGE.md). + Fresh diagnostics retain explicit source-case identities so confidence intervals cluster whole conversations even when question IDs do not encode their case. Metrics with no eligible questions remain `null` (unscored), including fresh and @@ -137,176 +137,176 @@ resumed runs. Artifact parsing, checksum validation, and queue receipts bind the same byte snapshot. Comparisons require consistent dataset and repair bindings while allowing the producer implementation to change between versions. -## What we measure today (all offline, no API key) - -Most Engraphis evals score **retrieval**, not end-to-end QA. The separate productivity benchmark -runs a complete offline agent attempt and correction loop, but it is not an official -frontier-model QA score. - -- **Correctness gate**: `eval/harness.py` over `eval/datasets/sample.jsonl` and - `codemem.jsonl` (conflict resolution) and `graph_multihop.jsonl` (multi-hop graph recall). - Runs on the deterministic embedder, so it is a plumbing/regression floor, not a public - performance claim. This is the gate CI enforces. -- **Ablation**: `eval/ablation.py`: vector-only vs. 1-hop graph vs. Personalized-PageRank arm, - to show the graph arm actually earns its place. -- **External benchmarks**: `eval/external.py` loads **LoCoMo** and **LongMemEval** and pushes - them through the *real* `MemoryEngine` write path (conflict resolution + evolution) and hybrid - recall with a real sentence-transformers embedder. It reports `recall_at_k` / `hit_at_k` / - `answer_token_recall`: i.e. *did the evidence come back*, not *did an LLM answer correctly*. - It retains source categories and abstention/no-evidence questions as explicit exclusions from - retrieval-only aggregates rather than silently dropping them. `eval.longmemeval_v2` is a local, - text-only adapter for the official LongMemEval-V2 `insert(trajectory)` / `query(query, - query_image=None)` memory interface; it does not download data or call a model. -- **Grounded**: `eval/grounded.py`: answerable → cite, off-topic → abstain. Exact fixture - outcomes are evidence ID `offline-grounded` in the registry above. -- **Chunking (quality per token)**: `eval/chunking_eval.py` over `eval/datasets/longdoc.jsonl` - ingests a multi-topic corpus twice: once as one memory per document (`whole`) and once with - sub-file `ChunkingExtractor` (`chunked`), then queries both through the real recall pipeline. - The checked-in corpus is explicitly marked trusted eval data so the measurement isolates - chunking from the production trust gate, which excludes arbitrary raw imports from normal - agent context. On the deterministic embedder, **recall@5 is 1.000 for both modes; mean - retrieved top-5 content falls from 740.3 to 214.3 tokens (526.0 fewer, 71.1% lower, about - 3.5× smaller), while the smallest returned evidence-holding memory falls from 162.2 to 42.4 - tokens (119.8 fewer, 73.9% lower, about 3.8× smaller).** These aggregates are evidence ID - `offline-chunking` in the registry above. Pass `--embed-model - sentence-transformers/all-MiniLM-L6-v2` to run a model-dependent experiment; do not publish - that result without a new immutable artifact and pinned model revision. -- **Full-pipeline latency + quality**: `eval/performance.py` times the shipped semantic + - lexical + graph + fusion + scoring + rerank + packing path after warmup, with reinforcement - disabled so repeated measurements do not mutate their corpus. It reports p50/p95/p99 latency, - retrieval quality, packed context tokens, and full/compact JSON-shape payload proxies in one - JSON-safe schema. Its legacy `quality` fields score all candidate chunks returned before - context packing; additive `packed_quality` fields score only chunks admitted to reader context. - Payload proxies are sampled once per question, independently of the number of timed iterations; - they are not serialized MCP envelopes or transport responses. In the +## What we measure today (all offline, no API key) + +Most Engraphis evals score **retrieval**, not end-to-end QA. The separate productivity benchmark +runs a complete offline agent attempt and correction loop, but it is not an official +frontier-model QA score. + +- **Correctness gate**: `eval/harness.py` over `eval/datasets/sample.jsonl` and + `codemem.jsonl` (conflict resolution) and `graph_multihop.jsonl` (multi-hop graph recall). + Runs on the deterministic embedder, so it is a plumbing/regression floor, not a public + performance claim. This is the gate CI enforces. +- **Ablation**: `eval/ablation.py`: vector-only vs. 1-hop graph vs. Personalized-PageRank arm, + to show the graph arm actually earns its place. +- **External benchmarks**: `eval/external.py` loads **LoCoMo** and **LongMemEval** and pushes + them through the *real* `MemoryEngine` write path (conflict resolution + evolution) and hybrid + recall with a real sentence-transformers embedder. It reports `recall_at_k` / `hit_at_k` / + `answer_token_recall`: i.e. *did the evidence come back*, not *did an LLM answer correctly*. + It retains source categories and abstention/no-evidence questions as explicit exclusions from + retrieval-only aggregates rather than silently dropping them. `eval.longmemeval_v2` is a local, + text-only adapter for the official LongMemEval-V2 `insert(trajectory)` / `query(query, + query_image=None)` memory interface; it does not download data or call a model. +- **Grounded**: `eval/grounded.py`: answerable → cite, off-topic → abstain. Exact fixture + outcomes are evidence ID `offline-grounded` in the registry above. +- **Chunking (quality per token)**: `eval/chunking_eval.py` over `eval/datasets/longdoc.jsonl` + ingests a multi-topic corpus twice: once as one memory per document (`whole`) and once with + sub-file `ChunkingExtractor` (`chunked`), then queries both through the real recall pipeline. + The checked-in corpus is explicitly marked trusted eval data so the measurement isolates + chunking from the production trust gate, which excludes arbitrary raw imports from normal + agent context. On the deterministic embedder, **recall@5 is 1.000 for both modes; mean + retrieved top-5 content falls from 740.3 to 214.3 tokens (526.0 fewer, 71.1% lower, about + 3.5× smaller), while the smallest returned evidence-holding memory falls from 162.2 to 42.4 + tokens (119.8 fewer, 73.9% lower, about 3.8× smaller).** These aggregates are evidence ID + `offline-chunking` in the registry above. Pass `--embed-model + sentence-transformers/all-MiniLM-L6-v2` to run a model-dependent experiment; do not publish + that result without a new immutable artifact and pinned model revision. +- **Full-pipeline latency + quality**: `eval/performance.py` times the shipped semantic + + lexical + graph + fusion + scoring + rerank + packing path after warmup, with reinforcement + disabled so repeated measurements do not mutate their corpus. It reports p50/p95/p99 latency, + retrieval quality, packed context tokens, and full/compact JSON-shape payload proxies in one + JSON-safe schema. Its legacy `quality` fields score all candidate chunks returned before + context packing; additive `packed_quality` fields score only chunks admitted to reader context. + Payload proxies are sampled once per question, independently of the number of timed iterations; + they are not serialized MCP envelopes or transport responses. In the registered CodeMem run, 26 payload samples total **24,590** full-proxy `engraphis.regex.v1` tokens versus **11,138** compact-proxy tokens, avoiding **13,452** proxy tokens (**54.71% lower**), while 260 recalls are timed. Packed context across the same 26 - samples averages **85.38** tokens and reaches **108** under a 1,500-token cap; Recall@5, - hit@5, and answer-token recall remain 1.000 for the legacy candidate-page view. The registered - v9 artifact predates `packed_quality`, so no packed-quality aggregate is published from it. - These aggregates are evidence ID - `offline-performance` in the registry above. `--filler-memories`, `--candidate-k`, and - `--retrieval-profile` make scaling and routing experiments executable, but their results need - separate evidence before publication. -- **Exact vector scale envelope**: `eval/vector_scale.py` measures the production - `NumpyVectorIndex` directly at requested corpus sizes with deterministic normalized vectors and - queries. It records a corpus fingerprint, result hashes, environment, and observed - p50/p95/p99 search envelopes. It intentionally has no pass/fail latency threshold: the output - describes the measured machine and workload, not a universal capacity cutoff. Pair it with - `eval/performance.py` before making a deployment decision because direct vector search excludes - the rest of the recall pipeline. Its `engraphis-vector-scale/v1` JSON is a local diagnostic, not - an `engraphis-benchmark/v2` public evidence artifact. -- **Proactive ranking calibration**: `eval/proactive_ranking.py` compares the previous and current - importance-retention floors on a small deterministic queryless-ranking fixture. It reports - top-1 accuracy and minimum expected margins for that fixture only. It is a scoring regression, - not evidence of general recall quality or user-task performance. -- **Workload context economy**: `eval/context_economy.py` compares three executable strategies - across every question in a workload: uncapped full-history replay, a contiguous recency window - at the same hard budget, and shipped Engraphis hybrid recall + packing. It reports evidence and - answer-token quality, cumulative reader-context tokens, a conservative total that charges one - complete source-token pass to indexing, and the query-count break-even point. The default is - deterministic/offline; `--embed-model` enables a real retrieval model, while - `--format locomo|longmemeval` reuses the established external loaders. -- **Agent productivity**: `eval/productivity.py` compares a capped full-history baseline, - always-on retrieval, and - adaptive context through a complete answer-and-correction loop. It reports completed tasks, - first-attempt errors, abstentions, corrections, agent turns, memory calls, wall-clock latency, - and all question/context/output tokens. The bundled agent is deterministic, receives no gold - answer, and is identified in every report; inject a real agent callable for model-specific - results. Optional provider telemetry is reported separately from the deterministic token - counter and is not a provider billing estimate. -- **Dashboard graph layout settle**: `eval/graph_every_bench.py` drives the Every-node - dashboard engine's real worker (`engraphis-graph-every-worker.js`) through a - `prepare → settled` round-trip over synthetic node/link loads and reports wall-clock settle - time plus the scaling ratio across sizes. It measures initial layout cost only: camera pans - and zooms never touch the worker (they are GPU-uniform updates), so no per-frame number can - come out of this harness and none should be quoted. Results are host- and Node-version - dependent local diagnostics, not registered public evidence; run the harness on the target - class of machine before quoting a figure. - -The context-economy and productivity tools intentionally report when a small workload does not -benefit from memory, and the external loaders expose retrieval-quality tradeoffs rather than -hiding them. Their prior local results are not retained as public numbers because no matching -redacted immutable artifact is checked in. Run the registered protocol and publish the resulting -artifact before making a quantitative claim. - -### Reproduce - -```bash -# Correctness gate (deterministic, no download) -python -m pytest tests/ -q -python -m eval.harness --dataset eval/datasets/sample.jsonl --k 5 -python -m eval.harness --dataset eval/datasets/codemem.jsonl --k 5 -python -m eval.harness --dataset eval/datasets/graph_multihop.jsonl --k 5 -python -m eval.ablation -python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 -python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 \ - --candidate-k 25 --candidate-depth adaptive --retrieval-profile auto --iterations 10 -python -m eval.context_economy --dataset eval/datasets/codemem.jsonl \ - --token-budget 512 --k 5 -python -m eval.productivity --dataset eval/datasets/codemem.jsonl \ - --max-context-tokens 512 --retrieval-token-budget 256 -python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 \ - --iterations 5 --filler-memories 1000 -# Direct NumPy search envelope at representative corpus sizes; timings are machine-specific. -python -m eval.vector_scale --sizes 1000,10000,100000 --queries 20 --iterations 3 --json -# Deterministic queryless-ranking calibration fixture. -python -m eval.proactive_ranking -# Canonical latency/resource protocol: requires >=1,000 queries and five processes. -python -m eval.performance --dataset fixed-1000-plus.jsonl --acceptance-matrix --processes 5 - -# External retrieval diagnostics (downloads all-MiniLM-L6-v2; not QA/leaderboard results) -python -m eval.external --dataset longmemeval_s.json --format longmemeval --k 10 -python -m eval.external --dataset locomo10.json --format locomo --k 10 -# Complete external-dataset coverage with an immutable embedding revision. These runs are -# retrieval-only diagnostics, not official benchmark-harness or leaderboard results. Completed -# public-safe artifacts and measured results are listed in the benchmark expansion report. -python -m eval.external --dataset longmemeval_s.json --format longmemeval --canonical \ - --embed-revision <40-character-model-commit> --json external-longmemeval.json -python -m eval.external --dataset locomo10.json --format locomo --canonical --no-resolve \ - --embed-revision <40-character-model-commit> \ - --locomo-repair-manifest eval/datasets/locomo10_repair_manifest.json \ - --json external-locomo.json -python -m eval.context_economy --dataset locomo10.json --format locomo \ - --embed-model sentence-transformers/all-MiniLM-L6-v2 --token-budget 512 --k 10 --no-resolve -``` - -Canonical external mode requires an exact lowercase 40-character embedding commit and a semantic -embedder; dependency or model-load failure is fatal instead of silently falling back to hashing. -Every report records `embedding`, `dataset_sha256`, `source_cases`, `normalized_cases`, and -`configuration` provenance so a result can be attributed to the actual data and retrieval setup. - -The official ten-conversation LoCoMo JSON contains delimiter-packed IDs, two mechanical ID -typos, and three references that cannot be normalized syntactically. The adapter normalizes only -the unambiguous forms. The checked-in repair manifest is bound to the official source SHA-256, -names every remaining replacement/removal, must be fully consumed, and is recorded in the JSON -report with its own hash. Any source update, unused repair, or unresolved ID fails the run. This -repairs retrieval references only; it does not claim to correct LoCoMo's semantic answer labels. - -The earlier private pinned retrieval diagnostic was the pre-publication state. Current complete -LoCoMo and LongMemEval retrieval diagnostics are retained as separate public-safe artifacts in the -[benchmark expansion report](docs/BENCHMARK_EXPANSION_RESULTS.md), with source, model, configuration -and checksum boundaries. Those values remain evidence-retrieval metrics, not end-to-end QA accuracy -or an official LoCoMo leaderboard score. - -## What we do NOT yet claim - -- **No official end-to-end LLM QA accuracy.** The deterministic productivity agent measures the - complete local control loop, not a frontier answering model. Official LoCoMo / LongMemEval QA - still requires a pinned answering model and evaluator. -- **No hosted-service latency comparison.** The in-repo p50/p95/p99 benchmark covers the local - reference pipeline and records its environment; unlike environments are not compared. -- **No neutral third-party ranking.** We have not run an external eval platform. -- **No provider bill estimate.** Context-economy counts reader evidence under its named counter. - It excludes system/tool prompts, questions, completions, prompt caching, provider pricing, - compute, and storage. Its indexing-inclusive total is a conservative text-volume proxy. - -Every publishable run should emit the `engraphis-benchmark/v2` envelope: dataset/config hashes, -per-question records, explicit exclusions, fixed-budget context curves, and deterministic -stratified or paired bootstrap confidence intervals. Every run names its token counter. -Noncanonical offline fixtures may identify a deterministic estimate; canonical public evidence -requires the exact pinned reader tokenizer and immutable model revision. The lightweight CI + samples averages **85.38** tokens and reaches **108** under a 1,500-token cap; Recall@5, + hit@5, and answer-token recall remain 1.000 for the legacy candidate-page view. The registered + v9 artifact predates `packed_quality`, so no packed-quality aggregate is published from it. + These aggregates are evidence ID + `offline-performance` in the registry above. `--filler-memories`, `--candidate-k`, and + `--retrieval-profile` make scaling and routing experiments executable, but their results need + separate evidence before publication. +- **Exact vector scale envelope**: `eval/vector_scale.py` measures the production + `NumpyVectorIndex` directly at requested corpus sizes with deterministic normalized vectors and + queries. It records a corpus fingerprint, result hashes, environment, and observed + p50/p95/p99 search envelopes. It intentionally has no pass/fail latency threshold: the output + describes the measured machine and workload, not a universal capacity cutoff. Pair it with + `eval/performance.py` before making a deployment decision because direct vector search excludes + the rest of the recall pipeline. Its `engraphis-vector-scale/v1` JSON is a local diagnostic, not + an `engraphis-benchmark/v2` public evidence artifact. +- **Proactive ranking calibration**: `eval/proactive_ranking.py` compares the previous and current + importance-retention floors on a small deterministic queryless-ranking fixture. It reports + top-1 accuracy and minimum expected margins for that fixture only. It is a scoring regression, + not evidence of general recall quality or user-task performance. +- **Workload context economy**: `eval/context_economy.py` compares three executable strategies + across every question in a workload: uncapped full-history replay, a contiguous recency window + at the same hard budget, and shipped Engraphis hybrid recall + packing. It reports evidence and + answer-token quality, cumulative reader-context tokens, a conservative total that charges one + complete source-token pass to indexing, and the query-count break-even point. The default is + deterministic/offline; `--embed-model` enables a real retrieval model, while + `--format locomo|longmemeval` reuses the established external loaders. +- **Agent productivity**: `eval/productivity.py` compares a capped full-history baseline, + always-on retrieval, and + adaptive context through a complete answer-and-correction loop. It reports completed tasks, + first-attempt errors, abstentions, corrections, agent turns, memory calls, wall-clock latency, + and all question/context/output tokens. The bundled agent is deterministic, receives no gold + answer, and is identified in every report; inject a real agent callable for model-specific + results. Optional provider telemetry is reported separately from the deterministic token + counter and is not a provider billing estimate. +- **Dashboard graph layout settle**: `eval/graph_every_bench.py` drives the Every-node + dashboard engine's real worker (`engraphis-graph-every-worker.js`) through a + `prepare → settled` round-trip over synthetic node/link loads and reports wall-clock settle + time plus the scaling ratio across sizes. It measures initial layout cost only: camera pans + and zooms never touch the worker (they are GPU-uniform updates), so no per-frame number can + come out of this harness and none should be quoted. Results are host- and Node-version + dependent local diagnostics, not registered public evidence; run the harness on the target + class of machine before quoting a figure. + +The context-economy and productivity tools intentionally report when a small workload does not +benefit from memory, and the external loaders expose retrieval-quality tradeoffs rather than +hiding them. Their prior local results are not retained as public numbers because no matching +redacted immutable artifact is checked in. Run the registered protocol and publish the resulting +artifact before making a quantitative claim. + +### Reproduce + +```bash +# Correctness gate (deterministic, no download) +python -m pytest tests/ -q +python -m eval.harness --dataset eval/datasets/sample.jsonl --k 5 +python -m eval.harness --dataset eval/datasets/codemem.jsonl --k 5 +python -m eval.harness --dataset eval/datasets/graph_multihop.jsonl --k 5 +python -m eval.ablation +python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 +python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 \ + --candidate-k 25 --candidate-depth adaptive --retrieval-profile auto --iterations 10 +python -m eval.context_economy --dataset eval/datasets/codemem.jsonl \ + --token-budget 512 --k 5 +python -m eval.productivity --dataset eval/datasets/codemem.jsonl \ + --max-context-tokens 512 --retrieval-token-budget 256 +python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 \ + --iterations 5 --filler-memories 1000 +# Direct NumPy search envelope at representative corpus sizes; timings are machine-specific. +python -m eval.vector_scale --sizes 1000,10000,100000 --queries 20 --iterations 3 --json +# Deterministic queryless-ranking calibration fixture. +python -m eval.proactive_ranking +# Canonical latency/resource protocol: requires >=1,000 queries and five processes. +python -m eval.performance --dataset fixed-1000-plus.jsonl --acceptance-matrix --processes 5 + +# External retrieval diagnostics (downloads all-MiniLM-L6-v2; not QA/leaderboard results) +python -m eval.external --dataset longmemeval_s.json --format longmemeval --k 10 +python -m eval.external --dataset locomo10.json --format locomo --k 10 +# Complete external-dataset coverage with an immutable embedding revision. These runs are +# retrieval-only diagnostics, not official benchmark-harness or leaderboard results. Completed +# public-safe artifacts and measured results are listed in the benchmark expansion report. +python -m eval.external --dataset longmemeval_s.json --format longmemeval --canonical \ + --embed-revision <40-character-model-commit> --json external-longmemeval.json +python -m eval.external --dataset locomo10.json --format locomo --canonical --no-resolve \ + --embed-revision <40-character-model-commit> \ + --locomo-repair-manifest eval/datasets/locomo10_repair_manifest.json \ + --json external-locomo.json +python -m eval.context_economy --dataset locomo10.json --format locomo \ + --embed-model sentence-transformers/all-MiniLM-L6-v2 --token-budget 512 --k 10 --no-resolve +``` + +Canonical external mode requires an exact lowercase 40-character embedding commit and a semantic +embedder; dependency or model-load failure is fatal instead of silently falling back to hashing. +Every report records `embedding`, `dataset_sha256`, `source_cases`, `normalized_cases`, and +`configuration` provenance so a result can be attributed to the actual data and retrieval setup. + +The official ten-conversation LoCoMo JSON contains delimiter-packed IDs, two mechanical ID +typos, and three references that cannot be normalized syntactically. The adapter normalizes only +the unambiguous forms. The checked-in repair manifest is bound to the official source SHA-256, +names every remaining replacement/removal, must be fully consumed, and is recorded in the JSON +report with its own hash. Any source update, unused repair, or unresolved ID fails the run. This +repairs retrieval references only; it does not claim to correct LoCoMo's semantic answer labels. + +The earlier private pinned retrieval diagnostic was the pre-publication state. Current complete +LoCoMo and LongMemEval retrieval diagnostics are retained as separate public-safe artifacts in the +[benchmark expansion report](docs/BENCHMARK_EXPANSION_RESULTS.md), with source, model, configuration +and checksum boundaries. Those values remain evidence-retrieval metrics, not end-to-end QA accuracy +or an official LoCoMo leaderboard score. + +## What we do NOT yet claim + +- **No official end-to-end LLM QA accuracy.** The deterministic productivity agent measures the + complete local control loop, not a frontier answering model. Official LoCoMo / LongMemEval QA + still requires a pinned answering model and evaluator. +- **No hosted-service latency comparison.** The in-repo p50/p95/p99 benchmark covers the local + reference pipeline and records its environment; unlike environments are not compared. +- **No neutral third-party ranking.** We have not run an external eval platform. +- **No provider bill estimate.** Context-economy counts reader evidence under its named counter. + It excludes system/tool prompts, questions, completions, prompt caching, provider pricing, + compute, and storage. Its indexing-inclusive total is a conservative text-volume proxy. + +Every publishable run should emit the `engraphis-benchmark/v2` envelope: dataset/config hashes, +per-question records, explicit exclusions, fixed-budget context curves, and deterministic +stratified or paired bootstrap confidence intervals. Every run names its token counter. +Noncanonical offline fixtures may identify a deterministic estimate; canonical public evidence +requires the exact pinned reader tokenizer and immutable model revision. The lightweight CI fixtures validate that machinery; they are not a claim about external benchmark performance. Public journey and external retrieval exports identify producer code by unique @@ -316,183 +316,183 @@ LongMemEval-V2 inputs use stable role names such as `inputs/dataset` or Each name remains bound to its SHA-256 and byte count. The shared envelope keeps basename-only defaults for other callers and historical artifacts; exporters opt in with explicit `source_names` and verify the completed envelope against evaluated bytes. - -The benchmark context metric reads strict recall usage fields rather than inferring prompt size: -`budget_tokens`, `context_tokens`, `source_tokens`, `saved_tokens`, `savings_ratio`, -`packed_count`, `omitted_count`, and `token_counter`. Use `engraphis_recall_context` for a -hard-budget prompt packet; legacy `engraphis_recall` remains available in full or compact response -mode for compatibility. - -### Canonical public artifacts - -Use `python -m eval.benchmark --input report.json --output artifacts/run.json` to validate a -report and write sorted, immutable JSON plus `run.json.sha256`. The command permits an identical -retry but refuses to replace a different artifact at the same path. For an official -LongMemEval-V2 run, add `--canonical`: this requires a profile with an exact benchmark repository -revision, dataset revision, reader model revision, and embedding model revision. The checked-in -profile pins immutable upstream commits; replacing any revision with a mutable tag fails -validation. Canonical profiles label the baseline (`no_retrieval`, `lexical_only`, `dense_only`, -`dense_lexical_rrf`, `full_hybrid`, `full_history`, `no_graph`, `no_reranker`, -`no_temporal_resolution`, or `whole_document`) and declare the required fixed context-budget -matrix: 256, 512, 1024, 2048, and 4096 tokens. Canonical in-repo reports rerun every question at -all five budgets and validate each aggregate against its per-question evidence. The checked-in -LongMemEval-V2 memory-module configuration sets the official adapter's operating point to 1,024 -tokens; that single official point must not be presented as a five-point curve. - -`eval.external --canonical` refuses `--limit` and rejects a normalized output that omitted source -cases. Retrieval-only abstention/no-evidence records remain visible in the artifact's -`exclusions`; they are not counted as evidence-retrieval scores. - -Official LongMemEval-V2 output can be converted into a public-safe QA artifact with -`python -m eval.longmemeval_v2_evidence`. The exporter requires the completion manifest written by -the pinned runner after a successful, complete official run. It binds the exact per-question -output, questions, haystack, trajectories, memory configuration, matrix manifest, seed, clean -official checkout, and recorded environment. The public artifact keeps the official QA score, -fixed-reader context token count, aggregate source-file digests, repository state, and artifact -checksum. It removes raw questions, answers, prompts, reader output, and retrieved context, and -does not publish per-record content fingerprints. See the -[`public benchmark runbook`](docs/PUBLIC_BENCHMARK_RUNBOOK.md) for the end-to-end operator sequence. - -### LongMemEval-V2 memory-module adapter - -`eval.longmemeval_v2.EngraphisLongMemEvalV2Memory` follows the official -`memory_modules.memory.Memory` interface at LongMemEval-V2 commit -`6f020ac2fc3275e46c706d3406e02c3ed79b7be2`. When imported in that environment, its -`@register_memory` decorator registers `memory_type="engraphis"`; use the checked-in -[`eval/configs/longmemeval_v2_engraphis.json`](eval/configs/longmemeval_v2_engraphis.json) -with the official harness. The config pins `Qwen/Qwen3-Embedding-8B` to revision -`1d8ad4ca9b3dd8059ad90a75d4983776a23d44af`; mutable embedding revisions are rejected, and a -canonical adapter run fails instead of relabeling the deterministic offline fallback as Qwen. -First materialize the six declared variants at all five token budgets: - -```bash -python -m eval.longmemeval_v2_matrix \ - --output "$ENGRAPHIS_EVIDENCE_RUN_DIR/configs" -``` - -This writes a 30-run manifest: balanced, planner, episodic-cap, planner-plus-episodic-cap, and -matched `context_k=2` comparators for both capped variants, each at 256, 512, 1,024, 2,048, and -4,096 evidence tokens. Run each manifest cell through `python -m eval.run_longmemeval_v2` with all -eight `--engraphis-*` completion-receipt arguments. The wrapper imports the adapter before the -official registry builds the memory module, forces the pinned reader processor revision, and -delegates the remaining official harness arguments unchanged. Only after a successful return does -it verify that the output question IDs exactly cover the source question IDs and write the -immutable execution manifest. - -The checked-in configuration is canonical only when the adapter resolves the pinned Qwen reader -processor at `c202236235762e1c871ad0ccb60c8ee5ba337b9a`. The wrapper refuses a dirty or non-pinned -official checkout and refuses to start if the optional processor dependency or immutable revision -is unavailable; the local regex counter is never silently relabeled as a reader budget. The -recorded budget counts each returned context item's content with that reader tokenizer (without -prompt framing or inter-item separators), so it is a hard **evidence-item content** budget, not a -claim about total chat-prompt tokens. Packed sources are returned as separate context items, -preserving the largest fitting evidence prefix instead of dropping one oversized monolithic item. -Every official per-question row reports inserted and retrieved counts by memory type. A -memory-type-cap claim additionally requires at least two populated inserted types, so a nominal cap -over a single-type workload cannot qualify as evidence. The adapter does not download benchmark -data or call the reader/evaluator; the official harness owns those steps. - -## External evidence status and remaining executions - -1. **Run the official LongMemEval-V2 reader and evaluator.** The adapter, pinned runner, and - redacted evidence exporter are implemented. The exact upstream commit boots in an isolated - Python 3.11 environment and the wrapper reaches the official harness CLI. The dataset, pinned - Qwen reader, and embedding assets require substantial storage and compute; no canonical QA - score is claimed until that run completes. -2. **Publish production-backend latency.** Run `eval/performance.py` with the real embedder and - sqlite-vec/backend configuration on a fixed machine class and corpus scale. -3. **Run the fixed-budget curve on the complete official datasets.** The v2 harness now measures - every question at 256, 512, 1,024, 2,048, and 4,096 evidence tokens and validates the - per-question records, aggregates, and pinned reader-tokenizer identity. Publish the curve only - after complete official runs produce immutable artifacts for every point. -4. **Run an external evaluation platform** once (1)–(3) exist. - -Do not make all evidence lanes variants of explicit factual recall. Executable offline adapters -now cover: - -- [MemoryAgentBench](https://github.com/HUST-AI-HYZ/MemoryAgentBench): incremental multi-turn - learning, long-range understanding, and conflict/consolidation inputs. -- [LoCoMo-Plus](https://github.com/xjtuleeyf/Locomo-Plus): an old implicit constraint must affect - a later response even when the later cue does not restate the remembered fact. -- [Mem2ActBench](https://github.com/Cantaloupe-M/Mem2ActBench): memory must select a tool and - ground its arguments, not merely return a passage. The current adapter measures retrieval and - expected tool-argument context coverage, not generated tool-call success. - -```bash -python -m eval.agent_benchmarks --dataset memoryagentbench.json \ - --format memoryagentbench -python -m eval.agent_benchmarks --dataset locomo_plus.json \ - --format locomo_plus -python -m eval.agent_benchmarks --dataset qa_dataset.jsonl \ - --conversations toolmem_conversation.jsonl --format mem2actbench \ - --artifact artifacts/mem2actbench.json -``` - -Use `--artifact` on any of these commands to write a redacted, immutable evidence envelope plus -an adjacent SHA256 file. The ordinary console/`--json` report is private run material and may -contain source questions for debugging. - -### Upstream-data diagnostics and publication scope - -The LoCoMo-Plus and MemoryAgentBench adapters have been exercised against upstream data and remain -queued for their own public-safe retrieval artifacts. Rerun each pending adapter with `--artifact`, -publish the redacted immutable envelope and checksum, and add its suite/config binding before -quoting a number. Mem2ActBench's declared small retrieval diagnostic is complete and has a checked-in -artifact; its exclusion and memory-cardinality figures are source-preparation metadata in the -public source lock, not product or action-success metrics. None of these lanes is an official -leaderboard, answer-quality, or marketing result. - -The MemoryAgentBench loader accepts both its aligned public JSON export and the Hugging Face -dataset-server `rows[].row` envelope. Rows without gold evidence remain useful for answer-token -coverage, but are excluded from retrieval aggregates and counted separately as -`retrieval_scored_questions`. - -For paired code-agent runs, execute the same tasks with the same model, tools, machine, and -deterministic success oracle under `full_history` and `engraphis`. Then analyze the content-free -run records with: - -```bash -python -m eval.code_agent_ab --full-history full-history.jsonl \ - --engraphis engraphis.jsonl --output paired-report.json -``` - -The analyzer rejects unmatched task IDs and different success oracles, then reports paired -bootstrap intervals for task success, input/output/tool tokens, retries, latency, and optional -cost. Its aggregate output does not echo task IDs or oracle commands. It does not launch an agent -or invent a task-success oracle. - -## Optimization experiments to run before changing defaults - -1. **Budget-aware packing**: compare full source, safe summary, sentence-aligned safe summary - excerpt, and raw-source excerpt at fixed budgets. Gate on support/answer retention and - qualifier preservation, not token count alone. -2. **Adaptive retrieval work**: `--candidate-depth adaptive` is an opt-in performance experiment. - It keeps wider graph/code pools and reduces routine lexical/balanced pools while reporting the - requested and actual depth. A local experiment motivated this option, but no public number is - retained because its machine-specific artifact is not in the evidence registry. Keep the - default fixed until complete external categories meet predeclared quality margins. -3. **Packing-pressure consolidation**: prioritize memory families that are frequently recalled, - repeatedly omitted, or costly per useful token. Count write/index/storage cost as well as later - reader-context savings. -4. **Tokenizer-aware ingestion**: implemented behind the chunk extractor. The dependency-free - default remains `engraphis.chars4.v1`; an explicitly configured Hugging Face reader tokenizer - enforces prose chunk and overlap budgets and records its identity in chunk metadata. Continue - measuring tokens-to-evidence, recall, and storage/index growth together before recommending a - model-specific default. -5. **Bulk ingestion**: add batch embedding plus a transaction-aware vector upsert path, then rerun - the complete MemoryAgentBench Test-Time Learning input. Gate this on identical stored-memory, - provenance, graph-link, and temporal-resolution outcomes, not throughput alone. -6. **Scoped caches**: benchmark query embeddings and repeat-recall results keyed by workspace, - repo, time anchors, profile, and corpus version. Test invalidation correctness before claiming - latency gains. -7. **Privacy-safe real usage**: use `engraphis_context_savings` to let each workspace inspect - aggregate source/context/saved tokens already present in content-free receipts. Keep unlike - token counters separate and require a valid receipt chain before treating totals as auditable. - -## Evaluation question - -The predeclared question is whether the full vector + lexical/BM25 + sparse PPR graph + calibrated -rerank pipeline, bi-temporal resolution, and grounded abstention produce higher evidence recall -per injected token than the registered baselines. The answer must come from a complete, -machine-readable artifact with paired confidence intervals; otherwise the release reports -“no demonstrated improvement.” + +The benchmark context metric reads strict recall usage fields rather than inferring prompt size: +`budget_tokens`, `context_tokens`, `source_tokens`, `saved_tokens`, `savings_ratio`, +`packed_count`, `omitted_count`, and `token_counter`. Use `engraphis_recall_context` for a +hard-budget prompt packet; legacy `engraphis_recall` remains available in full or compact response +mode for compatibility. + +### Canonical public artifacts + +Use `python -m eval.benchmark --input report.json --output artifacts/run.json` to validate a +report and write sorted, immutable JSON plus `run.json.sha256`. The command permits an identical +retry but refuses to replace a different artifact at the same path. For an official +LongMemEval-V2 run, add `--canonical`: this requires a profile with an exact benchmark repository +revision, dataset revision, reader model revision, and embedding model revision. The checked-in +profile pins immutable upstream commits; replacing any revision with a mutable tag fails +validation. Canonical profiles label the baseline (`no_retrieval`, `lexical_only`, `dense_only`, +`dense_lexical_rrf`, `full_hybrid`, `full_history`, `no_graph`, `no_reranker`, +`no_temporal_resolution`, or `whole_document`) and declare the required fixed context-budget +matrix: 256, 512, 1024, 2048, and 4096 tokens. Canonical in-repo reports rerun every question at +all five budgets and validate each aggregate against its per-question evidence. The checked-in +LongMemEval-V2 memory-module configuration sets the official adapter's operating point to 1,024 +tokens; that single official point must not be presented as a five-point curve. + +`eval.external --canonical` refuses `--limit` and rejects a normalized output that omitted source +cases. Retrieval-only abstention/no-evidence records remain visible in the artifact's +`exclusions`; they are not counted as evidence-retrieval scores. + +Official LongMemEval-V2 output can be converted into a public-safe QA artifact with +`python -m eval.longmemeval_v2_evidence`. The exporter requires the completion manifest written by +the pinned runner after a successful, complete official run. It binds the exact per-question +output, questions, haystack, trajectories, memory configuration, matrix manifest, seed, clean +official checkout, and recorded environment. The public artifact keeps the official QA score, +fixed-reader context token count, aggregate source-file digests, repository state, and artifact +checksum. It removes raw questions, answers, prompts, reader output, and retrieved context, and +does not publish per-record content fingerprints. See the +[`public benchmark runbook`](docs/PUBLIC_BENCHMARK_RUNBOOK.md) for the end-to-end operator sequence. + +### LongMemEval-V2 memory-module adapter + +`eval.longmemeval_v2.EngraphisLongMemEvalV2Memory` follows the official +`memory_modules.memory.Memory` interface at LongMemEval-V2 commit +`6f020ac2fc3275e46c706d3406e02c3ed79b7be2`. When imported in that environment, its +`@register_memory` decorator registers `memory_type="engraphis"`; use the checked-in +[`eval/configs/longmemeval_v2_engraphis.json`](eval/configs/longmemeval_v2_engraphis.json) +with the official harness. The config pins `Qwen/Qwen3-Embedding-8B` to revision +`1d8ad4ca9b3dd8059ad90a75d4983776a23d44af`; mutable embedding revisions are rejected, and a +canonical adapter run fails instead of relabeling the deterministic offline fallback as Qwen. +First materialize the six declared variants at all five token budgets: + +```bash +python -m eval.longmemeval_v2_matrix \ + --output "$ENGRAPHIS_EVIDENCE_RUN_DIR/configs" +``` + +This writes a 30-run manifest: balanced, planner, episodic-cap, planner-plus-episodic-cap, and +matched `context_k=2` comparators for both capped variants, each at 256, 512, 1,024, 2,048, and +4,096 evidence tokens. Run each manifest cell through `python -m eval.run_longmemeval_v2` with all +eight `--engraphis-*` completion-receipt arguments. The wrapper imports the adapter before the +official registry builds the memory module, forces the pinned reader processor revision, and +delegates the remaining official harness arguments unchanged. Only after a successful return does +it verify that the output question IDs exactly cover the source question IDs and write the +immutable execution manifest. + +The checked-in configuration is canonical only when the adapter resolves the pinned Qwen reader +processor at `c202236235762e1c871ad0ccb60c8ee5ba337b9a`. The wrapper refuses a dirty or non-pinned +official checkout and refuses to start if the optional processor dependency or immutable revision +is unavailable; the local regex counter is never silently relabeled as a reader budget. The +recorded budget counts each returned context item's content with that reader tokenizer (without +prompt framing or inter-item separators), so it is a hard **evidence-item content** budget, not a +claim about total chat-prompt tokens. Packed sources are returned as separate context items, +preserving the largest fitting evidence prefix instead of dropping one oversized monolithic item. +Every official per-question row reports inserted and retrieved counts by memory type. A +memory-type-cap claim additionally requires at least two populated inserted types, so a nominal cap +over a single-type workload cannot qualify as evidence. The adapter does not download benchmark +data or call the reader/evaluator; the official harness owns those steps. + +## External evidence status and remaining executions + +1. **Run the official LongMemEval-V2 reader and evaluator.** The adapter, pinned runner, and + redacted evidence exporter are implemented. The exact upstream commit boots in an isolated + Python 3.11 environment and the wrapper reaches the official harness CLI. The dataset, pinned + Qwen reader, and embedding assets require substantial storage and compute; no canonical QA + score is claimed until that run completes. +2. **Publish production-backend latency.** Run `eval/performance.py` with the real embedder and + sqlite-vec/backend configuration on a fixed machine class and corpus scale. +3. **Run the fixed-budget curve on the complete official datasets.** The v2 harness now measures + every question at 256, 512, 1,024, 2,048, and 4,096 evidence tokens and validates the + per-question records, aggregates, and pinned reader-tokenizer identity. Publish the curve only + after complete official runs produce immutable artifacts for every point. +4. **Run an external evaluation platform** once (1)–(3) exist. + +Do not make all evidence lanes variants of explicit factual recall. Executable offline adapters +now cover: + +- [MemoryAgentBench](https://github.com/HUST-AI-HYZ/MemoryAgentBench): incremental multi-turn + learning, long-range understanding, and conflict/consolidation inputs. +- [LoCoMo-Plus](https://github.com/xjtuleeyf/Locomo-Plus): an old implicit constraint must affect + a later response even when the later cue does not restate the remembered fact. +- [Mem2ActBench](https://github.com/Cantaloupe-M/Mem2ActBench): memory must select a tool and + ground its arguments, not merely return a passage. The current adapter measures retrieval and + expected tool-argument context coverage, not generated tool-call success. + +```bash +python -m eval.agent_benchmarks --dataset memoryagentbench.json \ + --format memoryagentbench +python -m eval.agent_benchmarks --dataset locomo_plus.json \ + --format locomo_plus +python -m eval.agent_benchmarks --dataset qa_dataset.jsonl \ + --conversations toolmem_conversation.jsonl --format mem2actbench \ + --artifact artifacts/mem2actbench.json +``` + +Use `--artifact` on any of these commands to write a redacted, immutable evidence envelope plus +an adjacent SHA256 file. The ordinary console/`--json` report is private run material and may +contain source questions for debugging. + +### Upstream-data diagnostics and publication scope + +The LoCoMo-Plus and MemoryAgentBench adapters have been exercised against upstream data and remain +queued for their own public-safe retrieval artifacts. Rerun each pending adapter with `--artifact`, +publish the redacted immutable envelope and checksum, and add its suite/config binding before +quoting a number. Mem2ActBench's declared small retrieval diagnostic is complete and has a checked-in +artifact; its exclusion and memory-cardinality figures are source-preparation metadata in the +public source lock, not product or action-success metrics. None of these lanes is an official +leaderboard, answer-quality, or marketing result. + +The MemoryAgentBench loader accepts both its aligned public JSON export and the Hugging Face +dataset-server `rows[].row` envelope. Rows without gold evidence remain useful for answer-token +coverage, but are excluded from retrieval aggregates and counted separately as +`retrieval_scored_questions`. + +For paired code-agent runs, execute the same tasks with the same model, tools, machine, and +deterministic success oracle under `full_history` and `engraphis`. Then analyze the content-free +run records with: + +```bash +python -m eval.code_agent_ab --full-history full-history.jsonl \ + --engraphis engraphis.jsonl --output paired-report.json +``` + +The analyzer rejects unmatched task IDs and different success oracles, then reports paired +bootstrap intervals for task success, input/output/tool tokens, retries, latency, and optional +cost. Its aggregate output does not echo task IDs or oracle commands. It does not launch an agent +or invent a task-success oracle. + +## Optimization experiments to run before changing defaults + +1. **Budget-aware packing**: compare full source, safe summary, sentence-aligned safe summary + excerpt, and raw-source excerpt at fixed budgets. Gate on support/answer retention and + qualifier preservation, not token count alone. +2. **Adaptive retrieval work**: `--candidate-depth adaptive` is an opt-in performance experiment. + It keeps wider graph/code pools and reduces routine lexical/balanced pools while reporting the + requested and actual depth. A local experiment motivated this option, but no public number is + retained because its machine-specific artifact is not in the evidence registry. Keep the + default fixed until complete external categories meet predeclared quality margins. +3. **Packing-pressure consolidation**: prioritize memory families that are frequently recalled, + repeatedly omitted, or costly per useful token. Count write/index/storage cost as well as later + reader-context savings. +4. **Tokenizer-aware ingestion**: implemented behind the chunk extractor. The dependency-free + default remains `engraphis.chars4.v1`; an explicitly configured Hugging Face reader tokenizer + enforces prose chunk and overlap budgets and records its identity in chunk metadata. Continue + measuring tokens-to-evidence, recall, and storage/index growth together before recommending a + model-specific default. +5. **Bulk ingestion**: add batch embedding plus a transaction-aware vector upsert path, then rerun + the complete MemoryAgentBench Test-Time Learning input. Gate this on identical stored-memory, + provenance, graph-link, and temporal-resolution outcomes, not throughput alone. +6. **Scoped caches**: benchmark query embeddings and repeat-recall results keyed by workspace, + repo, time anchors, profile, and corpus version. Test invalidation correctness before claiming + latency gains. +7. **Privacy-safe real usage**: use `engraphis_context_savings` to let each workspace inspect + aggregate source/context/saved tokens already present in content-free receipts. Keep unlike + token counters separate and require a valid receipt chain before treating totals as auditable. + +## Evaluation question + +The predeclared question is whether the full vector + lexical/BM25 + sparse PPR graph + calibrated +rerank pipeline, bi-temporal resolution, and grounded abstention produce higher evidence recall +per injected token than the registered baselines. The answer must come from a complete, +machine-readable artifact with paired confidence intervals; otherwise the release reports +“no demonstrated improvement.” diff --git a/README.md b/README.md index 62d36693..f290bf88 100644 --- a/README.md +++ b/README.md @@ -1,561 +1,561 @@ -# Engraphis - -[![PyPI version](https://img.shields.io/pypi/v/engraphis.svg)](https://pypi.org/project/engraphis/) -[![License](https://img.shields.io/badge/license-Apache--2.0-green.svg)](https://github.com/Coding-Dev-Tools/engraphis/blob/main/LICENSE) -[![Support](https://img.shields.io/badge/Buy%20Me%20a%20Coffee-support-yellow?logo=buy-me-a-coffee)](https://buymeacoffee.com/Jaixii) - -[https://engraphis.com/](https://engraphis.com/) - -[https://discord.com/invite/Wfr2ejBmY](https://discord.com/invite/Wfr2ejBmY) - -**Give your AI agents a memory. See it, search it, and maintain it, all in a beautiful WebUI on your own machine.** - -

- Engraphis Knowledge Graph tab: force-directed entity-relation network -
- Knowledge Graph · run engraphis-dashboard to see it live -

- -**Grounded, not guessed.** Memory with receipts. Local by default. - ---- - -> **Open-core boundary:** this repository contains the free local engine, dashboard, MCP server, -> and customer-side clients. Hosted sync, analytics, automation, and team services run on the -> official hosted service; their server implementations are not distributed here. - -> **Support continued Engraphis development with Pro.** [Start a 3-day Pro trial](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro&trial=pro#billing) -> or [subscribe to Pro](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro#billing). - ---- - -## Measured token and context savings - -### Runtime estimator - -The dashboard Overview and Audit/Receipts views also show a receipt-backed estimate from -real context deliveries. It compares the host history or retrieved source baseline with the -context Engraphis actually emitted, keeps token counters and release versions separate, and -labels adaptive history reductions separately from packing savings. Receipts without estimator -metadata remain historical/unclassified. This measures estimated prompt-context reduction; it -does not measure provider billing. The `/context-savings` API and -`engraphis_context_savings` MCP tool aggregate the complete history across all visible workspaces -by default, or accept an explicit workspace plus optional `from_ts`, `to_ts`, and -`release_version` filters. - -

+# Engraphis + +[![PyPI version](https://img.shields.io/pypi/v/engraphis.svg)](https://pypi.org/project/engraphis/) +[![License](https://img.shields.io/badge/license-Apache--2.0-green.svg)](https://github.com/Coding-Dev-Tools/engraphis/blob/main/LICENSE) +[![Support](https://img.shields.io/badge/Buy%20Me%20a%20Coffee-support-yellow?logo=buy-me-a-coffee)](https://buymeacoffee.com/Jaixii) + +[https://engraphis.com/](https://engraphis.com/) + +[https://discord.com/invite/Wfr2ejBmY](https://discord.com/invite/Wfr2ejBmY) + +**Give your AI agents a memory. See it, search it, and maintain it, all in a beautiful WebUI on your own machine.** + +

+ Engraphis Knowledge Graph tab: force-directed entity-relation network +
+ Knowledge Graph · run engraphis-dashboard to see it live +

+ +**Grounded, not guessed.** Memory with receipts. Local by default. + +--- + +> **Open-core boundary:** this repository contains the free local engine, dashboard, MCP server, +> and customer-side clients. Hosted sync, analytics, automation, and team services run on the +> official hosted service; their server implementations are not distributed here. + +> **Support continued Engraphis development with Pro.** [Start a 3-day Pro trial](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro&trial=pro#billing) +> or [subscribe to Pro](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro#billing). + +--- + +## Measured token and context savings + +### Runtime estimator + +The dashboard Overview and Audit/Receipts views also show a receipt-backed estimate from +real context deliveries. It compares the host history or retrieved source baseline with the +context Engraphis actually emitted, keeps token counters and release versions separate, and +labels adaptive history reductions separately from packing savings. Receipts without estimator +metadata remain historical/unclassified. This measures estimated prompt-context reduction; it +does not measure provider billing. The `/context-savings` API and +`engraphis_context_savings` MCP tool aggregate the complete history across all visible workspaces +by default, or accept an explicit workspace plus optional `from_ts`, `to_ts`, and +`release_version` filters. + +

Dark chart of registered deterministic fixtures. Structure-aware chunks reduce retrieved context from 740.3 to 214.3 tokens and the smallest evidence-holding memory from 162.2 to 42.4 tokens. A compact JSON-shape proxy uses 11,138 rather than 24,590 tokens. Retrieved-candidate quality is labeled separately from packed-context quality, both measured in the selected report with packed-quality fields. Actual MCP transport and provider billing are not measured. -
- Less repeated history means more room for the task, tools, and useful evidence. -

- -
-See benchmark details and reproduce the results - -### Controlled before-and-after example - -| Retrieval mode | Mean returned memory content | Recall@5 | -|---|---:|---:| -| Whole documents | 740.3 tokens | 1.000 | -| Engraphis structure-aware chunks | 214.3 tokens | 1.000 | - -The chunked mode returns the relevant passage instead of the whole document: **526.0 fewer tokens -per question**. Under the same model-context budget, that leaves roughly **526 tokens** for task -instructions or other relevant evidence. This is evidence ID `offline-chunking` in the registered -artifact below. - -### Measurement details and reproducibility - -The table below contains every exact token/context aggregate currently published here and keeps -its counting boundary explicit. - -| What is counted | Comparison | Measured reduction | Quality held constant | -|---|---|---|---| -| Retrieved top-5 memory content, averaged per question | Whole documents: **740.3** tokens → structure-aware chunks: **214.3** tokens | **526.0 fewer tokens per question** (**71.1% lower**, about **3.5× smaller**) | Recall@5 **1.000** in both modes across 6 documents and 18 questions | -| Smallest returned memory that contains the reference evidence | Whole documents: **162.2** tokens → chunks: **42.4** tokens | **119.8 fewer tokens to evidence** (**73.9% lower**, about **3.8× smaller**) | The same 18 questions had a returned evidence-holding memory in both modes | +
+ Less repeated history means more room for the task, tools, and useful evidence. +

+ +
+See benchmark details and reproduce the results + +### Controlled before-and-after example + +| Retrieval mode | Mean returned memory content | Recall@5 | +|---|---:|---:| +| Whole documents | 740.3 tokens | 1.000 | +| Engraphis structure-aware chunks | 214.3 tokens | 1.000 | + +The chunked mode returns the relevant passage instead of the whole document: **526.0 fewer tokens +per question**. Under the same model-context budget, that leaves roughly **526 tokens** for task +instructions or other relevant evidence. This is evidence ID `offline-chunking` in the registered +artifact below. + +### Measurement details and reproducibility + +The table below contains every exact token/context aggregate currently published here and keeps +its counting boundary explicit. + +| What is counted | Comparison | Measured reduction | Quality held constant | +|---|---|---|---| +| Retrieved top-5 memory content, averaged per question | Whole documents: **740.3** tokens → structure-aware chunks: **214.3** tokens | **526.0 fewer tokens per question** (**71.1% lower**, about **3.5× smaller**) | Recall@5 **1.000** in both modes across 6 documents and 18 questions | +| Smallest returned memory that contains the reference evidence | Whole documents: **162.2** tokens → chunks: **42.4** tokens | **119.8 fewer tokens to evidence** (**73.9% lower**, about **3.8× smaller**) | The same 18 questions had a returned evidence-holding memory in both modes | | Full versus compact recall payload proxy across one 26-question pass within a 260-timed-recall CodeMem run | Full proxy: **24,590** `engraphis.regex.v1` tokens → compact proxy: **11,138** tokens | **13,452 proxy tokens avoided** (**54.71% lower**) | 26 payload samples; 260 timed recalls; Recall@5, hit@5, and answer-token recall all **1.000** | -| Packed prompt-context usage in the same 26-question CodeMem sample pass | Hard budget: **1,500** tokens; observed mean: **85.38**; observed maximum: **108** | A hard cap prevents a recall from exceeding its configured context budget | This is usage accounting, not a before/after savings comparison | - -The performance report keeps its legacy `quality` fields for all candidate chunks returned before -context packing and adds `packed_quality` for evidence admitted to the reader context. The checked-in +| Packed prompt-context usage in the same 26-question CodeMem sample pass | Hard budget: **1,500** tokens; observed mean: **85.38**; observed maximum: **108** | A hard cap prevents a recall from exceeding its configured context budget | This is usage accounting, not a before/after savings comparison | + +The performance report keeps its legacy `quality` fields for all candidate chunks returned before +context packing and adds `packed_quality` for evidence admitted to the reader context. The checked-in v19 artifact includes both quality views, with Recall@5, hit@5 and answer-token evidence coverage -of 1.000 for the 26-question fixture in each view. Both views measure retrieved evidence; -neither is an end-to-end question-answer score. Coding outcomes, external datasets, and staged -operational capacity remain separate pending evaluation tracks until their artifacts are selected. - +of 1.000 for the 26-question fixture in each view. Both views measure retrieved evidence; +neither is an end-to-end question-answer score. Coding outcomes, external datasets, and staged +operational capacity remain separate pending evaluation tracks until their artifacts are selected. + These values are evidence IDs `offline-chunking` and `offline-performance` in -[`offline-fixtures-v73.json`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/benchmark-evidence/offline-fixtures-v73.json), +[`offline-fixtures-v76.json`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/benchmark-evidence/offline-fixtures-v76.json), SHA-256 -`aa7ed9c141afcc82fc2a05b63ed9037842f2ea8372667f3142cf9bb795833988`. -[`BENCHMARKS.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md#public-numeric-evidence-registry) -records the matching suite digest, exact commands, and per-command config digests. The offline -fixture registry intentionally excludes external, model-dependent, consolidation, productivity, -and latency results. Completed retrieval-only diagnostics are published separately in the -[benchmark expansion results](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/BENCHMARK_EXPANSION_RESULTS.md) with redacted immutable -artifacts; no generated-answer, official leaderboard, hosted-latency, or paid result is claimed -here. - -The compact payload shape avoids duplicating full memory bodies when the packed context and source -list are enough. The evaluator tokenizes JSON-shaped full and compact payload proxies built from -recall results; it does **not** serialize the MCP envelope or measure a transport response. The -fixture therefore does not measure model-provider charges, end-to-end task time, or customer cost -savings. - -The measures are deliberately separate and **must not be added together**: chunking counts the -content of retrieved memory records before `ContextPacker`, whereas compact recall counts a -serialized JSON-shape payload proxy. “Tokens to evidence” is the size of the smallest -retrieved memory record holding the reference evidence; it is not latency or end-to-end answer -accuracy. Chunking creates more focused stored records, so this is a context-efficiency result, -not a storage-reduction claim. - -Reproduce the registered quality and token/context measurements without a network connection or -API key: - -```bash -python -m eval.grounded -python -m eval.chunking_eval --dataset eval/datasets/longdoc.jsonl --k 5 -python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 --json -``` - -These are small deterministic correctness and efficiency fixtures, not official LoCoMo / -LongMemEval QA scores or a third-party leaderboard result. Compact-response counts use the exact -`engraphis.regex.v1` counter; the chunking evaluation uses its documented deterministic -normalized-character estimator. Chunking measures retrieved memory content, while compact recall -measures a serialized JSON-shape payload proxy, not an MCP transport response. See the registered -artifact and [`BENCHMARKS.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md) -for definitions, limitations, and canonical external-evaluation requirements. - -
- ---- - -## Full Engraphis install: pip install "engraphis[all]" - -The complete `engraphis[all]` install is the default way to use Engraphis: it includes the local -dashboard, Smart MCP server, documents, Cloud Sync client, and supported optional integrations. -Python 3.10+ is required. - -```bash -pip install "engraphis[all]" -engraphis-dashboard -``` - -The dashboard opens at [http://127.0.0.1:8700](http://127.0.0.1:8700). Local memory needs no -account or API key. - -### Smaller installation options - -Use a smaller package only when you intentionally need a limited surface. The NumPy-only core -continues to support Python 3.9+. - -| Goal | Install | Start | -|---|---|---| -| Local dashboard and REST API | `pip install "engraphis[server]"` | `engraphis-dashboard` | -| Coding-agent memory over Smart MCP | `pip install "engraphis[mcp]"` | `codex mcp add engraphis -- engraphis-mcp` | -| Native SQLite vector acceleration | `pip install "engraphis[vector]"` | Server entrypoints select it automatically | -| Offline Python library | `pip install engraphis` | `MemoryService.create("engraphis.db")` | - -For MCP clients other than Codex, configure a stdio server whose command is `engraphis-mcp`; see -the [agent connection guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/AGENT_CONNECT.md). - -### Updating - -Use `engraphis-update` to upgrade the installation using its detected install method. Package -metadata does not record which extras were selected, so the updater defaults to the safe -superset `engraphis[all]` rather than silently dropping an optional surface. For a deliberate -selection, set `ENGRAPHIS_UPDATE_EXTRAS` to a comma-separated list (for example -`server,mcp`), or set it to `none` for the base package only. - -> **Upgrading to 1.4:** `engraphis-mcp` now exposes the nine-tool Smart gateway. Integrations that -> require the former 35 direct tool names should run `engraphis-mcp-classic`. The SQLite schema -> in the 1.4.0 release was version 9. Existing v7-to-v8 databases already contain `confidence` -> and `pinned_at`/`unpinned_at`; v9 adds the `memory_tombstones` repository-scope column/table -> and performs a one-time entity-canonicalization repair, then migrates automatically on first -> open. A tombstone with a known `repo_id` is terminal only in that repository; legacy repo-less -> tombstones remain global. See the [1.4.0 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#140---2026-08-02). - -> **Upgrading to 1.5:** schema 10 bounds legacy retention state and schema 11 backfills explicit -> approval only for eligible pre-review local memories. Pending and quarantined evidence remains -> gated. Existing 1.4.x databases migrate automatically when Engraphis 1.5 opens them; see the -> [1.5 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#15---2026-08-04). - -> **Upgrading to 1.6:** existing 1.5 databases migrate automatically through schema 12, which -> classifies content-free erasure markers before sync: existing markers become local-only -> `never_export`, while new secure erasures become `remote_erasure` only for non-secret -> `workspace`/`repo` records already eligible for sharing. Schema 13 adds per-memory hybrid -> logical clocks for deterministic descriptive-state sync and durable, content-free proof that a -> memory crossed a sync boundary. Schema 14 adds the Obsidian collection and import manifests; -> schema 15 generalizes them to source-neutral local documents, preserves temporal source lineage -> across re-imports, binds adapters and target scopes, and retains only bounded, content-free -> per-job format/result metadata. The schema 16 migration persists each import job's optional session target -> and requires source lineage and job-item attachments to remain in that exact session. See the -> [1.6 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#16---2026-08-15). - ---- - -## What Engraphis gives an agent - -An agent should not have to reconstruct a project from scattered chat history on every task. -Engraphis turns local project knowledge into scoped, time-aware memory; retrieves the evidence -that supports the current question; and returns a bounded, attributable context packet. - -The core task is continuity: retrieve the current, supported project decision without dragging the -whole history into the next prompt. See [measured token and context savings](#measured-token-and-context-savings) -for the short version of how much less history an agent has to carry. - -| Agent need | What Engraphis changes | -|---|---| -| Remember a project across sessions | Stores typed memory in a `workspace → repo → session` hierarchy and provides a last-session handoff. | -| Find support for the current task | Fuses vector, lexical, graph, and code-aware retrieval instead of relying on one search signal; `fast` can skip graph traversal for small or latency-sensitive vaults. | -| Know what is true now and what changed | Preserves bi-temporal history and supersession chains instead of silently overwriting a fact. | -| Avoid confident guesses | Returns cited evidence or explicitly abstains when support is too weak. | -| Avoid dragging the whole project into every prompt | Packs context to a configured hard budget and can return a compact MCP response. | -| Keep knowledge in the operator's control | Runs local-first and offline-capable, with scopes, audit records, and optional privacy-safe receipts. | - -## Dashboard and local UI - -The Engraphis dashboard opens `http://127.0.0.1:8700`. Local memory needs no cloud account, -signup, or API key and stays in a SQLite file on your machine. - -**Ledger** is the primary local interface for recall, memories, graph exploration, provenance, -workspaces, and manual consolidation. **Classic** preserves the former full tool suite; both use -the same local data. Switch in **Manage → Settings → Interface** (Ledger) or **Settings → -Appearance & Engine** (Classic). - -### Start it on every platform - -| Platform | How | -|----------|-----| -| **Windows** | Double-click **Engraphis Dashboard** on your Desktop or Start Menu (install: `engraphis-dashboard --install-shortcuts`) | -| **macOS** | Double-click **Engraphis Dashboard.app** on your Desktop (install: same command) | -| **Linux** | Desktop entry in Applications → Development (GNOME/KDE/etc.) | -| **Docker** | `docker compose up`: see `docker-compose.yml` for the one-command deployment | -| **Any** | `engraphis-dashboard` in a terminal | - -In a source checkout, `scripts/launch_dashboard.ps1` is only a Windows convenience wrapper. It -delegates configuration, startup health, browser opening, and process lifecycle to the same -`engraphis-dashboard` entrypoint rather than maintaining a second behavior path. - -### Accessibility-first inspection, built in - -Inspect memories, supersession diffs, recall scores, timelines, links, consolidation, and audit -records in the dashboard. The offline graph renderer is vendored, and the interface is keyboard- -navigable with light and dark themes. Graph exploration offers a focused **High quality** view and -an explicit worker-backed **Every node** view for complete entity projections up to 20,000 -nodes and 200,000 relationships; see the [graph performance profiles](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/GRAPH_PERFORMANCE.md). - ---- - -## How it works - -Engraphis gives agents durable, scoped, *explainable* project knowledge. The local engine combines -Ebbinghaus decay, bi-temporal facts, and hybrid vector/lexical/graph recall; it runs offline with -SQLite, local embeddings, and `numpy` only. - -- **Grounded and governed:** deterministic conflict resolution, cited answers or abstention, - explicit correction/promotion/forgetting, and a complete history. -- **Agent-ready:** MCP tools, hard-budget context packets, handoffs, and code-aware retrieval. -- **Auditable:** content-free receipt chains, provenance, and temporal/entity/code relationships. -- **Practical:** local file and code ingest, optional PDF/OCR/transcription, and SQLCipher at rest. - -### Optional LLM providers - -The memory engine, embeddings, conflict resolution, and recall stay local without an LLM. An -explicitly configured provider adds structured extraction, cited synthesis, consolidation, and -retention supervision. Configure it in **Settings → Connect an LLM**. The activity view records -outcomes, never keys, prompts, or raw provider responses. See the -[LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md) for setup and privacy choices. - -> Privacy boundary: text sent to an explicitly selected provider leaves the local process under -> that provider's terms. Use `ENGRAPHIS_RETENTION_SUPERVISOR=none` (the default) and the offline -> `chunk` extractor when ingestion must remain entirely local. - -Choose and configure an external LLM with the [LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md), -including OpenAI, Anthropic, Google, OpenRouter, Ollama, Cohere Command, Command Code Provider, -and other compatible endpoints. The guide also covers Codex subscription MCP connections. - ---- - -## Install - -```bash -pip install "engraphis[all]" # self-hosted dashboard, MCP, code graph, documents, transcription, PostgreSQL, and Cloud Sync -pip install "engraphis[server]" # dashboard + REST API -pip install "engraphis[mcp]" # MCP server only -pip install "engraphis[documents]" # PDF + image OCR bindings -pip install "engraphis[transcription]" # faster-whisper audio/video -pip install "engraphis[postgres]" # PostgreSQL schema introspection -pip install "engraphis[code]" # tree-sitter code graph indexing -pip install "engraphis[vector]" # native sqlite-vec exact-KNN acceleration -pip install "engraphis[cloud-sync]" # Cloud Sync client crypto/runtime -pip install "engraphis[encryption]" # SQLCipher encryption-at-rest extra -pip install engraphis # core library: numpy only, fully offline -``` - -The official Docker image includes the local Tesseract executable for image OCR. Outside -Docker, the `documents` extra installs its Python bindings; install Tesseract through your -operating system as well if you enable image OCR. - -The NumPy-only core library supports Python 3.9+. Current patched releases of the WebUI -stack, MCP SDK, image parser, and Cloud Sync client require Python 3.10+, so use Python 3.10 -or newer for the `server`, `mcp`, `documents`, `cloud-sync`, or `all` installation paths. - -The default `NumpyVectorIndex` performs an exact full scan. There is no universal memory-count -cutoff because latency depends on vector size, hardware, filters, and the rest of the recall -pipeline. Measure your machine with `python -m eval.vector_scale --backend numpy`, then run -`python -m eval.performance` on a representative corpus. If exact scans miss your latency target, -install `engraphis[vector]`, create the engine with `vector_backend="sqlite-vec"`, and remeasure. -The stable sqlite-vec `vec0` backend executes exact KNN in native code; it is acceleration, not a -claim of sublinear ANN scaling. See [BENCHMARKS.md](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md) for the reproducible commands -and reporting limits. - -Dashboard, REST, and MCP entrypoints default to `ENGRAPHIS_VECTOR_BACKEND=auto`: they use -sqlite-vec when the `vector` extra is installed and compatible, then safely fall back to NumPy. -Programmatic `MemoryEngine.create()` and `MemoryService.create()` retain the deterministic -`numpy` default unless a backend is requested explicitly. -Use `python -m eval.vector_scale --backend sqlite-vec` for an input-identical direct-search -comparison; setup/index-build time is explicitly excluded from the timed search envelope. - -Persistent vectors fail closed unless the embedder can publish a durable, secret-free space -fingerprint. Sentence Transformers use the loaded Hub commit or a manifest of local artifacts; -when a remote model's immutable identity cannot be resolved, persistent vector recall remains -gated instead of mixing spaces. For programmatic OpenAI-compatible embeddings, construct -`ApiEmbedder` with an operator/provider `space_version`; without it the adapter remains usable for -ephemeral embedding only. Its `base_url` may be a provider root or a `/v1` root and is normalized -to exactly one `/v1/embeddings` endpoint. - -`sqlcipher3-binary` publishes CPython manylinux x86-64 wheels. On that target, -`engraphis[encryption]` installs the driver. The cross-platform `all` extra deliberately -omits it so `all` remains resolvable on macOS, Windows, Linux ARM, and musl; on those -targets, provision a compatible SQLCipher driver separately before enabling a database -key. The programmatic core remains plaintext unless a database key is configured. For a -fresh database, `engraphis-init` enables SQLCipher automatically when a compatible driver is -available, creates a private key sidecar, and can be overridden with `--no-encryption`. - -> **Linux / macOS:** if `pip install` fails with `error: externally-managed-environment`, -> your system Python is marked read-only (PEP 668). Install into a virtual environment -> instead. Run `python3 -m venv venv && source venv/bin/activate && pip install "engraphis[server]"` -> Alternatively, use Docker (`docker compose up`). `pipx install "engraphis[server]"` also works. - -> First run downloads `all-MiniLM-L6-v2` (~80 MB). Without it, the engine falls back -> to deterministic feature hashing so it always runs offline. That fallback captures lexical -> overlap, not meaning: recall and grounded MCP responses set `degraded_mode=true` and -> `semantic_support=false`, and disable vector retrieval plus semantic-cosine evidence. Install -> a declared embedding model for semantic retrieval. - -> To require a model that is already local, set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` -> or `local:`. This path never downloads a model. If it is unavailable, Engraphis -> explicitly enters lexical degraded mode instead of presenting hash-vector scores as semantic. - ---- - -## Quickstart: dashboard - -```bash -pip install "engraphis[server]" -engraphis-dashboard # → http://127.0.0.1:8700 -engraphis-dashboard --install-shortcuts # → Desktop + Start Menu icons -``` - -> **Offline first run:** the first launch downloads the `all-MiniLM-L6-v2` embedding model -> (~80 MB), then runs fully offline. To stay offline-only, set -> `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` (never downloads; unknown local models -> enter lexical degraded mode instead of faking semantic scores). Extraction defaults to -> `ENGRAPHIS_EXTRACTOR=none` (verbatim writes), the vector backend defaults to `auto` (native -> acceleration when installed, otherwise NumPy), and recall without a usable semantic space -> reports `degraded_mode=true` with lexical/graph recall. Run `engraphis-init --check` to -> verify the install, extras, and database writability. - -### Docker - -```bash -docker compose up # → http://127.0.0.1:8700 -``` - -For Docker Compose persistence and loopback-port configuration, see the -[Docker deployment guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCKER.md). -`engraphis-server` and `engraphis server` are headless compatibility aliases -for this same v2 service, so every public surface has the same scoped recall and retention model. - -For optional LAN exposure, token configuration, and HTTP MCP setup, see the -[Docker deployment guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCKER.md). - -Set `ENGRAPHIS_API_TOKEN` to require API authentication and `ENGRAPHIS_DB_KEY` to encrypt -the local database at rest. Hosted-plan credentials configure customer clients; they do not -install premium server implementations into this image. See `docker-compose.yml` for options. - ---- - -## Quickstart: MCP server (for coding agents) - -```bash -pip install "engraphis[mcp]" -engraphis-init # writes ~/.engraphis/config.env + prints config snippets -claude mcp add engraphis -- engraphis-mcp -codex mcp add engraphis -- engraphis-mcp # Codex subscription - -``` - -> **Offline first run:** the first tool call lazily loads the `all-MiniLM-L6-v2` embedding -> model (~80 MB, same download as the dashboard), then memory runs fully offline with no API -> key. To stay offline-only, set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` (never -> downloads); extraction defaults to `ENGRAPHIS_EXTRACTOR=none`, the vector backend `auto` -> falls back to NumPy without the `vector` extra, and recall without a usable semantic space -> reports `degraded_mode=true` with lexical/graph recall. Run `engraphis-init --check` to -> verify the install and database path before registering the server. - -For Codex subscription setup and verification, see the [agent connection guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/AGENT_CONNECT.md) -and the [LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md). - -`engraphis-mcp` is zero-configuration Smart MCP: agents begin with nine compact tools for sessions, -prompt-ready recall, durable memory, governed record read/update, conflict review, action discovery, -and safe execution. For code graphs, -governance, audit, or other advanced work, the agent calls `engraphis_discover_actions` and then -the indicated read or action executor; no profile selection is required. The gateway validates -the discovered capability again before it runs it, and clients remain responsible for their -normal destructive-action approval boundary. - -Existing clients that pin the historical 35 named tools can use -`engraphis-mcp-classic` (or `engraphis-mcp-http --classic`). The complete classic inventory, -including `engraphis_check_update`, is in the [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md). - -### Pi extension - -For installation, configuration, lifecycle commands, and the local trust boundary, see the -[Pi extension guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/pi/README.md). - -### Command Code SessionStart hook - -`integrations/commandcode/` ships a SessionStart hook that warms up a new -session with bounded, recalled context from the local Engraphis gateway. Fails -open on timeout and is installed via `python scripts/install_cc_hook.py`. - -### prime-agent fleet - -`integrations/prime_agent/` ships a first-party Python package for -[PrimeIntellect prime-agent](https://github.com/PrimeIntellect-ai/prime-agent) -that exposes the same nine Smart MCP tools, with a `PrimeAgentFleet` of eight -named sub-agents (`researcher`, `planner`, `coder`, `reviewer`, `tester`, -`documenter`, `monitor`, `integrator`) sharing one `engraphis-mcp` stdio -subprocess. Install via `pip install ./integrations/prime_agent` and register -with `python scripts/install_prime_agent.py`. See the -[prime-agent integration guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/prime_agent/README.md). - -**What the integration is.** A `PrimeAgentFleet` is a thin Python layer -around the same `engraphis-mcp` Smart gateway every other host uses. At -runtime the fleet holds one shared `EngraphisMcpClient`, which owns one -`engraphis-mcp` subprocess over JSON-RPC stdio. Each of the eight named -sub-agents gets its own Engraphis session (started lazily on first tool use) -and its own default `repo` scope, so per-role memory is isolated while the -local gateway stays single-process. The eight sub-agent names -(`researcher`, `planner`, `coder`, `reviewer`, `tester`, `documenter`, -`monitor`, `integrator`) are the fixed default; pass `agent_names=[...]` to -`PrimeAgentFleet(...)` for a custom set. Concurrent tool calls serialize at -the JSON-RPC frame layer through an `asyncio.Lock`, so framework-level -parallelism (eight sub-agents reasoning at once) is preserved while the -underlying MCP transport remains one ordered stream. The only integration -surface is `EngraphisPrimeAgent.register()` in -`integrations/prime_agent/src/engraphis_prime_agent/agent.py` -- that is the -single adapter point to override if prime-agent's tool-registration API -differs from the assumed `target.register_tool(name, fn, schema=...)` -contract. - -The design -- eight named sub-agents, one shared stdio subprocess, -per-agent session bootstrap, and `ENGRAPHIS_*`-only environment forwarding -to the gateway -- is recorded in `~/.commandcode/plans/prime-agent-integration.md` -on the host where the integration was developed. When that host plan is not -available (other contributor machines, CI), the same design is summarized in -the PR description that introduced the integration and in the -[prime-agent integration guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/prime_agent/README.md) -("Architecture" and "Concurrency model" sections). - -## Quickstart: repository graph - -```bash -pip install "engraphis[code]" -engraphis-graph index -w acme -r api --root . -engraphis-graph search -w acme -r api "UserService" -# `query`/`explain` blend code search with your stored memories: query matches symbol -# and file NAMES (a full question sentence won't match anything), and explain's answer -# is drawn from memories recorded against the repo; both are empty on a fresh index. -engraphis-graph query -w acme -r api "UserService" -engraphis-graph explain -w acme -r api "why does deploy depend on approval?" -engraphis-graph path -w acme -r api UserService DatabasePool -engraphis-graph impact -w acme -r api --root . --git-range origin/main...HEAD -engraphis-graph prs -w acme -r api --base main --head HEAD -engraphis-graph export -w acme -r api -o engraphis-graph-out -engraphis-graph install-merge-driver --root . -``` - -The export contains `graph.json`, a self-contained `graph.html`, and `GRAPH_REPORT.md`. -Indexing supports Python, JavaScript, TypeScript, Go, Rust, Java, C#, C, C++, SQL, and -Terraform. Tree-sitter is used when available; the dependency-free regex backend remains a -functional fallback. Definitions, methods, calls, imports, ownership, variables, -inheritance/implementation, and docstrings/comments are indexed. Indexing is incremental by -content hash, honors `.engraphisignore`, and does not follow file symlinks outside the repository -root. Call edges are name-based and best-effort rather than type-resolved. The optional Git merge -driver validates bounded graph JSON and deterministically unions nodes and edges instead of -choosing one export side. - -For a read-only recall and graph API that can be shared without exposing write operations: - -```bash -pip install "engraphis[server]" -engraphis-graph-server # API at http://127.0.0.1:8720; schema at /openapi.json -``` - -A non-loopback bind fails closed unless `ENGRAPHIS_GRAPH_TOKEN` (or -`ENGRAPHIS_API_TOKEN`) is set. See [the v3 architecture/design document](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md). - ---- - -## Quickstart: Python library - -```python -from engraphis.service import MemoryService - -mem = MemoryService.create("engraphis.db") -mem.remember("Auth migrated from JWT to PASETO.", workspace="acme", repo="api") -hit = mem.recall("why did we change auth?", workspace="acme", repo="api") -print(hit["context"]) -``` - -The same `MemoryService` backs the dashboard and the MCP server. The package root also -intentionally exposes the low-level engine facade (`MemoryEngine`, `create_memory_engine`) -for advanced composition, while `MemoryService` remains the high-level service API. - -New writes support `session`, `repo`, and `workspace` visibility. `scope="user"` is reserved and -rejected until records carry an immutable owner identity; it must not be treated as private -per-person memory. Historical user-scope rows remain workspace-bound for compatibility. - -After an upgrade, `stats()` reports prompt-eligibility counts and active embedding-space -coverage. Zero-result recall identifies a review-gated scope instead of silently looking empty, -and `engraphis-cli review list|approve` provides a dry-run-first local bulk workflow. Embedding -model changes trigger a guarded rebuild; vector recall stays disabled until every stored vector -matches the new fingerprint. See [recall recovery](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/RECALL_RECOVERY.md). - -Agent hosts can avoid retrieval when their existing history already fits: - -```python -decision = mem.adaptive_context( - "what should the agent do next?", - current_history, - workspace="acme", - repo="api", - max_context_tokens=8_192, - retrieval_token_budget=1_024, -) -prompt_context = decision["context"] -``` - -The decision is `history_bypass` when the history fits, `retrieval` when compact evidence is -strong, and `history_fallback` when weak retrieval should widen back to recent raw history. - -For an agent prompt, prefer `engraphis_recall_context`: it returns one hard-budget packed -`context` plus compact `sources`, deterministic `usage` accounting (`budget_tokens`, `context_tokens`, -`source_tokens`, `saved_tokens`, `savings_ratio`, `packed_count`, `omitted_count`, and -`token_counter`), and optional diagnostics. Accounting is exact for the named counter; inject the -reader's tokenizer when reader-model token parity is required. `engraphis_recall` remains the compatible full-recall +`2fb5ce5b2cbc21541f7ae9cad5d2ef615014c0e86f9881f00621d00986422ba8`. +[`BENCHMARKS.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md#public-numeric-evidence-registry) +records the matching suite digest, exact commands, and per-command config digests. The offline +fixture registry intentionally excludes external, model-dependent, consolidation, productivity, +and latency results. Completed retrieval-only diagnostics are published separately in the +[benchmark expansion results](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/BENCHMARK_EXPANSION_RESULTS.md) with redacted immutable +artifacts; no generated-answer, official leaderboard, hosted-latency, or paid result is claimed +here. + +The compact payload shape avoids duplicating full memory bodies when the packed context and source +list are enough. The evaluator tokenizes JSON-shaped full and compact payload proxies built from +recall results; it does **not** serialize the MCP envelope or measure a transport response. The +fixture therefore does not measure model-provider charges, end-to-end task time, or customer cost +savings. + +The measures are deliberately separate and **must not be added together**: chunking counts the +content of retrieved memory records before `ContextPacker`, whereas compact recall counts a +serialized JSON-shape payload proxy. “Tokens to evidence” is the size of the smallest +retrieved memory record holding the reference evidence; it is not latency or end-to-end answer +accuracy. Chunking creates more focused stored records, so this is a context-efficiency result, +not a storage-reduction claim. + +Reproduce the registered quality and token/context measurements without a network connection or +API key: + +```bash +python -m eval.grounded +python -m eval.chunking_eval --dataset eval/datasets/longdoc.jsonl --k 5 +python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 --json +``` + +These are small deterministic correctness and efficiency fixtures, not official LoCoMo / +LongMemEval QA scores or a third-party leaderboard result. Compact-response counts use the exact +`engraphis.regex.v1` counter; the chunking evaluation uses its documented deterministic +normalized-character estimator. Chunking measures retrieved memory content, while compact recall +measures a serialized JSON-shape payload proxy, not an MCP transport response. See the registered +artifact and [`BENCHMARKS.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md) +for definitions, limitations, and canonical external-evaluation requirements. + +
+ +--- + +## Full Engraphis install: pip install "engraphis[all]" + +The complete `engraphis[all]` install is the default way to use Engraphis: it includes the local +dashboard, Smart MCP server, documents, Cloud Sync client, and supported optional integrations. +Python 3.10+ is required. + +```bash +pip install "engraphis[all]" +engraphis-dashboard +``` + +The dashboard opens at [http://127.0.0.1:8700](http://127.0.0.1:8700). Local memory needs no +account or API key. + +### Smaller installation options + +Use a smaller package only when you intentionally need a limited surface. The NumPy-only core +continues to support Python 3.9+. + +| Goal | Install | Start | +|---|---|---| +| Local dashboard and REST API | `pip install "engraphis[server]"` | `engraphis-dashboard` | +| Coding-agent memory over Smart MCP | `pip install "engraphis[mcp]"` | `codex mcp add engraphis -- engraphis-mcp` | +| Native SQLite vector acceleration | `pip install "engraphis[vector]"` | Server entrypoints select it automatically | +| Offline Python library | `pip install engraphis` | `MemoryService.create("engraphis.db")` | + +For MCP clients other than Codex, configure a stdio server whose command is `engraphis-mcp`; see +the [agent connection guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/AGENT_CONNECT.md). + +### Updating + +Use `engraphis-update` to upgrade the installation using its detected install method. Package +metadata does not record which extras were selected, so the updater defaults to the safe +superset `engraphis[all]` rather than silently dropping an optional surface. For a deliberate +selection, set `ENGRAPHIS_UPDATE_EXTRAS` to a comma-separated list (for example +`server,mcp`), or set it to `none` for the base package only. + +> **Upgrading to 1.4:** `engraphis-mcp` now exposes the nine-tool Smart gateway. Integrations that +> require the former 35 direct tool names should run `engraphis-mcp-classic`. The SQLite schema +> in the 1.4.0 release was version 9. Existing v7-to-v8 databases already contain `confidence` +> and `pinned_at`/`unpinned_at`; v9 adds the `memory_tombstones` repository-scope column/table +> and performs a one-time entity-canonicalization repair, then migrates automatically on first +> open. A tombstone with a known `repo_id` is terminal only in that repository; legacy repo-less +> tombstones remain global. See the [1.4.0 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#140---2026-08-02). + +> **Upgrading to 1.5:** schema 10 bounds legacy retention state and schema 11 backfills explicit +> approval only for eligible pre-review local memories. Pending and quarantined evidence remains +> gated. Existing 1.4.x databases migrate automatically when Engraphis 1.5 opens them; see the +> [1.5 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#15---2026-08-04). + +> **Upgrading to 1.6:** existing 1.5 databases migrate automatically through schema 12, which +> classifies content-free erasure markers before sync: existing markers become local-only +> `never_export`, while new secure erasures become `remote_erasure` only for non-secret +> `workspace`/`repo` records already eligible for sharing. Schema 13 adds per-memory hybrid +> logical clocks for deterministic descriptive-state sync and durable, content-free proof that a +> memory crossed a sync boundary. Schema 14 adds the Obsidian collection and import manifests; +> schema 15 generalizes them to source-neutral local documents, preserves temporal source lineage +> across re-imports, binds adapters and target scopes, and retains only bounded, content-free +> per-job format/result metadata. The schema 16 migration persists each import job's optional session target +> and requires source lineage and job-item attachments to remain in that exact session. See the +> [1.6 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#16---2026-08-15). + +--- + +## What Engraphis gives an agent + +An agent should not have to reconstruct a project from scattered chat history on every task. +Engraphis turns local project knowledge into scoped, time-aware memory; retrieves the evidence +that supports the current question; and returns a bounded, attributable context packet. + +The core task is continuity: retrieve the current, supported project decision without dragging the +whole history into the next prompt. See [measured token and context savings](#measured-token-and-context-savings) +for the short version of how much less history an agent has to carry. + +| Agent need | What Engraphis changes | +|---|---| +| Remember a project across sessions | Stores typed memory in a `workspace → repo → session` hierarchy and provides a last-session handoff. | +| Find support for the current task | Fuses vector, lexical, graph, and code-aware retrieval instead of relying on one search signal; `fast` can skip graph traversal for small or latency-sensitive vaults. | +| Know what is true now and what changed | Preserves bi-temporal history and supersession chains instead of silently overwriting a fact. | +| Avoid confident guesses | Returns cited evidence or explicitly abstains when support is too weak. | +| Avoid dragging the whole project into every prompt | Packs context to a configured hard budget and can return a compact MCP response. | +| Keep knowledge in the operator's control | Runs local-first and offline-capable, with scopes, audit records, and optional privacy-safe receipts. | + +## Dashboard and local UI + +The Engraphis dashboard opens `http://127.0.0.1:8700`. Local memory needs no cloud account, +signup, or API key and stays in a SQLite file on your machine. + +**Ledger** is the primary local interface for recall, memories, graph exploration, provenance, +workspaces, and manual consolidation. **Classic** preserves the former full tool suite; both use +the same local data. Switch in **Manage → Settings → Interface** (Ledger) or **Settings → +Appearance & Engine** (Classic). + +### Start it on every platform + +| Platform | How | +|----------|-----| +| **Windows** | Double-click **Engraphis Dashboard** on your Desktop or Start Menu (install: `engraphis-dashboard --install-shortcuts`) | +| **macOS** | Double-click **Engraphis Dashboard.app** on your Desktop (install: same command) | +| **Linux** | Desktop entry in Applications → Development (GNOME/KDE/etc.) | +| **Docker** | `docker compose up`: see `docker-compose.yml` for the one-command deployment | +| **Any** | `engraphis-dashboard` in a terminal | + +In a source checkout, `scripts/launch_dashboard.ps1` is only a Windows convenience wrapper. It +delegates configuration, startup health, browser opening, and process lifecycle to the same +`engraphis-dashboard` entrypoint rather than maintaining a second behavior path. + +### Accessibility-first inspection, built in + +Inspect memories, supersession diffs, recall scores, timelines, links, consolidation, and audit +records in the dashboard. The offline graph renderer is vendored, and the interface is keyboard- +navigable with light and dark themes. Graph exploration offers a focused **High quality** view and +an explicit worker-backed **Every node** view for complete entity projections up to 20,000 +nodes and 200,000 relationships; see the [graph performance profiles](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/GRAPH_PERFORMANCE.md). + +--- + +## How it works + +Engraphis gives agents durable, scoped, *explainable* project knowledge. The local engine combines +Ebbinghaus decay, bi-temporal facts, and hybrid vector/lexical/graph recall; it runs offline with +SQLite, local embeddings, and `numpy` only. + +- **Grounded and governed:** deterministic conflict resolution, cited answers or abstention, + explicit correction/promotion/forgetting, and a complete history. +- **Agent-ready:** MCP tools, hard-budget context packets, handoffs, and code-aware retrieval. +- **Auditable:** content-free receipt chains, provenance, and temporal/entity/code relationships. +- **Practical:** local file and code ingest, optional PDF/OCR/transcription, and SQLCipher at rest. + +### Optional LLM providers + +The memory engine, embeddings, conflict resolution, and recall stay local without an LLM. An +explicitly configured provider adds structured extraction, cited synthesis, consolidation, and +retention supervision. Configure it in **Settings → Connect an LLM**. The activity view records +outcomes, never keys, prompts, or raw provider responses. See the +[LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md) for setup and privacy choices. + +> Privacy boundary: text sent to an explicitly selected provider leaves the local process under +> that provider's terms. Use `ENGRAPHIS_RETENTION_SUPERVISOR=none` (the default) and the offline +> `chunk` extractor when ingestion must remain entirely local. + +Choose and configure an external LLM with the [LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md), +including OpenAI, Anthropic, Google, OpenRouter, Ollama, Cohere Command, Command Code Provider, +and other compatible endpoints. The guide also covers Codex subscription MCP connections. + +--- + +## Install + +```bash +pip install "engraphis[all]" # self-hosted dashboard, MCP, code graph, documents, transcription, PostgreSQL, and Cloud Sync +pip install "engraphis[server]" # dashboard + REST API +pip install "engraphis[mcp]" # MCP server only +pip install "engraphis[documents]" # PDF + image OCR bindings +pip install "engraphis[transcription]" # faster-whisper audio/video +pip install "engraphis[postgres]" # PostgreSQL schema introspection +pip install "engraphis[code]" # tree-sitter code graph indexing +pip install "engraphis[vector]" # native sqlite-vec exact-KNN acceleration +pip install "engraphis[cloud-sync]" # Cloud Sync client crypto/runtime +pip install "engraphis[encryption]" # SQLCipher encryption-at-rest extra +pip install engraphis # core library: numpy only, fully offline +``` + +The official Docker image includes the local Tesseract executable for image OCR. Outside +Docker, the `documents` extra installs its Python bindings; install Tesseract through your +operating system as well if you enable image OCR. + +The NumPy-only core library supports Python 3.9+. Current patched releases of the WebUI +stack, MCP SDK, image parser, and Cloud Sync client require Python 3.10+, so use Python 3.10 +or newer for the `server`, `mcp`, `documents`, `cloud-sync`, or `all` installation paths. + +The default `NumpyVectorIndex` performs an exact full scan. There is no universal memory-count +cutoff because latency depends on vector size, hardware, filters, and the rest of the recall +pipeline. Measure your machine with `python -m eval.vector_scale --backend numpy`, then run +`python -m eval.performance` on a representative corpus. If exact scans miss your latency target, +install `engraphis[vector]`, create the engine with `vector_backend="sqlite-vec"`, and remeasure. +The stable sqlite-vec `vec0` backend executes exact KNN in native code; it is acceleration, not a +claim of sublinear ANN scaling. See [BENCHMARKS.md](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md) for the reproducible commands +and reporting limits. + +Dashboard, REST, and MCP entrypoints default to `ENGRAPHIS_VECTOR_BACKEND=auto`: they use +sqlite-vec when the `vector` extra is installed and compatible, then safely fall back to NumPy. +Programmatic `MemoryEngine.create()` and `MemoryService.create()` retain the deterministic +`numpy` default unless a backend is requested explicitly. +Use `python -m eval.vector_scale --backend sqlite-vec` for an input-identical direct-search +comparison; setup/index-build time is explicitly excluded from the timed search envelope. + +Persistent vectors fail closed unless the embedder can publish a durable, secret-free space +fingerprint. Sentence Transformers use the loaded Hub commit or a manifest of local artifacts; +when a remote model's immutable identity cannot be resolved, persistent vector recall remains +gated instead of mixing spaces. For programmatic OpenAI-compatible embeddings, construct +`ApiEmbedder` with an operator/provider `space_version`; without it the adapter remains usable for +ephemeral embedding only. Its `base_url` may be a provider root or a `/v1` root and is normalized +to exactly one `/v1/embeddings` endpoint. + +`sqlcipher3-binary` publishes CPython manylinux x86-64 wheels. On that target, +`engraphis[encryption]` installs the driver. The cross-platform `all` extra deliberately +omits it so `all` remains resolvable on macOS, Windows, Linux ARM, and musl; on those +targets, provision a compatible SQLCipher driver separately before enabling a database +key. The programmatic core remains plaintext unless a database key is configured. For a +fresh database, `engraphis-init` enables SQLCipher automatically when a compatible driver is +available, creates a private key sidecar, and can be overridden with `--no-encryption`. + +> **Linux / macOS:** if `pip install` fails with `error: externally-managed-environment`, +> your system Python is marked read-only (PEP 668). Install into a virtual environment +> instead. Run `python3 -m venv venv && source venv/bin/activate && pip install "engraphis[server]"` +> Alternatively, use Docker (`docker compose up`). `pipx install "engraphis[server]"` also works. + +> First run downloads `all-MiniLM-L6-v2` (~80 MB). Without it, the engine falls back +> to deterministic feature hashing so it always runs offline. That fallback captures lexical +> overlap, not meaning: recall and grounded MCP responses set `degraded_mode=true` and +> `semantic_support=false`, and disable vector retrieval plus semantic-cosine evidence. Install +> a declared embedding model for semantic retrieval. + +> To require a model that is already local, set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` +> or `local:`. This path never downloads a model. If it is unavailable, Engraphis +> explicitly enters lexical degraded mode instead of presenting hash-vector scores as semantic. + +--- + +## Quickstart: dashboard + +```bash +pip install "engraphis[server]" +engraphis-dashboard # → http://127.0.0.1:8700 +engraphis-dashboard --install-shortcuts # → Desktop + Start Menu icons +``` + +> **Offline first run:** the first launch downloads the `all-MiniLM-L6-v2` embedding model +> (~80 MB), then runs fully offline. To stay offline-only, set +> `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` (never downloads; unknown local models +> enter lexical degraded mode instead of faking semantic scores). Extraction defaults to +> `ENGRAPHIS_EXTRACTOR=none` (verbatim writes), the vector backend defaults to `auto` (native +> acceleration when installed, otherwise NumPy), and recall without a usable semantic space +> reports `degraded_mode=true` with lexical/graph recall. Run `engraphis-init --check` to +> verify the install, extras, and database writability. + +### Docker + +```bash +docker compose up # → http://127.0.0.1:8700 +``` + +For Docker Compose persistence and loopback-port configuration, see the +[Docker deployment guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCKER.md). +`engraphis-server` and `engraphis server` are headless compatibility aliases +for this same v2 service, so every public surface has the same scoped recall and retention model. + +For optional LAN exposure, token configuration, and HTTP MCP setup, see the +[Docker deployment guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCKER.md). + +Set `ENGRAPHIS_API_TOKEN` to require API authentication and `ENGRAPHIS_DB_KEY` to encrypt +the local database at rest. Hosted-plan credentials configure customer clients; they do not +install premium server implementations into this image. See `docker-compose.yml` for options. + +--- + +## Quickstart: MCP server (for coding agents) + +```bash +pip install "engraphis[mcp]" +engraphis-init # writes ~/.engraphis/config.env + prints config snippets +claude mcp add engraphis -- engraphis-mcp +codex mcp add engraphis -- engraphis-mcp # Codex subscription + +``` + +> **Offline first run:** the first tool call lazily loads the `all-MiniLM-L6-v2` embedding +> model (~80 MB, same download as the dashboard), then memory runs fully offline with no API +> key. To stay offline-only, set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` (never +> downloads); extraction defaults to `ENGRAPHIS_EXTRACTOR=none`, the vector backend `auto` +> falls back to NumPy without the `vector` extra, and recall without a usable semantic space +> reports `degraded_mode=true` with lexical/graph recall. Run `engraphis-init --check` to +> verify the install and database path before registering the server. + +For Codex subscription setup and verification, see the [agent connection guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/AGENT_CONNECT.md) +and the [LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md). + +`engraphis-mcp` is zero-configuration Smart MCP: agents begin with nine compact tools for sessions, +prompt-ready recall, durable memory, governed record read/update, conflict review, action discovery, +and safe execution. For code graphs, +governance, audit, or other advanced work, the agent calls `engraphis_discover_actions` and then +the indicated read or action executor; no profile selection is required. The gateway validates +the discovered capability again before it runs it, and clients remain responsible for their +normal destructive-action approval boundary. + +Existing clients that pin the historical 35 named tools can use +`engraphis-mcp-classic` (or `engraphis-mcp-http --classic`). The complete classic inventory, +including `engraphis_check_update`, is in the [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md). + +### Pi extension + +For installation, configuration, lifecycle commands, and the local trust boundary, see the +[Pi extension guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/pi/README.md). + +### Command Code SessionStart hook + +`integrations/commandcode/` ships a SessionStart hook that warms up a new +session with bounded, recalled context from the local Engraphis gateway. Fails +open on timeout and is installed via `python scripts/install_cc_hook.py`. + +### prime-agent fleet + +`integrations/prime_agent/` ships a first-party Python package for +[PrimeIntellect prime-agent](https://github.com/PrimeIntellect-ai/prime-agent) +that exposes the same nine Smart MCP tools, with a `PrimeAgentFleet` of eight +named sub-agents (`researcher`, `planner`, `coder`, `reviewer`, `tester`, +`documenter`, `monitor`, `integrator`) sharing one `engraphis-mcp` stdio +subprocess. Install via `pip install ./integrations/prime_agent` and register +with `python scripts/install_prime_agent.py`. See the +[prime-agent integration guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/prime_agent/README.md). + +**What the integration is.** A `PrimeAgentFleet` is a thin Python layer +around the same `engraphis-mcp` Smart gateway every other host uses. At +runtime the fleet holds one shared `EngraphisMcpClient`, which owns one +`engraphis-mcp` subprocess over JSON-RPC stdio. Each of the eight named +sub-agents gets its own Engraphis session (started lazily on first tool use) +and its own default `repo` scope, so per-role memory is isolated while the +local gateway stays single-process. The eight sub-agent names +(`researcher`, `planner`, `coder`, `reviewer`, `tester`, `documenter`, +`monitor`, `integrator`) are the fixed default; pass `agent_names=[...]` to +`PrimeAgentFleet(...)` for a custom set. Concurrent tool calls serialize at +the JSON-RPC frame layer through an `asyncio.Lock`, so framework-level +parallelism (eight sub-agents reasoning at once) is preserved while the +underlying MCP transport remains one ordered stream. The only integration +surface is `EngraphisPrimeAgent.register()` in +`integrations/prime_agent/src/engraphis_prime_agent/agent.py` -- that is the +single adapter point to override if prime-agent's tool-registration API +differs from the assumed `target.register_tool(name, fn, schema=...)` +contract. + +The design -- eight named sub-agents, one shared stdio subprocess, +per-agent session bootstrap, and `ENGRAPHIS_*`-only environment forwarding +to the gateway -- is recorded in `~/.commandcode/plans/prime-agent-integration.md` +on the host where the integration was developed. When that host plan is not +available (other contributor machines, CI), the same design is summarized in +the PR description that introduced the integration and in the +[prime-agent integration guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/prime_agent/README.md) +("Architecture" and "Concurrency model" sections). + +## Quickstart: repository graph + +```bash +pip install "engraphis[code]" +engraphis-graph index -w acme -r api --root . +engraphis-graph search -w acme -r api "UserService" +# `query`/`explain` blend code search with your stored memories: query matches symbol +# and file NAMES (a full question sentence won't match anything), and explain's answer +# is drawn from memories recorded against the repo; both are empty on a fresh index. +engraphis-graph query -w acme -r api "UserService" +engraphis-graph explain -w acme -r api "why does deploy depend on approval?" +engraphis-graph path -w acme -r api UserService DatabasePool +engraphis-graph impact -w acme -r api --root . --git-range origin/main...HEAD +engraphis-graph prs -w acme -r api --base main --head HEAD +engraphis-graph export -w acme -r api -o engraphis-graph-out +engraphis-graph install-merge-driver --root . +``` + +The export contains `graph.json`, a self-contained `graph.html`, and `GRAPH_REPORT.md`. +Indexing supports Python, JavaScript, TypeScript, Go, Rust, Java, C#, C, C++, SQL, and +Terraform. Tree-sitter is used when available; the dependency-free regex backend remains a +functional fallback. Definitions, methods, calls, imports, ownership, variables, +inheritance/implementation, and docstrings/comments are indexed. Indexing is incremental by +content hash, honors `.engraphisignore`, and does not follow file symlinks outside the repository +root. Call edges are name-based and best-effort rather than type-resolved. The optional Git merge +driver validates bounded graph JSON and deterministically unions nodes and edges instead of +choosing one export side. + +For a read-only recall and graph API that can be shared without exposing write operations: + +```bash +pip install "engraphis[server]" +engraphis-graph-server # API at http://127.0.0.1:8720; schema at /openapi.json +``` + +A non-loopback bind fails closed unless `ENGRAPHIS_GRAPH_TOKEN` (or +`ENGRAPHIS_API_TOKEN`) is set. See [the v3 architecture/design document](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md). + +--- + +## Quickstart: Python library + +```python +from engraphis.service import MemoryService + +mem = MemoryService.create("engraphis.db") +mem.remember("Auth migrated from JWT to PASETO.", workspace="acme", repo="api") +hit = mem.recall("why did we change auth?", workspace="acme", repo="api") +print(hit["context"]) +``` + +The same `MemoryService` backs the dashboard and the MCP server. The package root also +intentionally exposes the low-level engine facade (`MemoryEngine`, `create_memory_engine`) +for advanced composition, while `MemoryService` remains the high-level service API. + +New writes support `session`, `repo`, and `workspace` visibility. `scope="user"` is reserved and +rejected until records carry an immutable owner identity; it must not be treated as private +per-person memory. Historical user-scope rows remain workspace-bound for compatibility. + +After an upgrade, `stats()` reports prompt-eligibility counts and active embedding-space +coverage. Zero-result recall identifies a review-gated scope instead of silently looking empty, +and `engraphis-cli review list|approve` provides a dry-run-first local bulk workflow. Embedding +model changes trigger a guarded rebuild; vector recall stays disabled until every stored vector +matches the new fingerprint. See [recall recovery](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/RECALL_RECOVERY.md). + +Agent hosts can avoid retrieval when their existing history already fits: + +```python +decision = mem.adaptive_context( + "what should the agent do next?", + current_history, + workspace="acme", + repo="api", + max_context_tokens=8_192, + retrieval_token_budget=1_024, +) +prompt_context = decision["context"] +``` + +The decision is `history_bypass` when the history fits, `retrieval` when compact evidence is +strong, and `history_fallback` when weak retrieval should widen back to recent raw history. + +For an agent prompt, prefer `engraphis_recall_context`: it returns one hard-budget packed +`context` plus compact `sources`, deterministic `usage` accounting (`budget_tokens`, `context_tokens`, +`source_tokens`, `saved_tokens`, `savings_ratio`, `packed_count`, `omitted_count`, and +`token_counter`), and optional diagnostics. Accounting is exact for the named counter; inject the +reader's tokenizer when reader-model token parity is required. `engraphis_recall` remains the compatible full-recall surface; use `response_mode="compact"` when the packed context is enough and full memory bodies would duplicate it. For advanced query-planning configuration, see the [architecture guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md#query-planning). @@ -577,338 +577,338 @@ and title-only revisions preserve valid bindings. History preserves the original MCP response trimming removes binding metadata whenever its supporting context is omitted. For bi-temporal reads, `valid_at` selects what was true at a Unix timestamp and `known_at` selects -what Engraphis had learned then. `as_of` remains a compatibility alias for `valid_at`; supplying -both is allowed only when they match. - -For a mutable claim, pass a stable `subject_key` and optional `claim_kind`, such as -`subject_key="api.rate_limit", claim_kind="configured_value"`. Offline conflict resolution -deterministically adds, reinforces, relates, or supersedes records while preserving temporal -history; it does not need an LLM. Matching claim identities let it supersede substantially -reworded mutable facts. Without them, the dependency-free lexical embedder cannot reliably infer -that a paraphrase is a contradiction, so keep both records or use an explicit `correct` operation. - ---- - -## Govern memories without losing history - -Engraphis separates automatic write resolution from explicit human governance: - -| Operation | Use it when | What happens to history | -|---|---|---| -| `remember` | Adding or restating one fact | Adds, reinforces, safely supersedes, or relates an uncertain neighbor | -| `correct` | Replacing one known-wrong memory | Closes the old validity window and links the replacement | -| `promote` | A narrow learning now applies more broadly | Writes a wider-scope successor and closes/links the source instead of editing scope in place | -| `merge` | Combining two or more overlapping memories | Retires every source and creates one memory that supersedes all of them | -| `retire` | Removing a memory from live recall | Bi-temporally closes it; the audit/history record remains | -| `consolidate` | Distilling recurring episodic memories automatically | Creates linked semantic digests; source episodes remain live | - -Manual N→1 merge is available through `MemoryService.merge()` and `POST /api/merge`: - -```python -a = mem.remember("Deploys happen Friday at 3pm.", workspace="acme") -b = mem.remember("We deploy Fridays around 15:00.", workspace="acme") - -merged = mem.merge( - [a["id"], b["id"]], - "Deploys ship every Friday at approximately 15:00.", - workspace="acme", - reason="deduplicate the deployment schedule", -) -print(merged["compaction"]) -``` - -`retire` is intentionally not deletion: it preserves temporal history, FTS, and vector -evidence for historical reads. If a credential was captured, new writes are blocked before -storage; for a legacy leak use the explicitly destructive `MemoryService.secure_erase()` or -`POST /api/secure-erase`/`engraphis_secure_erase`. That flow removes the one memory and local -FTS/vector-index and derived graph/link rows, runs SQLite secure-delete, WAL checkpoint, and -VACUUM, and scans recognised local SQLite recovery backups. It cannot erase exports, filesystem -snapshots, remote peers, unknown backups, or information a running/compromised agent already -read; rotate the credential. See [secure-erasure limits](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SECURE_ERASURE.md). `forget` -remains a deprecated compatibility alias for `retire`. - -All sources must belong to the named workspace. The result inherits the strictest source -sensitivity, remains untrusted if any source was untrusted, and stays pinned if any source was -pinned. The full multi-predecessor chain remains visible through inspection, Why, and Timeline. - ---- - -## Free forever vs. hosted plans - -The core engine, local dashboard, MCP server, and manual consolidation are Apache-2.0 and free. -**Pro and Team are services** that provide optional access to the official hosted service; its -control-plane, billing, relay, compute, and Team identity modules live in a private repository. -They do not limit the local core. See -[hosted plans](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/HOSTED_PLANS.md), [licensing](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md), and -[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for service boundaries, lifecycle, and pricing. - -[Subscribe to Pro](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_pricing#billing) -to support the project and add hosted services. - -[Compare hosted plans](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro#billing) -when you are ready to evaluate the service boundary and billing options. - -| | Free (available now) | Pro: $10/mo or $100/yr | Team: $20/seat/mo or $200/seat/yr | -|---|---|---|---| -| Dashboard WebUI (with built-in inspector) | ✓ | ✓ | ✓ | -| Memory engine + Smart MCP (Classic 35-tool compatibility) | ✓ | ✓ | ✓ | -| Version-chain diffs, offline knowledge graph | ✓ | ✓ | ✓ | -| Manual local consolidation (dry-run by default) | ✓ | ✓ | ✓ | -| Local workspace export (portable v2 JSON: memories, source manifests, graph/code evidence, sessions, audit, and receipts) | ✓ | ✓ | ✓ | -| Hosted Cloud Sync | | ✓ | ✓ | -| Hosted Analytics | | ✓ | ✓ | -| Hosted Auto Consolidation + retention policy | | ✓ | ✓ | -| Hosted Auto Dreaming + managed proposals | | ✓ | ✓ | -| Priority support | | ✓ | ✓ | -| Hosted multi-user dashboard: invitations, logins, roles, seat management | | | ✓ | -| Hosted Team audit log + CSV export | | | ✓ | -| 72-hour pending invitations (resend/revoke) | | | ✓ | -| Scoped, expiring per-user agent and sync tokens | | | ✓ | - ---- - -## MCP tools - -Engraphis exposes a zero-configuration Smart MCP gateway plus a 35-tool Classic compatibility -server across memory, recall, code graphs, governance, sessions, and privacy-safe audit receipts. -The focused [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md) is the source for -the full inventory and parameters. - ---- - -## Graphs and privacy-safe receipts - -Memory, entity, and code relationships live in one local graph. Engraphis also provides -content-free operation receipts for inspectable audit evidence. See the -[architecture](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md), [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md), and -[security policy](https://github.com/Coding-Dev-Tools/engraphis/blob/main/SECURITY.md) for the data model, tools, and guarantees. - ---- - -## Cloud sync - -Cloud Sync is an optional hosted Pro/Team service. The public package includes the customer client -and deterministic merge implementation; hosted relay and account operations are separate. See -[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for setup, encryption, merge behavior, and the local folder exchange. - -The public package ships the same sync client as a console script and CLI verb: -`engraphis-sync` (installed entry point), `engraphis sync ...`, and -`python -m scripts.sync --status` for local-only state without network activity. See -[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for -flags, encryption, merge behavior, and the local folder exchange. - ---- - -## Security and trust boundaries - -Engraphis is local-first and binds to loopback by default. Read the -[security policy](https://github.com/Coding-Dev-Tools/engraphis/blob/main/SECURITY.md) before remote deployment or integrating external resources; it -covers supported versions, data protections, threat model, and vulnerability reporting. - ---- - -## Encryption at rest - -Set `ENGRAPHIS_DB_KEY` (or `ENGRAPHIS_DB_KEY_FILE`) and install the extra: - -```bash -pip install "engraphis[encryption]" -``` - -The entire main memory database file is transparently encrypted with AES-256 via SQLCipher; -full-text search, the graph, and every query keep working unchanged. Customer authentication -and managed-service state use their respective deployment protections. When a key is set for the -main database, Engraphis **fails closed with an error** rather than silently falling back to -plaintext. Generate a strong key: - -```bash -python -c "import secrets; print(secrets.token_hex(32))" -``` - -When using `ENGRAPHIS_DB_KEY_FILE`, provision a regular secret file readable only by the -service identity. Engraphis rejects links, reparse points, hard links, malformed text, and -oversized key files rather than following an unexpected filesystem object. - -> An existing plaintext database cannot be opened with a key: migrate it (dump → import -> into a fresh keyed DB). See `.env.example` for all encryption options. - ---- - -## Import files and folders - -The dependency-free universal core scans Markdown, plain text, RST, HTML, JSON/JSONL, CSV/TSV, -configuration/XML text, source code, RTF, DOCX/ODT, XLSX/ODS, PPTX/ODP, and EPUB into the normal -v2 memory path. Installed local resource adapters add PDF text, image OCR, and explicitly -local-model audio/video transcription. -Start with a zero-write -preview, then confirm the same source collection explicitly: - -```bash -engraphis import documents /path/to/collection --workspace acme --dry-run -engraphis import documents /path/to/collection --workspace acme --repo product --yes -``` - -The CLI never downloads an embedding model during import. Use a model that is already cached, -set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path`, or explicitly set -`ENGRAPHIS_EMBED_MODEL` to an empty value to use dependency-free deterministic hashing in -lexical degraded mode. - -The dashboard’s **Import local documents** flow offers the same preview, target scope, source -label, conflict policy, cancellation, and resumable progress. Re-imports are idempotent, -preserve temporal history, and report source removals without hard-deleting memories. Obsidian -remains the rich Markdown adapter for frontmatter, aliases, wikilinks, and attachment references: - -```bash -engraphis import obsidian /path/to/vault --workspace acme --dry-run -``` - -See the [document import guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCUMENT_IMPORT.md) -for supported formats, source safety, resume and conflict behavior, optional adapters, and -limitations; see the [Obsidian adapter guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/OBSIDIAN_IMPORT.md) -for Markdown-specific behavior. - ---- - -## Consolidation and automation - -Manual consolidation is free, local, and dry-run by default; use the dashboard, SDK, CLI, or -MCP. Hosted Pro and Team automation is optional managed compute that produces reviewable -proposals rather than silently changing local data. See [hosted plans](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/HOSTED_PLANS.md), -[licensing](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md), and the [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md) for scope and use. - ---- - -## Configuration - -Values come from the process environment. Engraphis also loads the owner-private -`~/.engraphis/config.env`; `ENGRAPHIS_ENV_FILE` can select another absolute owner-private regular -file. It never searches the working directory for `.env`, and explicit process variables win. - -| Env Var | Default | Description | -|---------|---------|-------------| -| `ENGRAPHIS_ENV_FILE` | `~/.engraphis/config.env` | Optional trusted config leaf selected before trusted values load. Its bounded dependency-free parser performs no interpolation. An explicit value must be an absolute path to an owner-private regular file; arbitrary working-directory `.env` files are ignored. | -| `ENGRAPHIS_DB_PATH` | Source: `/engraphis.db`; installed: platform user-data directory | SQLite database file. Installed defaults are `%LOCALAPPDATA%\engraphis\engraphis.db` (Windows), `~/Library/Application Support/engraphis/engraphis.db` (macOS), and `$XDG_DATA_HOME/engraphis/engraphis.db` or `~/.local/share/engraphis/engraphis.db` (Linux). The environment variable overrides every default; a relative value is resolved from the trusted `~/.engraphis/config.env` directory so launch CWD cannot select a different workspace database. | -| `ENGRAPHIS_SQLITE_DURABILITY` | `durable` | Writable file databases use WAL and FULL commit synchronization. Explicit `balanced` selects NORMAL, which can lose recent acknowledged writes after OS/power failure. Effective settings appear in diagnostics; see [SQLite durability](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SQLITE_DURABILITY.md). | -| `ENGRAPHIS_HOST` | `127.0.0.1` | Server bind address | -| `ENGRAPHIS_PORT` | `8700` | Dashboard port. A platform-injected `$PORT` (Railway/Fly/Heroku) takes precedence over this value for the dashboard bind; Compose pins both to `ENGRAPHIS_COMPOSE_PORT` so the mapping stays in sync | -| `ENGRAPHIS_SERVICE_MODE` | `customer` | The public package supports only `customer`; hosted vendor, relay, compute, and worker roles are not distributed here | -| `ENGRAPHIS_API_TOKEN` | Not set | Optional bearer credential for this single-user local customer node; never reuse a hosted credential | -| `ENGRAPHIS_CORS_ORIGINS` | loopback on `ENGRAPHIS_PORT` | Comma-separated REST CORS allow-list; defaults to `127.0.0.1` and `localhost` on the configured port | -| `ENGRAPHIS_INDEX_ROOTS` | Working, home, and temporary directories | Optional path-separator-delimited absolute-path allow-list that replaces the default roots accepted by local code indexing | -| `ENGRAPHIS_HTTP_INDEX_ROOT` | First `ENGRAPHIS_INDEX_ROOTS` entry, or current directory | Single root for dashboard and REST `POST /api/code/index`; submitted paths resolve beneath it. An explicit root (or fallback entry) must be absolute; an explicit HTTP root is included in the engine-approved set. MCP and CLI indexing continue to use `ENGRAPHIS_INDEX_ROOTS`. | -| `ENGRAPHIS_DB_KEY` | Not set | Encrypt the database at rest (SQLCipher). Or use `ENGRAPHIS_DB_KEY_FILE` | -| `ENGRAPHIS_EMBED_MODEL` | `sentence-transformers/all-MiniLM-L6-v2` | sentence-transformers model | -| `ENGRAPHIS_EMBED_REVISION` | Not set | Optional immutable lowercase 40-hex Hugging Face commit for the embedding model. Loaded Hub commits or local artifact manifests identify persistent vector spaces; unresolved mutable identities keep vector recall fail-closed. | -| `ENGRAPHIS_RERANK_MODEL` | Not set | Optional sentence-transformers cross-encoder reranker | -| `ENGRAPHIS_RERANK_REVISION` | Not set | Optional immutable lowercase 40-hex Hugging Face commit for the reranker | -| `ENGRAPHIS_REQUIRE_IMMUTABLE_MODELS` | `false` | When enabled, require a 40-hex commit before loading remote embedding models, rerankers, or chunk tokenizers; `local:` selectors and filesystem paths remain permitted | -| `ENGRAPHIS_REQUIRE_EXACT_BACKENDS` | `false` | When enabled, dashboard and standalone MCP startup fails if a configured optional backend is unavailable instead of silently falling back | -| `ENGRAPHIS_EXTRACTOR` | `none` | `none` = verbatim; `chunk` = offline structure-aware chunks; `llm` = free-form LLM facts; `llm_structured` = schema-validated facts + graph metadata | -| `ENGRAPHIS_CHUNK_TOKENIZER_MODEL` | Not set | Optional Hugging Face tokenizer used to enforce chunk budgets with the downstream reader's real tokenization; requires the optional `transformers` package | -| `ENGRAPHIS_CHUNK_TOKENIZER_REVISION` | Not set | Optional immutable tokenizer/model revision recorded in the chunk-counter identity; pin this for reproducible benchmark artifacts | -| `ENGRAPHIS_GRAPH_EXTRACTOR` | `regex` | `regex` = offline heuristic NER; `none` = disable heuristic text extraction (validated `llm_structured` metadata still feeds the graph) | -| `ENGRAPHIS_RETENTION_SUPERVISOR` | `none` | `none` = deterministic only; `llm` = sends a bounded excerpt to the configured provider for advisory ephemeral/normal/critical classification | -| `ENGRAPHIS_ALLOW_AUTOMATIC_CRITICAL_RETENTION` | `false` | Opt in only when an LLM supervisor may automatically assign the long-lived `critical` class; explicit user-selected critical retention is unaffected | -| `ENGRAPHIS_WHISPER_MODEL` | Not set | Enables local faster-whisper audio/video transcription | -| `ENGRAPHIS_POSTGRES_DSN` | Not set | CLI-only PostgreSQL source; used for the connection and never stored | -| `ENGRAPHIS_POSTGRES_CONNECT_TIMEOUT` | `10` | PostgreSQL introspection connection timeout in seconds (bounded to 1--120) | -| `ENGRAPHIS_POSTGRES_STATEMENT_TIMEOUT_MS` | `30000` | Per-introspection PostgreSQL statement timeout in milliseconds (bounded to 1--300000) | -| `ENGRAPHIS_GRAPH_TOKEN` | Not set | Bearer token for `engraphis-graph-server`; required off-loopback | -| `ENGRAPHIS_GRAPH_HOST` / `ENGRAPHIS_GRAPH_PORT` | `127.0.0.1` / `8720` | Read-only graph/recall server bind address | -| `ENGRAPHIS_LLM_PROVIDER` | `openai` | `openai \| anthropic \| google \| openrouter \| custom` | -| `ENGRAPHIS_LLM_MODEL` | `gpt-4o-mini` | Model name (provider-specific) | -| `ENGRAPHIS_LLM_API_KEY` | Not set | API key for chat/synthesis, `llm` / `llm_structured` extraction, and structured consolidation | -| `ENGRAPHIS_LLM_BASE_URL` | Not set | Base URL for openrouter / custom OpenAI-compatible endpoints | -| `ENGRAPHIS_LLM_AUTO_EXTRACT` | `0` | Opt in to switching the running engine to `llm_structured` after a successful live connection test; the dashboard's extraction Off button persists `0`, and its On button restores `1` | -| `ENGRAPHIS_FORWARDED_ALLOW_IPS` | *(none)* | Proxies trusted for forwarded client/TLS headers (`*` only when the service is reachable exclusively through that proxy) | -| `ENGRAPHIS_LOCAL_TRUSTED_PEERS` | *(none)* | Exact peers/CIDRs treated as local without forwarding headers; use only for trusted Docker/LAN peers, never public deployments | -| `ENGRAPHIS_UPDATE_CACHE` | `86400` | Update-check cache TTL in seconds, bounded to `1..31622400`; this is never a cache-file path | -| `ENGRAPHIS_UPDATE_CHECK` | Off | Opt-in release reminder surfaced in the dashboard, server startup log, and MCP. Update checks run only when this is set to an affirmative value; `0` keeps them off. | -| `ENGRAPHIS_UPDATE_URL` | Not set | Overrides the release-check source URL; the outbound client accepts HTTPS and rejects private/reserved destinations. | -| `ENGRAPHIS_CLOUD_CONTROL_URL` | hosted default | Official entitlement, organization, and credential control API. A saved rotating credential stays bound to the control endpoint recorded for its family; reconnect to change it. | -| `ENGRAPHIS_CLOUD_COMPUTE_URL` | hosted default | Official Analytics and managed-automation API. A saved rotating credential stays bound to its recorded compute endpoint; reconnect to change it. | -| `ENGRAPHIS_CLOUD_ORGANIZATION_ID` | Not set | Hosted organization bound to this customer session | -| `ENGRAPHIS_CLOUD_REFRESH_CREDENTIAL` | Not set | Bootstrap-only rotating hosted credential; after first use the owner-only cloud session replacement takes precedence | -| `ENGRAPHIS_CLOUD_TOKEN_SUBJECT` | `member` | Subject fixed during hosted bootstrap (`device` or `member`); set explicitly with an environment-only refresh credential | -| `ENGRAPHIS_CLOUD_ACCESS_TOKEN` | Not set | Optional short-lived access token for ephemeral jobs | -| `ENGRAPHIS_MANAGED_COMPUTE_CONSENT` | *(unset)* | Deny-only operator override: `0` pauses readable managed processing. A truthy value cannot grant approval. Each workspace requires explicit confirmation in Manage → Settings; encrypted sync is separate | - -The optional cross-encoder reranker is model- and hardware-dependent. Treat its quality and -latency as deployment-specific until a versioned model identity, exact configuration, and -reproducible evaluation artifact are available for the comparison being reported. - -See `.env.example` for the full variable inventory. Supply those values through the process -environment or the trusted config file above; copying it to an arbitrary `./.env` does not make -Engraphis load it. - -> **Ablation fixture:** `python -m eval.ablation` is an offline deterministic check that prints -> `recall@5` comparisons for vector-only and hybrid retrieval, multi-hop graph arms, and -> retrieval policies, plus ordinary-recall age and semantic-confidence checks. It does not -> produce MRR, hit@5, or ms/query results. Use `python -m eval.reinforcement` for retention -> trajectories, and register evidence before quoting any benchmark results. - ---- - -## Project structure - -``` -engraphis/ -├── engraphis/ -│ ├── core/ # v2 engine: interfaces, store, recall, scoring, schema, sync -│ ├── backends/ # pluggable embedder / vector index / reranker / codegraph / sync transports / encryption -│ ├── factory.py # outer v2 composition root; selects and injects concrete backends -│ ├── service.py # validated MemoryService facade -│ ├── mcp_server.py # Smart MCP gateway + 35-tool Classic compatibility server -│ ├── dashboard_app.py # dashboard WebUI (FastAPI) -│ ├── dashboard_assets/ # primary Ledger interface + graph engine -│ ├── classic_assets/ # selectable full operator dashboard backup -│ ├── read_only_api.py # token-protected recall/repository-graph HTTP surface -│ ├── hosted_client.py # hosted URLs, plan labels, and endpoint validation only -│ ├── licensing.py # compatibility facade for hosted presentation metadata -│ ├── cloud_session.py # rotating hosted customer-session client -│ ├── cloud_features.py # consented managed-feature protocol client -│ ├── config.py / app.py # env settings / REST server -│ └── static/ # compatibility dashboard asset paths -├── eval/ # offline retrieval eval harness + datasets -├── tests/ # offline-first pytest suite and release/security contracts -├── scripts/ # dashboard, server, graph, CLI, connect, update, consolidation, sync -├── docs/ # product, API, hosting, sync, and provider guides -├── Dockerfile / docker-compose.yml -└── pyproject.toml -``` - -New capability belongs in the v2 path (`engraphis/core/`, `engraphis/backends/`, and -`MemoryService`) behind the interfaces in `core/interfaces.py`. Algorithm modules in `core/` -remain backend-agnostic; `engraphis/factory.py` is the outer composition root used by -`engraphis.create_memory_engine()` and the compatibility `MemoryEngine.create()` entry point, then -injects the selected collaborators into `core/engine.py`. The flat-namespace v1 server under -`engraphis/app.py`, `routes/`, `stores/`, and `engines/` remains a -compatibility/reference surface; `engraphis-dashboard`, the MCP server, and the Python quickstart -above use v2. - ---- - -## License - -Apache-2.0. See [LICENSE](https://github.com/Coding-Dev-Tools/engraphis/blob/main/LICENSE) and [NOTICE](https://github.com/Coding-Dev-Tools/engraphis/blob/main/NOTICE). "Engraphis" is a trademark of the -Engraphis project; the license does not grant trademark rights. Code already distributed -under Apache-2.0 keeps that grant; later releases cannot retroactively withdraw it. The -official hosted control plane, its production credentials and records, managed operations, -support, and future separately delivered commercial modules are outside the public source -grant. See [`docs/LICENSING.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md) for the complete boundary. - -### Reliability implementation candidate - -The current source uses schema 18 for durable, content-free vector-index repair and -atomic memory-command receipts. Upgrades use the existing verified-backup migration path. -The [rework execution register](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/REWORK_EXECUTION.md) records the current findings, -compatibility decisions, acceptance evidence, remaining work and recovery procedure. -See [the reliability program](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/RELIABILITY_PROGRAM.md) for exact implementation, -validation, migration and release boundaries. Managed processing now requires explicit -workspace approval in Settings. Existing installations start with readable -uploads paused until confirmed; connecting an account does not grant approval. - -For setup diagnostics use `engraphis-init --check --json`. New configurations get an -owner-private local API token. Existing configs are preserved. Record selected install -capabilities with `engraphis-init --extras server,mcp` or `--extras none`; future updates -preserve that choice. `ENGRAPHIS_UPDATE_EXTRAS` remains an explicit override. +what Engraphis had learned then. `as_of` remains a compatibility alias for `valid_at`; supplying +both is allowed only when they match. + +For a mutable claim, pass a stable `subject_key` and optional `claim_kind`, such as +`subject_key="api.rate_limit", claim_kind="configured_value"`. Offline conflict resolution +deterministically adds, reinforces, relates, or supersedes records while preserving temporal +history; it does not need an LLM. Matching claim identities let it supersede substantially +reworded mutable facts. Without them, the dependency-free lexical embedder cannot reliably infer +that a paraphrase is a contradiction, so keep both records or use an explicit `correct` operation. + +--- + +## Govern memories without losing history + +Engraphis separates automatic write resolution from explicit human governance: + +| Operation | Use it when | What happens to history | +|---|---|---| +| `remember` | Adding or restating one fact | Adds, reinforces, safely supersedes, or relates an uncertain neighbor | +| `correct` | Replacing one known-wrong memory | Closes the old validity window and links the replacement | +| `promote` | A narrow learning now applies more broadly | Writes a wider-scope successor and closes/links the source instead of editing scope in place | +| `merge` | Combining two or more overlapping memories | Retires every source and creates one memory that supersedes all of them | +| `retire` | Removing a memory from live recall | Bi-temporally closes it; the audit/history record remains | +| `consolidate` | Distilling recurring episodic memories automatically | Creates linked semantic digests; source episodes remain live | + +Manual N→1 merge is available through `MemoryService.merge()` and `POST /api/merge`: + +```python +a = mem.remember("Deploys happen Friday at 3pm.", workspace="acme") +b = mem.remember("We deploy Fridays around 15:00.", workspace="acme") + +merged = mem.merge( + [a["id"], b["id"]], + "Deploys ship every Friday at approximately 15:00.", + workspace="acme", + reason="deduplicate the deployment schedule", +) +print(merged["compaction"]) +``` + +`retire` is intentionally not deletion: it preserves temporal history, FTS, and vector +evidence for historical reads. If a credential was captured, new writes are blocked before +storage; for a legacy leak use the explicitly destructive `MemoryService.secure_erase()` or +`POST /api/secure-erase`/`engraphis_secure_erase`. That flow removes the one memory and local +FTS/vector-index and derived graph/link rows, runs SQLite secure-delete, WAL checkpoint, and +VACUUM, and scans recognised local SQLite recovery backups. It cannot erase exports, filesystem +snapshots, remote peers, unknown backups, or information a running/compromised agent already +read; rotate the credential. See [secure-erasure limits](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SECURE_ERASURE.md). `forget` +remains a deprecated compatibility alias for `retire`. + +All sources must belong to the named workspace. The result inherits the strictest source +sensitivity, remains untrusted if any source was untrusted, and stays pinned if any source was +pinned. The full multi-predecessor chain remains visible through inspection, Why, and Timeline. + +--- + +## Free forever vs. hosted plans + +The core engine, local dashboard, MCP server, and manual consolidation are Apache-2.0 and free. +**Pro and Team are services** that provide optional access to the official hosted service; its +control-plane, billing, relay, compute, and Team identity modules live in a private repository. +They do not limit the local core. See +[hosted plans](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/HOSTED_PLANS.md), [licensing](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md), and +[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for service boundaries, lifecycle, and pricing. + +[Subscribe to Pro](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_pricing#billing) +to support the project and add hosted services. + +[Compare hosted plans](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro#billing) +when you are ready to evaluate the service boundary and billing options. + +| | Free (available now) | Pro: $10/mo or $100/yr | Team: $20/seat/mo or $200/seat/yr | +|---|---|---|---| +| Dashboard WebUI (with built-in inspector) | ✓ | ✓ | ✓ | +| Memory engine + Smart MCP (Classic 35-tool compatibility) | ✓ | ✓ | ✓ | +| Version-chain diffs, offline knowledge graph | ✓ | ✓ | ✓ | +| Manual local consolidation (dry-run by default) | ✓ | ✓ | ✓ | +| Local workspace export (portable v2 JSON: memories, source manifests, graph/code evidence, sessions, audit, and receipts) | ✓ | ✓ | ✓ | +| Hosted Cloud Sync | | ✓ | ✓ | +| Hosted Analytics | | ✓ | ✓ | +| Hosted Auto Consolidation + retention policy | | ✓ | ✓ | +| Hosted Auto Dreaming + managed proposals | | ✓ | ✓ | +| Priority support | | ✓ | ✓ | +| Hosted multi-user dashboard: invitations, logins, roles, seat management | | | ✓ | +| Hosted Team audit log + CSV export | | | ✓ | +| 72-hour pending invitations (resend/revoke) | | | ✓ | +| Scoped, expiring per-user agent and sync tokens | | | ✓ | + +--- + +## MCP tools + +Engraphis exposes a zero-configuration Smart MCP gateway plus a 35-tool Classic compatibility +server across memory, recall, code graphs, governance, sessions, and privacy-safe audit receipts. +The focused [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md) is the source for +the full inventory and parameters. + +--- + +## Graphs and privacy-safe receipts + +Memory, entity, and code relationships live in one local graph. Engraphis also provides +content-free operation receipts for inspectable audit evidence. See the +[architecture](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md), [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md), and +[security policy](https://github.com/Coding-Dev-Tools/engraphis/blob/main/SECURITY.md) for the data model, tools, and guarantees. + +--- + +## Cloud sync + +Cloud Sync is an optional hosted Pro/Team service. The public package includes the customer client +and deterministic merge implementation; hosted relay and account operations are separate. See +[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for setup, encryption, merge behavior, and the local folder exchange. + +The public package ships the same sync client as a console script and CLI verb: +`engraphis-sync` (installed entry point), `engraphis sync ...`, and +`python -m scripts.sync --status` for local-only state without network activity. See +[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for +flags, encryption, merge behavior, and the local folder exchange. + +--- + +## Security and trust boundaries + +Engraphis is local-first and binds to loopback by default. Read the +[security policy](https://github.com/Coding-Dev-Tools/engraphis/blob/main/SECURITY.md) before remote deployment or integrating external resources; it +covers supported versions, data protections, threat model, and vulnerability reporting. + +--- + +## Encryption at rest + +Set `ENGRAPHIS_DB_KEY` (or `ENGRAPHIS_DB_KEY_FILE`) and install the extra: + +```bash +pip install "engraphis[encryption]" +``` + +The entire main memory database file is transparently encrypted with AES-256 via SQLCipher; +full-text search, the graph, and every query keep working unchanged. Customer authentication +and managed-service state use their respective deployment protections. When a key is set for the +main database, Engraphis **fails closed with an error** rather than silently falling back to +plaintext. Generate a strong key: + +```bash +python -c "import secrets; print(secrets.token_hex(32))" +``` + +When using `ENGRAPHIS_DB_KEY_FILE`, provision a regular secret file readable only by the +service identity. Engraphis rejects links, reparse points, hard links, malformed text, and +oversized key files rather than following an unexpected filesystem object. + +> An existing plaintext database cannot be opened with a key: migrate it (dump → import +> into a fresh keyed DB). See `.env.example` for all encryption options. + +--- + +## Import files and folders + +The dependency-free universal core scans Markdown, plain text, RST, HTML, JSON/JSONL, CSV/TSV, +configuration/XML text, source code, RTF, DOCX/ODT, XLSX/ODS, PPTX/ODP, and EPUB into the normal +v2 memory path. Installed local resource adapters add PDF text, image OCR, and explicitly +local-model audio/video transcription. +Start with a zero-write +preview, then confirm the same source collection explicitly: + +```bash +engraphis import documents /path/to/collection --workspace acme --dry-run +engraphis import documents /path/to/collection --workspace acme --repo product --yes +``` + +The CLI never downloads an embedding model during import. Use a model that is already cached, +set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path`, or explicitly set +`ENGRAPHIS_EMBED_MODEL` to an empty value to use dependency-free deterministic hashing in +lexical degraded mode. + +The dashboard’s **Import local documents** flow offers the same preview, target scope, source +label, conflict policy, cancellation, and resumable progress. Re-imports are idempotent, +preserve temporal history, and report source removals without hard-deleting memories. Obsidian +remains the rich Markdown adapter for frontmatter, aliases, wikilinks, and attachment references: + +```bash +engraphis import obsidian /path/to/vault --workspace acme --dry-run +``` + +See the [document import guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCUMENT_IMPORT.md) +for supported formats, source safety, resume and conflict behavior, optional adapters, and +limitations; see the [Obsidian adapter guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/OBSIDIAN_IMPORT.md) +for Markdown-specific behavior. + +--- + +## Consolidation and automation + +Manual consolidation is free, local, and dry-run by default; use the dashboard, SDK, CLI, or +MCP. Hosted Pro and Team automation is optional managed compute that produces reviewable +proposals rather than silently changing local data. See [hosted plans](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/HOSTED_PLANS.md), +[licensing](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md), and the [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md) for scope and use. + +--- + +## Configuration + +Values come from the process environment. Engraphis also loads the owner-private +`~/.engraphis/config.env`; `ENGRAPHIS_ENV_FILE` can select another absolute owner-private regular +file. It never searches the working directory for `.env`, and explicit process variables win. + +| Env Var | Default | Description | +|---------|---------|-------------| +| `ENGRAPHIS_ENV_FILE` | `~/.engraphis/config.env` | Optional trusted config leaf selected before trusted values load. Its bounded dependency-free parser performs no interpolation. An explicit value must be an absolute path to an owner-private regular file; arbitrary working-directory `.env` files are ignored. | +| `ENGRAPHIS_DB_PATH` | Source: `/engraphis.db`; installed: platform user-data directory | SQLite database file. Installed defaults are `%LOCALAPPDATA%\engraphis\engraphis.db` (Windows), `~/Library/Application Support/engraphis/engraphis.db` (macOS), and `$XDG_DATA_HOME/engraphis/engraphis.db` or `~/.local/share/engraphis/engraphis.db` (Linux). The environment variable overrides every default; a relative value is resolved from the trusted `~/.engraphis/config.env` directory so launch CWD cannot select a different workspace database. | +| `ENGRAPHIS_SQLITE_DURABILITY` | `durable` | Writable file databases use WAL and FULL commit synchronization. Explicit `balanced` selects NORMAL, which can lose recent acknowledged writes after OS/power failure. Effective settings appear in diagnostics; see [SQLite durability](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SQLITE_DURABILITY.md). | +| `ENGRAPHIS_HOST` | `127.0.0.1` | Server bind address | +| `ENGRAPHIS_PORT` | `8700` | Dashboard port. A platform-injected `$PORT` (Railway/Fly/Heroku) takes precedence over this value for the dashboard bind; Compose pins both to `ENGRAPHIS_COMPOSE_PORT` so the mapping stays in sync | +| `ENGRAPHIS_SERVICE_MODE` | `customer` | The public package supports only `customer`; hosted vendor, relay, compute, and worker roles are not distributed here | +| `ENGRAPHIS_API_TOKEN` | Not set | Optional bearer credential for this single-user local customer node; never reuse a hosted credential | +| `ENGRAPHIS_CORS_ORIGINS` | loopback on `ENGRAPHIS_PORT` | Comma-separated REST CORS allow-list; defaults to `127.0.0.1` and `localhost` on the configured port | +| `ENGRAPHIS_INDEX_ROOTS` | Working, home, and temporary directories | Optional path-separator-delimited absolute-path allow-list that replaces the default roots accepted by local code indexing | +| `ENGRAPHIS_HTTP_INDEX_ROOT` | First `ENGRAPHIS_INDEX_ROOTS` entry, or current directory | Single root for dashboard and REST `POST /api/code/index`; submitted paths resolve beneath it. An explicit root (or fallback entry) must be absolute; an explicit HTTP root is included in the engine-approved set. MCP and CLI indexing continue to use `ENGRAPHIS_INDEX_ROOTS`. | +| `ENGRAPHIS_DB_KEY` | Not set | Encrypt the database at rest (SQLCipher). Or use `ENGRAPHIS_DB_KEY_FILE` | +| `ENGRAPHIS_EMBED_MODEL` | `sentence-transformers/all-MiniLM-L6-v2` | sentence-transformers model | +| `ENGRAPHIS_EMBED_REVISION` | Not set | Optional immutable lowercase 40-hex Hugging Face commit for the embedding model. Loaded Hub commits or local artifact manifests identify persistent vector spaces; unresolved mutable identities keep vector recall fail-closed. | +| `ENGRAPHIS_RERANK_MODEL` | Not set | Optional sentence-transformers cross-encoder reranker | +| `ENGRAPHIS_RERANK_REVISION` | Not set | Optional immutable lowercase 40-hex Hugging Face commit for the reranker | +| `ENGRAPHIS_REQUIRE_IMMUTABLE_MODELS` | `false` | When enabled, require a 40-hex commit before loading remote embedding models, rerankers, or chunk tokenizers; `local:` selectors and filesystem paths remain permitted | +| `ENGRAPHIS_REQUIRE_EXACT_BACKENDS` | `false` | When enabled, dashboard and standalone MCP startup fails if a configured optional backend is unavailable instead of silently falling back | +| `ENGRAPHIS_EXTRACTOR` | `none` | `none` = verbatim; `chunk` = offline structure-aware chunks; `llm` = free-form LLM facts; `llm_structured` = schema-validated facts + graph metadata | +| `ENGRAPHIS_CHUNK_TOKENIZER_MODEL` | Not set | Optional Hugging Face tokenizer used to enforce chunk budgets with the downstream reader's real tokenization; requires the optional `transformers` package | +| `ENGRAPHIS_CHUNK_TOKENIZER_REVISION` | Not set | Optional immutable tokenizer/model revision recorded in the chunk-counter identity; pin this for reproducible benchmark artifacts | +| `ENGRAPHIS_GRAPH_EXTRACTOR` | `regex` | `regex` = offline heuristic NER; `none` = disable heuristic text extraction (validated `llm_structured` metadata still feeds the graph) | +| `ENGRAPHIS_RETENTION_SUPERVISOR` | `none` | `none` = deterministic only; `llm` = sends a bounded excerpt to the configured provider for advisory ephemeral/normal/critical classification | +| `ENGRAPHIS_ALLOW_AUTOMATIC_CRITICAL_RETENTION` | `false` | Opt in only when an LLM supervisor may automatically assign the long-lived `critical` class; explicit user-selected critical retention is unaffected | +| `ENGRAPHIS_WHISPER_MODEL` | Not set | Enables local faster-whisper audio/video transcription | +| `ENGRAPHIS_POSTGRES_DSN` | Not set | CLI-only PostgreSQL source; used for the connection and never stored | +| `ENGRAPHIS_POSTGRES_CONNECT_TIMEOUT` | `10` | PostgreSQL introspection connection timeout in seconds (bounded to 1--120) | +| `ENGRAPHIS_POSTGRES_STATEMENT_TIMEOUT_MS` | `30000` | Per-introspection PostgreSQL statement timeout in milliseconds (bounded to 1--300000) | +| `ENGRAPHIS_GRAPH_TOKEN` | Not set | Bearer token for `engraphis-graph-server`; required off-loopback | +| `ENGRAPHIS_GRAPH_HOST` / `ENGRAPHIS_GRAPH_PORT` | `127.0.0.1` / `8720` | Read-only graph/recall server bind address | +| `ENGRAPHIS_LLM_PROVIDER` | `openai` | `openai \| anthropic \| google \| openrouter \| custom` | +| `ENGRAPHIS_LLM_MODEL` | `gpt-4o-mini` | Model name (provider-specific) | +| `ENGRAPHIS_LLM_API_KEY` | Not set | API key for chat/synthesis, `llm` / `llm_structured` extraction, and structured consolidation | +| `ENGRAPHIS_LLM_BASE_URL` | Not set | Base URL for openrouter / custom OpenAI-compatible endpoints | +| `ENGRAPHIS_LLM_AUTO_EXTRACT` | `0` | Opt in to switching the running engine to `llm_structured` after a successful live connection test; the dashboard's extraction Off button persists `0`, and its On button restores `1` | +| `ENGRAPHIS_FORWARDED_ALLOW_IPS` | *(none)* | Proxies trusted for forwarded client/TLS headers (`*` only when the service is reachable exclusively through that proxy) | +| `ENGRAPHIS_LOCAL_TRUSTED_PEERS` | *(none)* | Exact peers/CIDRs treated as local without forwarding headers; use only for trusted Docker/LAN peers, never public deployments | +| `ENGRAPHIS_UPDATE_CACHE` | `86400` | Update-check cache TTL in seconds, bounded to `1..31622400`; this is never a cache-file path | +| `ENGRAPHIS_UPDATE_CHECK` | Off | Opt-in release reminder surfaced in the dashboard, server startup log, and MCP. Update checks run only when this is set to an affirmative value; `0` keeps them off. | +| `ENGRAPHIS_UPDATE_URL` | Not set | Overrides the release-check source URL; the outbound client accepts HTTPS and rejects private/reserved destinations. | +| `ENGRAPHIS_CLOUD_CONTROL_URL` | hosted default | Official entitlement, organization, and credential control API. A saved rotating credential stays bound to the control endpoint recorded for its family; reconnect to change it. | +| `ENGRAPHIS_CLOUD_COMPUTE_URL` | hosted default | Official Analytics and managed-automation API. A saved rotating credential stays bound to its recorded compute endpoint; reconnect to change it. | +| `ENGRAPHIS_CLOUD_ORGANIZATION_ID` | Not set | Hosted organization bound to this customer session | +| `ENGRAPHIS_CLOUD_REFRESH_CREDENTIAL` | Not set | Bootstrap-only rotating hosted credential; after first use the owner-only cloud session replacement takes precedence | +| `ENGRAPHIS_CLOUD_TOKEN_SUBJECT` | `member` | Subject fixed during hosted bootstrap (`device` or `member`); set explicitly with an environment-only refresh credential | +| `ENGRAPHIS_CLOUD_ACCESS_TOKEN` | Not set | Optional short-lived access token for ephemeral jobs | +| `ENGRAPHIS_MANAGED_COMPUTE_CONSENT` | *(unset)* | Deny-only operator override: `0` pauses readable managed processing. A truthy value cannot grant approval. Each workspace requires explicit confirmation in Manage → Settings; encrypted sync is separate | + +The optional cross-encoder reranker is model- and hardware-dependent. Treat its quality and +latency as deployment-specific until a versioned model identity, exact configuration, and +reproducible evaluation artifact are available for the comparison being reported. + +See `.env.example` for the full variable inventory. Supply those values through the process +environment or the trusted config file above; copying it to an arbitrary `./.env` does not make +Engraphis load it. + +> **Ablation fixture:** `python -m eval.ablation` is an offline deterministic check that prints +> `recall@5` comparisons for vector-only and hybrid retrieval, multi-hop graph arms, and +> retrieval policies, plus ordinary-recall age and semantic-confidence checks. It does not +> produce MRR, hit@5, or ms/query results. Use `python -m eval.reinforcement` for retention +> trajectories, and register evidence before quoting any benchmark results. + +--- + +## Project structure + +``` +engraphis/ +├── engraphis/ +│ ├── core/ # v2 engine: interfaces, store, recall, scoring, schema, sync +│ ├── backends/ # pluggable embedder / vector index / reranker / codegraph / sync transports / encryption +│ ├── factory.py # outer v2 composition root; selects and injects concrete backends +│ ├── service.py # validated MemoryService facade +│ ├── mcp_server.py # Smart MCP gateway + 35-tool Classic compatibility server +│ ├── dashboard_app.py # dashboard WebUI (FastAPI) +│ ├── dashboard_assets/ # primary Ledger interface + graph engine +│ ├── classic_assets/ # selectable full operator dashboard backup +│ ├── read_only_api.py # token-protected recall/repository-graph HTTP surface +│ ├── hosted_client.py # hosted URLs, plan labels, and endpoint validation only +│ ├── licensing.py # compatibility facade for hosted presentation metadata +│ ├── cloud_session.py # rotating hosted customer-session client +│ ├── cloud_features.py # consented managed-feature protocol client +│ ├── config.py / app.py # env settings / REST server +│ └── static/ # compatibility dashboard asset paths +├── eval/ # offline retrieval eval harness + datasets +├── tests/ # offline-first pytest suite and release/security contracts +├── scripts/ # dashboard, server, graph, CLI, connect, update, consolidation, sync +├── docs/ # product, API, hosting, sync, and provider guides +├── Dockerfile / docker-compose.yml +└── pyproject.toml +``` + +New capability belongs in the v2 path (`engraphis/core/`, `engraphis/backends/`, and +`MemoryService`) behind the interfaces in `core/interfaces.py`. Algorithm modules in `core/` +remain backend-agnostic; `engraphis/factory.py` is the outer composition root used by +`engraphis.create_memory_engine()` and the compatibility `MemoryEngine.create()` entry point, then +injects the selected collaborators into `core/engine.py`. The flat-namespace v1 server under +`engraphis/app.py`, `routes/`, `stores/`, and `engines/` remains a +compatibility/reference surface; `engraphis-dashboard`, the MCP server, and the Python quickstart +above use v2. + +--- + +## License + +Apache-2.0. See [LICENSE](https://github.com/Coding-Dev-Tools/engraphis/blob/main/LICENSE) and [NOTICE](https://github.com/Coding-Dev-Tools/engraphis/blob/main/NOTICE). "Engraphis" is a trademark of the +Engraphis project; the license does not grant trademark rights. Code already distributed +under Apache-2.0 keeps that grant; later releases cannot retroactively withdraw it. The +official hosted control plane, its production credentials and records, managed operations, +support, and future separately delivered commercial modules are outside the public source +grant. See [`docs/LICENSING.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md) for the complete boundary. + +### Reliability implementation candidate + +The current source uses schema 18 for durable, content-free vector-index repair and +atomic memory-command receipts. Upgrades use the existing verified-backup migration path. +The [rework execution register](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/REWORK_EXECUTION.md) records the current findings, +compatibility decisions, acceptance evidence, remaining work and recovery procedure. +See [the reliability program](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/RELIABILITY_PROGRAM.md) for exact implementation, +validation, migration and release boundaries. Managed processing now requires explicit +workspace approval in Settings. Existing installations start with readable +uploads paused until confirmed; connecting an account does not grant approval. + +For setup diagnostics use `engraphis-init --check --json`. New configurations get an +owner-private local API token. Existing configs are preserved. Record selected install +capabilities with `engraphis-init --extras server,mcp` or `--extras none`; future updates +preserve that choice. `ENGRAPHIS_UPDATE_EXTRAS` remains an explicit override. diff --git a/docs/RELEASE_QUALIFICATION.md b/docs/RELEASE_QUALIFICATION.md index 2c2433d0..41b3375e 100644 --- a/docs/RELEASE_QUALIFICATION.md +++ b/docs/RELEASE_QUALIFICATION.md @@ -110,8 +110,11 @@ the peeled release tag commit, never the repair workflow's `main` checkout commi For the existing `v1.7.6` release only, the repository owner explicitly directed a qualification waiver on 2026-09-27. The `workflow_dispatch` input `waive_v176_qualification` skips the owner qualification verifier only when repairing -`v1.7.6`; the workflow records the actor and run URL, and the GitHub Release notes -state that full-product qualification was waived. This is not a qualification and +`v1.7.6` at commit `6a441a75c8dd159607fa3933da83f600864b9146`. Reusing that +tag for another commit cannot use this exception. The workflow records the actor +and run URL, and publishes the waiver in GitHub Release notes before the first +PyPI write. A failed disclosure prevents publication; a later repair failure +leaves the public disclosure in place. This is not a qualification and does not mark any unverified gate as passing. All ordinary tag publications and repairs for other versions still require a valid owner-signed qualification. diff --git a/docs/benchmark-evidence/offline-fixtures-v76.json b/docs/benchmark-evidence/offline-fixtures-v76.json new file mode 100644 index 00000000..1d986d58 --- /dev/null +++ b/docs/benchmark-evidence/offline-fixtures-v76.json @@ -0,0 +1,690 @@ +{ + "environment": { + "embedding": "deterministic", + "numpy": "2.4.5", + "platform": "win32", + "python": "3.12.10", + "vector_backend": "numpy" + }, + "generated_on": "2026-09-27", + "privacy": { + "contains_answers": false, + "contains_customer_data": false, + "contains_per_record_fingerprints": false, + "contains_prompts": false, + "contains_raw_questions": false + }, + "runs": [ + { + "boundary": "Deterministic offline retrieval fixture; normalized-character token estimator; not external QA or provider billing.", + "command": "python -m eval.chunking_eval --dataset eval/datasets/longdoc.jsonl --k 5", + "config_digest": "c1c8196aa7e1568ef3844a9fb2d76b87f342c39108e32d6ad144b885a76143b8", + "config_digest_method": "sha256(UTF-8 exact command)", + "id": "offline-chunking", + "result": { + "chunked": { + "max_stored_tokens": 59, + "mean_context_tokens": 214.3, + "mean_evidence_tokens": 42.4, + "memories": 24, + "recall_at_k": 1.0 + }, + "context_reduction_pct": 71.1, + "documents": 6, + "k": 5, + "questions": 18, + "token_counter": "engraphis.chars4.v1", + "whole": { + "max_stored_tokens": 213, + "mean_context_tokens": 740.3, + "mean_evidence_tokens": 162.2, + "memories": 6, + "recall_at_k": 1.0 + } + } + }, + { + "boundary": "Deterministic offline CodeMem fixture; serialized JSON-shape payload proxies, not MCP transport responses, provider billing, or latency claims.", + "command": "python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 --json", + "config_digest": "bbe4aca81e58d4830e50a8fc7729a1d15b71d97a6299bccd79432b7f119677d7", + "config_digest_method": "sha256(UTF-8 exact command)", + "id": "offline-performance", + "result": { + "answer_token_recall": 1.0, + "compact_serialized_payload_tokens": 11138, + "dataset_cases": 14, + "full_serialized_payload_tokens": 24590, + "hit_at_k": 1.0, + "k": 5, + "max_context_tokens": 108, + "mean_context_tokens": 85.38, + "memories": 44, + "packed_quality": { + "answer_token_recall": 1.0, + "hit_at_k": 1.0, + "recall_at_k": 1.0, + "sample_count": 26 + }, + "payload_boundary": { + "kind": "serialized_json_shape_proxy", + "mcp_envelope_serialized": false, + "token_counter": "engraphis.regex.v1", + "transport_measured": false + }, + "quality_scope": { + "packed": "packed_quality fields score only chunks admitted to reader context", + "retrieved": "legacy quality fields score all candidate chunks returned before context packing" + }, + "questions": 26, + "recall_at_k": 1.0, + "saved_serialized_payload_tokens": 13452, + "serialized_payload_savings_ratio": 0.5471, + "timed_recalls": 260, + "token_budget": 1500, + "token_counter": "engraphis.regex.v1" + } + }, + { + "boundary": "Deterministic offline support/abstention fixture; not a frontier-model answer-quality score.", + "command": "python -m eval.grounded", + "config_digest": "590442e51e3642c10489165759919dc86ffac62c182937330c153e7f8d5fc26f", + "config_digest_method": "sha256(UTF-8 exact command)", + "id": "offline-grounded", + "result": { + "abstained": 6, + "answerable": 5, + "decision_accuracy": 1.0, + "grounded": 5, + "off_topic": 6, + "quarantine_hits": 1, + "quarantined": 1 + } + } + ], + "schema": "engraphis-public-offline-fixtures/v1", + "suite": { + "digest": "20131c25c86e3ac5a53285d0631d4ff0c60946879b04ff7a7c38e488f42cda51", + "digest_method": "sha256(canonical compact JSON mapping each sorted path to its file SHA-256)", + "files": { + "engraphis/__init__.py": "a2c5d147ea025e6249d99f4ae2752576c8b1df382393ec450a3ef1b03fcf315a", + "engraphis/ai_context.py": "4dfd5d39eb95d05c591d1981e9e855eced73232f53efd9fe9966e0e08957d030", + "engraphis/app.py": "44d68ad8c0ff46baed01978c9b04e40b8be5031609d5a69e205b7f05da3777fb", + "engraphis/backends/__init__.py": "a9f22b9278362904166614081f1df78469d453601b298ce4e8afdba8a3722b25", + "engraphis/backends/codegraph.py": "83e723a91068d23694092fbe00157fcb2597061d453eb36f955bf55bc5b4d35e", + "engraphis/backends/embedder_api.py": "56a6bceea4f757325dcea987b0339e41d3875634b1ae5103f0537a358cf5878b", + "engraphis/backends/embedder_deterministic.py": "ec8b23de7e7e8273416125f5876ca96f55e4ae7881841bae55783ab0ba9130ad", + "engraphis/backends/embedder_st.py": "e1c20fd980e07060387e3f9fa37fe02a916959fc8abce4e6067a62699c726de1", + "engraphis/backends/encrypted_db.py": "25f6c1480d296a88f317213700a8b3c81e2732399bfac464b0d893465b25e846", + "engraphis/backends/extractor.py": "f2e3455ab7f14caee1d5b5c4ef071e498e90118f1b0ddcb8510c969583b0fc57", + "engraphis/backends/graph_extractor.py": "88561efa0d3fabc447a0a005b10e36261379d46218cf62d905e6928cd2fda676", + "engraphis/backends/model_source.py": "8c3c7681f95214a2bbabd8de222e5ee11f42fe13402d27365654ae75fb363d4e", + "engraphis/backends/postgres_schema.py": "8468578c3add701d30d5eaa36d768ded2375d116e55f1836e09f6107a07a267e", + "engraphis/backends/query_planner.py": "bbdd77afc9b5523421b85b2ae63c8da7f5a7b777265450e0d21708a83e7bb23c", + "engraphis/backends/reranker.py": "747761d6cbfa421388974bcfd98d844f92391d80f4bf6a4feca00b0c7a6908ca", + "engraphis/backends/resources.py": "47cc867c3aecc8bd95fa284bc5bb04715f3339c19a0a11512973ef6171c95944", + "engraphis/backends/retention.py": "381d9371e3951d762f8b55eb54711de5697642acb39de99a714f246c059ecbd0", + "engraphis/backends/sync_folder.py": "e4f70a92a17f6a365910670df041e6e3ca421d44ada2827917cd66b4dc067bfa", + "engraphis/backends/sync_relay.py": "b8b9ad265453aba17ba7c27a355e12a793469b3e44cb217943c6fad9382a3006", + "engraphis/backends/vector_numpy.py": "c598831bea547824cfe08844816fa79857d3617cb0631238f95dad955a424f72", + "engraphis/backends/vector_sqlitevec.py": "6148e14ceaacc19239b64c642a3fba0e98797c78cec356210157afadf475a08b", + "engraphis/build_info.py": "624c22471e56d4c4047160808c4245488292af611564d1a63ca437605bbb414f", + "engraphis/classic_assets/__init__.py": "a7c1d52b285e3faa670ce231814b5758754aa0fbd05e1428e74c20c3ec51a4f1", + "engraphis/cloud_authz.py": "e80500579cb3a1d5fbf30814dc94e3e3967e50b311e8ed2fa56afcc13f7eb565", + "engraphis/cloud_features.py": "90e876f8993d99f01760cd01ae9fb63140e3f65036706e8fb0c8224560416330", + "engraphis/cloud_session.py": "7c83d7b85665c05aa4da2597a6b4ad2b951f1f8f4af20105d9d1750b9b123c2d", + "engraphis/commercial.py": "184f312066a9e682e51a0abeff042f1c0e8eed2d47470157b23930b5a17633aa", + "engraphis/config.py": "a46f3a335fad343e7df32942be91c04094fbafaf14b8b11f5fc433ef74743226", + "engraphis/core/__init__.py": "dd5143729c3939237f04636f437032b1f2d3a5f7d82c91bbc2a5a283c3f0ebaa", + "engraphis/core/adaptive_context.py": "cc5ce48109bb0d5230a5b2b8424b829c853feec5b5d5b82596413f2279b0c9f9", + "engraphis/core/browsing.py": "cfae752d52ef51b17c4ffbe44dde62d5b0e1ed0ca34ec0e1788f3991d94bf05f", + "engraphis/core/codegraph_export.py": "4641074258d7b23498f92dd45053a0fbb111863eaad2001c08e5e3c2dc2fd54f", + "engraphis/core/conflicts.py": "28530be25a4af0bffd8f609b965b33ca7f93199a70887789b2148fda8a61a486", + "engraphis/core/consolidate.py": "f66eedfe08319a64b761261ccf1c99b163eba564d6d4060534bf8c575aecf232", + "engraphis/core/context.py": "8cc15746e4de88bce7d34db8c11dfaa545f17305788e2167a286179171717fb9", + "engraphis/core/diagnostics.py": "5ba449bdf5087d8ccb5808379596e695c8f405da497372e99f2bac509f96e8e4", + "engraphis/core/documents.py": "84385db39ba44e06b58b4b26dbf954228ff4abed7230f28a7280166fa6457861", + "engraphis/core/engine.py": "f1ab3940f7a185f1e12b1a0b49857db8873c83313e8741822a3c80d818e59599", + "engraphis/core/evidence.py": "97912b52a3d22de909218f83c09c92572ed695d7b3c2117e5330271c6d84a7bd", + "engraphis/core/fsutil.py": "6db770fa8bd3e1a57dfa70eb8e8bc46d48c2dc53ead1b0b43eeecf085ef58cfc", + "engraphis/core/graph_layers.py": "64d74ab01c77119f6343ba6f1d6a84f9653f1a5d34d47ce7966f3ac31b29d2ea", + "engraphis/core/graph_policy.py": "ec5b373d01adb2de87df31d9f543130018e9a73faaaed239a184f14a32646615", + "engraphis/core/graph_scene.py": "cab0aef9ce464620d1ee9ff88fab13fd89707c831212e1d8f1ae2c603c8b6010", + "engraphis/core/graphrank.py": "1279a58396104d3f906bfd5ec75b32efedfefe52201467bf19d80be3517017a5", + "engraphis/core/grounded.py": "ac40057dc37aa907e09c68e6b8cd54a989525fb310097df2970858708870ba3e", + "engraphis/core/ids.py": "e47eeadfaf560bc7638e1879fc491fa82d981cd42b9e4b05ef87033b2d7dfa60", + "engraphis/core/interfaces.py": "be11f909f0f59aca774b4cf7aeacf831123a5dc3fc1b889c47435b959e981266", + "engraphis/core/mutations.py": "dbb46a97686994e1652b2698e50c3309ff428d7258d6e8e53cbad7bb55b459aa", + "engraphis/core/obsidian.py": "991267c153cb7c4c40f7fe8f50aa71688892e250aaaf383b22c9e5910dc263b7", + "engraphis/core/poisoning.py": "5bc67169ee8032f3777f2d3969dcf71821437bd473ce4f917c4b41a50e845fb6", + "engraphis/core/query_planner.py": "249062d67392ab7c203cc71e9040e99bee91bf570604e90949149a93cb652120", + "engraphis/core/read_snapshots.py": "be08e63a88bd38ed91d61b28657a65201c38994856db2798b209a73151dd202a", + "engraphis/core/recall.py": "34959532fac07089ca8654fad0a1e945b11791918b3890420380ac7a9b78ccac", + "engraphis/core/resolve.py": "f01a6f55e44320ab04b97e516342f20155668863b2fc4765305e066d87586524", + "engraphis/core/retention_policy.py": "864c03bdb6e743cd0002c706de471e920f1fa1f1a9918ab343ef4c2042b47429", + "engraphis/core/retrieval_policy.py": "d169eb442115bd06c1e1da776fa6c1849e0795e658edd732a8d6821326b46b03", + "engraphis/core/savings.py": "cfbcfc7e476f4e28028555cd519696e23099f6210cf0b733225832aeaa0bc7dc", + "engraphis/core/schema.py": "ac273d3f0383995be815bbd866f2a36ea1398f30833aed57b8d4459afc87096b", + "engraphis/core/scoring.py": "f5b6ac291edf0968b3de83cb1951a97d5bfd8a8d079199cb2ef95bba0884c89a", + "engraphis/core/secrets.py": "a4835ba06e2616156528df6365ca1aba6cba0c97cdf834a709d2797a371099d2", + "engraphis/core/store.py": "2bcfda14325c6a846a45d87a7511a445d197618570eca0cf7f0c833596c2260c", + "engraphis/core/sync.py": "69f75b50fdb1ec9352f92460c89efeec8d72ca642bd10beb29474a8c62b4f58a", + "engraphis/core/textutil.py": "acd65031729fa5d91d09527b8eb52518b83cfce77a55e6b7ae2d94686e35c1a6", + "engraphis/core/user_model.py": "3147ec8ee7cfd855783f639874b63f331cdc822bd8bd9298cfb326dd18666026", + "engraphis/core/vector_repair.py": "a8c1812de4e3ed288eda3e54a505e136d6ebd3ec296788bcbb8804b11e13cfc8", + "engraphis/core/vector_search.py": "75800052e573e9af01c6fd098eaf8647c3c45cd7eb2601dae05d42359e19b234", + "engraphis/dashboard_app.py": "9305c9c72c43027b2b79c0ce559d398fc232929968d6ea03c37065e3227c8d51", + "engraphis/dashboard_assets/__init__.py": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "engraphis/device_connect.py": "c8cd0a22e9fd2d92a65bd74047cc3a8b7159bfc9a297cb639c1c7f1699f8802e", + "engraphis/document_import.py": "94fa0ca340ad0ebd060b143f46657798a81c440a11b82aa46fefa83fb65295dc", + "engraphis/engines/__init__.py": "111232af583889195c5f5a60298e32484348608e81d2bdd822fd3ecd2a33c1e4", + "engraphis/engines/embedder.py": "998b65dd566966bb6581fd9f09cdf46c58a3923ef5df3073ccdaa541153a3581", + "engraphis/engines/ingest.py": "1a5d4b52c13e533864329f9fff11c0f6ebc299ff7093a9d9275e5c9626a39ce2", + "engraphis/engines/intelligence.py": "b561589b98deb98271f104dfec6276aeba13c4815e627259be5f8769f2e7ad82", + "engraphis/engines/recall.py": "f979580d065599c07acbc3add59f71e52d9a186a2104b0e257daa7959ea88d26", + "engraphis/engines/reweight.py": "91ec5815f5d356a7068c36133d24405334450f361f902f99275bed3ccebffd49", + "engraphis/engines/thoughts.py": "4adb9c8a9bcfe736cb42fff9b5ce24631da6ec473d178e83f1c176fde3b8b814", + "engraphis/factory.py": "3b8765834cb04b052b89a9964b357b4ffb9c41a0f769bf1261768033653f708d", + "engraphis/graphdata.py": "195e583a4d4aefe97d8cff635cabea1ff9cd3f11c54ad392d951765d89910863", + "engraphis/hosted_client.py": "9c67aff6881c29704498f4ca99f836009be4ea58a248d74459c536dac7341a88", + "engraphis/http_security.py": "596981e96741fd47064d03db409605bbb435f062c0c6ef69b040aacdd8763a20", + "engraphis/inspector/__init__.py": "720cac28b8a6019d0a0c53809d5905b7d6eb9d4ecbac767505904c6ec3c39071", + "engraphis/inspector/app.py": "8aefe8a935397dde3c6466e691a7e6a7203ebe42e07aa00c332b733465e59e04", + "engraphis/licensing.py": "7e73c28b0e1c3536e2080a129af614f838d2cae3ec3a48a3d8ed7e5153c9e935", + "engraphis/llm/__init__.py": "f3096d2ddd652b99e6fc0b4c5a8786d9bbaba41b259df0b44ce93b57bde3a8b1", + "engraphis/llm/client.py": "304330bf7a45b0eef8f5beabee419b971d205c55f686de86c26d91fea69d172e", + "engraphis/local_auth.py": "b0ad3a1926d417a2aaa6c44a7dbf6e51f575290ddfa27875b78a03db176623c3", + "engraphis/logging_setup.py": "f7d2edc756458a852e0401453e9c71b785aba08fa7859aacbdc23fb30dc7982a", + "engraphis/managed_processing.py": "6d33cdfd10800d9552fcfae3c2b071d2b39fe10fb69d1a8e5d9ed6026882197e", + "engraphis/mcp_classic_cli.py": "778122b121a1c654b8f85810fad05d9c8d7799cf13903d7dc274938b51008da0", + "engraphis/mcp_cli.py": "ed2d997438d727180842dc5fb3f6776f5a5d97974f690ee7229264be1ff61957", + "engraphis/mcp_http_cli.py": "f6ac8bb04a0cb179e840502fdd543b8f55d0f1bf0045fc7900e06ae002cf422d", + "engraphis/mcp_server.py": "5a76a9b82e94a7be526307172fa5a0a594ddfec8c3989b0044db4cf40ff64775", + "engraphis/models.py": "6e76e97db0aca3805c6f582ea78cc6e1c0ccd2665e16fd8ab81eb91b94141b51", + "engraphis/netutil.py": "2e0f8a9095f6f31dcb5b96d323f023b9214369d59e1c488125a7d443a55972ff", + "engraphis/observability.py": "a3a6945bf33a0d8e216da56ca7efec0031b82dc64a66cff36f0b2161f5db4981", + "engraphis/obsidian_import.py": "c7cf4b5e993afce3ffb45f01637804719cb9e7cd2461ffe629c055be5975721f", + "engraphis/private_state.py": "7485570efcaee8a1dc00b64ebfdc23178517235ae17609aa78db8bdda7e45fe3", + "engraphis/read_only_api.py": "478c7efa4e462e349d758143d3d44d18b9152656067c0171904bfe675a2dedbe", + "engraphis/redirector.py": "5ba964b81f09008c9369180cb49d247519000763274aabc5e046215b12fa641d", + "engraphis/routes/__init__.py": "f0d59080212cfa0d9b50877bca28e5832d1ae917899500d52c68b1c8821231af", + "engraphis/routes/memory.py": "9ca066e1762eeeaafd9790ed57d730efcaceb57742dbe0c830ad8764500ff9e5", + "engraphis/routes/v2_api.py": "462c6dd7ef1b43d8714f054f320294d0eb335ca70ce4341d06be159e7721be56", + "engraphis/routes/vault.py": "1a7eee7c1a7c7aa11091042756aa2739a3eb963647020f40584a6a38da1988af", + "engraphis/service.py": "f4a432e823db8dd3561fbe30656d9813e02a8f2eef3347c3e468dd872822e40d", + "engraphis/service_context.py": "3de9289f49a977cdc206285ac42a9953a104eb1d1b1bc8ea77febd75c7cd82ab", + "engraphis/static/__init__.py": "1fff4c4e2554e7f5fcf3eace269feba09827524917193a4dd65df95bae64ad1f", + "engraphis/stores/__init__.py": "48ee4326c8f28eecf46f558b7aea21adb779226a7038299017ab90196d5d84be", + "engraphis/stores/graph.py": "ebf603b54cf8450e7c9a7319bd05db2f39fcda8491f5969d6af3e8da61571bc2", + "engraphis/stores/ledger.py": "df5cbb30d977decc0a9c3a2365c9c115cfbb5fe951bae48446436661c80e5d30", + "engraphis/stores/vaults.py": "2c986129b9d1e7aab33e18a3b9a278eb5ad2236895587f69b5816a5d96796bc3", + "engraphis/stores/vectors.py": "45a1baca381fc647548cc89424eb36d5853f7562c275b191b35b99339a190b7d", + "engraphis/update_check.py": "ffbf5ef682fb15177915073ebc0b5eab4dee9092bbf0f58d36a8a54c6605ccf4", + "eval/__init__.py": "639f0c6d9d6aac8ff6dc605a34a0a301058905cc53bff4eaed5912247f0e7c56", + "eval/ablation.py": "16f159dee75d2f96cc42f230c2403fa19ea0bda7091c4823660da553463a194a", + "eval/adversarial_memory_security.py": "35dd8d981bcbad50e9815465be420b05b62a9dea28a78eb6cc1e09ec51c320c4", + "eval/agent_benchmarks.py": "8e88943e45a1b2083b366cbb0329326ef918b119a2366ed395c404c2b7ef98b0", + "eval/benchmark.py": "b71832affdf87d23bc1db7b522a669a7888571b3989b03a235c70f5da9952bd6", + "eval/benchmark_analysis.py": "1200791425d029695fee0b7da1f188a8962f337aa31eb4217d50e4503641d2b3", + "eval/benchmark_campaign.py": "033dbebfb47df5fcd3a6588b29ea4ab5fd44d4387ae45334379278ab30f3e57f", + "eval/campaign_adapters.py": "3e1db1183358331143d2d560b3a122d5e6b817f02c597ac7b4cc2ed8b7ce04e8", + "eval/campaign_api.py": "323be4e1d9520e8047ab54c9cbb158013785402b06ee4e93e55227971c54bf95", + "eval/campaign_candidate.py": "88541840d16c3b7368586ed0ee3566054214b1f594826b746b28d0e407cb812a", + "eval/campaign_continuation.py": "7f77a20ac8f85cd96729bba0cf45872c98a3421af005bd735434b7b416dd7e1f", + "eval/campaign_ledger.py": "ba3079ba541cb1f70929a76f147d2bc5e8b264103a1cd7faad97685e0ecc7790", + "eval/campaign_oracle.py": "9ea5479786efd9caa2b59e612e891a93dd34b500a2b36aa20bcf749fb6d4d48a", + "eval/campaign_storage.py": "ebcd1f4aeceaa5ff9304494c64dd31fe31637ceb107d8cc39fa292e3b14055d4", + "eval/capacity_matrix.py": "8c25bd97754c1d7a8468a687042a3c87cc9c0299cf01c19b80dacfec2be4e552", + "eval/chunking_eval.py": "a16544353940c0a8c40cea3b9932d3399b35ea5994b809b78f5dbe4a952c467f", + "eval/code_agent_ab.py": "d98bba6b77700ff6bf86ff0d2cf518a67e5a2676ef51bbddabbb2ff26e1f3aaa", + "eval/code_arm.py": "d211166fce1b8a4173848e1617873b7e84aadeff43746effe7483c54bbdb6f1d", + "eval/codex_oauth.py": "f235fca482de4201d1850bbfb583765ac5f1a057b4ad7e2d504d036bfc03f391", + "eval/coding_acceptance.py": "4b39cbcb60d7fba503cca597399cf9d04ad9567a43cdb950015ccf552e6a2773", + "eval/coding_corpus.py": "7b5205e8544578fe99d9d9cf6cfc34e40e5d1d049238d502ff7bc6f3c14948f5", + "eval/consolidation_ranking.py": "917b578d4e0bcb929bf1a1a37611acf7716a520c076abf4ade0d8d12a1c455de", + "eval/context_economy.py": "709ac7cc866855f96d7717ab2bea12e8b0d3a140d15fb978a2a929ad085931f2", + "eval/context_efficiency_guardrails.py": "22afd1a6fe17219e74701dc587ec35f569a5bf22bea270944525f51746723f14", + "eval/datasets/codemem.jsonl": "341313023c22850a2e14f02742b571ad1deca824f886a1654a59541304c01f3c", + "eval/datasets/coding_memory_v1/oracles/atlas-green--code_relationships.py": "969aa71235e0ae3cb764b9ee12b789184cedb5fea6fcc0d86e5dcb2c0498fa56", + "eval/datasets/coding_memory_v1/oracles/atlas-green--condition_values.py": "648702646fb62990168e98fba0dc6a27be128876548fa7035a59d71aee9da02d", + "eval/datasets/coding_memory_v1/oracles/atlas-green--corrections.py": "c7132c7fe39b7afbdf11065037369c0f3c38a22e69d8742e8afddfd5785d2e61", + "eval/datasets/coding_memory_v1/oracles/atlas-green--long_documents.py": "65dffb25944cb612872655f12db16af73f7a095e7abcf6a7cc5d31dc91f083b1", + "eval/datasets/coding_memory_v1/oracles/atlas-green--multilingual.py": "ff49874a691bfc30c082695ee6d3c89b9f2e1e5eecf086026110770d8e1b0452", + "eval/datasets/coding_memory_v1/oracles/atlas-green--paraphrases.py": "dbdcf5d7abf136aa5816df8465bfde842284aef5ef002e955cee05ae69d574ca", + "eval/datasets/coding_memory_v1/oracles/atlas-green--poisoning.py": "58b8ca4a37f55c4d112640414dc82620e32e643ab0aa539260884a5ecc50740f", + "eval/datasets/coding_memory_v1/oracles/atlas-green--scope_boundaries.py": "c1e92b10b40004c95abd3dfbada1a38fe0ce56300ecddca8360b51a1e6671458", + "eval/datasets/coding_memory_v1/oracles/atlas-green--temporal_history.py": "a64e4e5a29ce1040cf53e94571491325347f443d72a93f05bcf87c2878665952", + "eval/datasets/coding_memory_v1/oracles/atlas-green--unsupported_questions.py": "c8b415052ea7320e1df3c5475a676d4dd24602c45dbf1868fba6880846e029ee", + "eval/datasets/coding_memory_v1/oracles/atlas-north--code_relationships.py": "bb4305d15a81b4b4ef80b374acb59fee6fc32b213b101dd4b173ac4144d8570d", + "eval/datasets/coding_memory_v1/oracles/atlas-north--condition_values.py": "4bc5b1aff48ef6ad5ece1bd915f6949f72d365bb4815c4d82515b438bc644112", + "eval/datasets/coding_memory_v1/oracles/atlas-north--corrections.py": "7851502a9bc842f840bee1001e42140ee1df08be2730b563b1b197285d14742c", + "eval/datasets/coding_memory_v1/oracles/atlas-north--long_documents.py": "6204123eded20fc0c7aa59b312f8b5bd29aa88a720f7318e32b3c6a17651b57b", + "eval/datasets/coding_memory_v1/oracles/atlas-north--multilingual.py": "5c989b34fdad1e076b61ffb1f7a1371dc635f44a537abc743b765d5f6009004e", + "eval/datasets/coding_memory_v1/oracles/atlas-north--paraphrases.py": "fd189285d0cde9991dea461af9cd18b4203656a69da1298494fccf972ed8a58b", + "eval/datasets/coding_memory_v1/oracles/atlas-north--poisoning.py": "66305d69419736bdbb11a896543933604d898cd5394d94eedda8b92e632a09ac", + "eval/datasets/coding_memory_v1/oracles/atlas-north--scope_boundaries.py": "d9fbbe3924b5178f32f1937485af3b8db242a0e3a08c7bf41a3d2a57f34a8879", + "eval/datasets/coding_memory_v1/oracles/atlas-north--temporal_history.py": "7e7be0d6e558a71d3baad14a8812a6c998ad25e3f52c997a301333e0711f820e", + "eval/datasets/coding_memory_v1/oracles/atlas-north--unsupported_questions.py": "91f574dd678bf244a1fa6f9708e945259bb8ff8e2eb213426be5c3b187be3c3b", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--code_relationships.py": "d0998e13db38d1fae9be6d254431b4fe592f89eb03ee25705f4c70985cd48c89", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--condition_values.py": "f97610aee190dca03b01aa48ffe4aa3f75e099a59fdd867db98f743f4a376729", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--corrections.py": "b7c31f7a9a1f978321ca5445997e0d284deea770c85f6d81a80204e548ad53ea", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--long_documents.py": "20ec2b41cd3b6691164bce096dff16f7bf5be674f922cb637120c0347e60d0b6", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--multilingual.py": "131e7dc978d3f2f7df395b6f1b4a9150051c0ca93c6dc5c386e58d46c8fdc573", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--paraphrases.py": "1e7d9da36e6634ca2d2b11dc97468bd671df133345dcb76859e1dc3cb77dc664", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--poisoning.py": "48460691ac0086c47102eec85387ac03bee9190ac83e3dafc4babc79b71d9bd7", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--scope_boundaries.py": "b96207e9b1fd09f49218102a5b88fde5b9656604df4cfd314e86426e3b6d6075", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--temporal_history.py": "7a196ff9a1db05b0c8f3a83086ae44cdc580dc58e4acc217d1db2575d9711b88", + "eval/datasets/coding_memory_v1/oracles/atlas-violet--unsupported_questions.py": "724564b5b75b9425b6889faae6d5e650bd5c940016c2b96054e5359d9d8b0887", + "eval/datasets/coding_memory_v1/oracles/atlas-west--code_relationships.py": "14295f735bcac4d4ae738eedfbd8fad4593afc914bc835233ec35e2263d43613", + "eval/datasets/coding_memory_v1/oracles/atlas-west--condition_values.py": "8ac70d9ca0c9122567bb9986d2cc5798104966df20e262b8087920e8fb5d82a2", + "eval/datasets/coding_memory_v1/oracles/atlas-west--corrections.py": "89d6a1a7781465c311830b7e769d2a76df41ab5c703ff662b3df1cc4c9a01668", + "eval/datasets/coding_memory_v1/oracles/atlas-west--long_documents.py": "38e67a26f69707089f852657a810ef1fca0395beb5511fff67c30f21ed63df57", + "eval/datasets/coding_memory_v1/oracles/atlas-west--multilingual.py": "bccf6579681d45940542ccb1f936ea1b225f4cf551a3ef31f3f8b1a026856e51", + "eval/datasets/coding_memory_v1/oracles/atlas-west--paraphrases.py": "89c8ce16c682553b27652e176b67a06f3228ae6209fbc8a7264260680f5651a9", + "eval/datasets/coding_memory_v1/oracles/atlas-west--poisoning.py": "fe9beb97d27eacdd87056fa1c9b04c62e2facae9354e765b912de60b35021621", + "eval/datasets/coding_memory_v1/oracles/atlas-west--scope_boundaries.py": "2eb3b1130c54644fcbade52da8f50cad1376a07b51856a962522913cbcbb7649", + "eval/datasets/coding_memory_v1/oracles/atlas-west--temporal_history.py": "116ee20af475da329cb535b0a02bd19ea1e7518b5b780ec12ba396ec896cc116", + "eval/datasets/coding_memory_v1/oracles/atlas-west--unsupported_questions.py": "080969e026215b9fe4ede6d0b4d02e9143e1293bea5d7383681e478426ddf121", + "eval/datasets/coding_memory_v1/oracles/borealis-green--code_relationships.py": "3965671e5e2432c8236c250f7982d4f478d8f1fe1268ffe57c555cf5fbbcd414", + "eval/datasets/coding_memory_v1/oracles/borealis-green--condition_values.py": "92994990327445258984ee6bf1d95f47359ff4b194e1f36e193980a523e6c5e0", + "eval/datasets/coding_memory_v1/oracles/borealis-green--corrections.py": "c5b473f962deacd6d59ea9967626e029888abbdb48f320704bcb6de876ce3f46", + "eval/datasets/coding_memory_v1/oracles/borealis-green--long_documents.py": "54aee2c1031c17b5678db2b0955eda8fafe5990aa9e52f02936584a9b75b9ab1", + "eval/datasets/coding_memory_v1/oracles/borealis-green--multilingual.py": "c458ac5c63e440e3d9486d096278759920d636f73e1901258679251afd86eb60", + "eval/datasets/coding_memory_v1/oracles/borealis-green--paraphrases.py": "250d027660f268cef1caa745cf3155de6e2f4adfaed02ba675eaa83224ffeee8", + "eval/datasets/coding_memory_v1/oracles/borealis-green--poisoning.py": "978f16eb5033c602ec09ed57ba0667edec9ac3e26accc1ce1456bc893ac9e6a7", + "eval/datasets/coding_memory_v1/oracles/borealis-green--scope_boundaries.py": "4b73a2e5dc5ac815c14e08a34acefd2ad426282cfae379ca2fd3060a3784c9b7", + "eval/datasets/coding_memory_v1/oracles/borealis-green--temporal_history.py": "6636d511bbcca51ca85c4eb0ec5ae956b3baaa5f9af89458ceb86329115743bd", + "eval/datasets/coding_memory_v1/oracles/borealis-green--unsupported_questions.py": "5cd4f70627bc60572c720a797bff122ef9bf219f32d265f694affa33b1f9f4fe", + "eval/datasets/coding_memory_v1/oracles/borealis-north--code_relationships.py": "769903aeac3ce360d845ddf1a6e89192737dbb460e1f2e251230b74d40763dd6", + "eval/datasets/coding_memory_v1/oracles/borealis-north--condition_values.py": "f6e69f525fd240a09152989b9d8a529e9abe076e7fc2c98d10bd84f9c6da2b56", + "eval/datasets/coding_memory_v1/oracles/borealis-north--corrections.py": "987ece1c7507a69a973c1e051b27c1d3b104ecda79df48959e49a62da31aa137", + "eval/datasets/coding_memory_v1/oracles/borealis-north--long_documents.py": "011134140cc486a3c089b407adbfbf5b958e1f8c1bffd572c29b8cd5d7f07f7a", + "eval/datasets/coding_memory_v1/oracles/borealis-north--multilingual.py": "1d29536a4a5836f64f56adcd1a965db85bc3371a12ad5bbb5644520c9d9606c8", + "eval/datasets/coding_memory_v1/oracles/borealis-north--paraphrases.py": "09a5fad90c24f75f2f448854c8778aab896c67854c97f884c2d689ead808583e", + "eval/datasets/coding_memory_v1/oracles/borealis-north--poisoning.py": "c1bec2ed9b31103e39db9dbab2be8db401cb0874bb3ce8573a0841a054e28a46", + "eval/datasets/coding_memory_v1/oracles/borealis-north--scope_boundaries.py": "faf51b368d7ba9406442edde29e290c931e67bff2799783dfd27f2226020e15b", + "eval/datasets/coding_memory_v1/oracles/borealis-north--temporal_history.py": "da52f43530149a4a293065674623fa49f83b37cd84b9871044b3bf32762fd8c1", + "eval/datasets/coding_memory_v1/oracles/borealis-north--unsupported_questions.py": "be0abc698692c9fe82f40f76a4694c472882a53843894dddd631ab1ab77aae45", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--code_relationships.py": "9497b7d2373bb9f858e994f7919cff72e7e3c4c4eaa1efda287524b1a182d1f4", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--condition_values.py": "aea13e4b7721019e7bc1f54878bb25b9f27a8a5044fe444e86ba5460deb13e09", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--corrections.py": "bfd446d431afb6c05ad9d8775b90887947b85981037756e05262f3b101d41e7b", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--long_documents.py": "eae829a374c0cd6237fac4ec52fec0d98d7340ec0b307b9b0b0c90c116d4b472", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--multilingual.py": "e821f91999b5f12cff53fdaa956af4e0bb2f929497f95b4e2ceb57a6621e58f7", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--paraphrases.py": "1a99c15898831a1c9aacab951868101d57e59e91ed90a2777015a26dbcc3043c", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--poisoning.py": "411877fc07b053201a63aa4e45cb499344abccb0a562c236d53a672078b1eb63", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--scope_boundaries.py": "b198dfb636910953ba372edf1cd6db382d4e5725bf7f820171be8c63f9fedf7d", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--temporal_history.py": "3de5efd6ef661572191c6aa426fb10caf9f4bed4aa495b6298815aab3acdce22", + "eval/datasets/coding_memory_v1/oracles/borealis-violet--unsupported_questions.py": "baf5dd85e86a97dedb0b395b5c574ea5ffcdad7a46bbbe9787cfdc33d6c7a06c", + "eval/datasets/coding_memory_v1/oracles/borealis-west--code_relationships.py": "48e53b3a561f91fdf4eedb2d63aad65eece8ae80188bb3d5bdcaf184fb0b6945", + "eval/datasets/coding_memory_v1/oracles/borealis-west--condition_values.py": "2dd7a524bd2a2aa2fbf212246880d4e4ecf03f503b70f25179f3ab6bab28a07a", + "eval/datasets/coding_memory_v1/oracles/borealis-west--corrections.py": "7d3080e05544e10243040fe0a969a8c524e4262f8f5c38abdc3d8e06c67fdabd", + "eval/datasets/coding_memory_v1/oracles/borealis-west--long_documents.py": "fc42e935879e7a81f6eef9fe72f8683244501ae10654008a7863d397ee73058c", + "eval/datasets/coding_memory_v1/oracles/borealis-west--multilingual.py": "19181c442ce5f6941485190ecf050e455914dae988f86efa42665cdef0d835a2", + "eval/datasets/coding_memory_v1/oracles/borealis-west--paraphrases.py": "f7a5cb842aae3d2bc8d62ea40325b332a8bd4f013b76bac8bf154fedb3dbcbe1", + "eval/datasets/coding_memory_v1/oracles/borealis-west--poisoning.py": "285a2526a6be4e556a564899e9a376ae1286fef6f0ad22eb9289fba1905a2b68", + "eval/datasets/coding_memory_v1/oracles/borealis-west--scope_boundaries.py": "fb994823743ff5b4853b142c17779778bdb11296958e033e5c6be598833a4cf5", + "eval/datasets/coding_memory_v1/oracles/borealis-west--temporal_history.py": "f30a74c85ad654a7450e20d07baceb3b4fb2a56b454d7b43b854bef3cdb2cee6", + "eval/datasets/coding_memory_v1/oracles/borealis-west--unsupported_questions.py": "01cbbd0b21a745720514aea6a21a98a3cccdbf0c0518bd5508019433b964776b", + "eval/datasets/coding_memory_v1/oracles/cinder-green--code_relationships.py": "8b08e7c65289515a9eee2ae5f4cd60ce94350e9a561225bef0e8f8815ea28f79", + "eval/datasets/coding_memory_v1/oracles/cinder-green--condition_values.py": "92e7d75d38f5dbcc685075e947005a9ae5a410207e7f17aa770b034b8690cbe5", + "eval/datasets/coding_memory_v1/oracles/cinder-green--corrections.py": "424cbf0924a8cbc1a59a6d7bc82b5527e41f52c8ee9f558ea3e398ab4d913919", + "eval/datasets/coding_memory_v1/oracles/cinder-green--long_documents.py": "2a52df9cef3ee66357847322794a382c927b828c6ab1a2819a799ff55d53657b", + "eval/datasets/coding_memory_v1/oracles/cinder-green--multilingual.py": "e47e791dd15b045a6a958ca114b1359ca8537e160274b015214746a2564363c6", + "eval/datasets/coding_memory_v1/oracles/cinder-green--paraphrases.py": "9e92a79ec00231cb381ce58e389dfdb2520a17ecf1e9c2e39e79501449e9ddd0", + "eval/datasets/coding_memory_v1/oracles/cinder-green--poisoning.py": "33213e6f194332969da88f4748c8f26ef93219b6c4bc710a142082f8147ccfcc", + "eval/datasets/coding_memory_v1/oracles/cinder-green--scope_boundaries.py": "8d2489da1fba137981e45e42c843f9cbd17e851c82ddcdc5904b4a86af0d9692", + "eval/datasets/coding_memory_v1/oracles/cinder-green--temporal_history.py": "29403fd08f0de7da3dfcc6c181e1f448a5c5264e95ff3924f3066508d85468b1", + "eval/datasets/coding_memory_v1/oracles/cinder-green--unsupported_questions.py": "ba21014188f9ae849e0bb79babe163194703c4cf01f0600f4c9f71058e7fed56", + "eval/datasets/coding_memory_v1/oracles/cinder-north--code_relationships.py": "46280659557bd08afff185352438bb63dfd308dabf5c7c1862c2f442bd171d48", + "eval/datasets/coding_memory_v1/oracles/cinder-north--condition_values.py": "6c62d673b1b6cb4b5c89ff2af4ecf57c2c9d73074a01ab92b8fde57815cab783", + "eval/datasets/coding_memory_v1/oracles/cinder-north--corrections.py": "8245b196072389517fad6d76ed7709d7e12eb52e08732e4ddcfc40fc1d986879", + "eval/datasets/coding_memory_v1/oracles/cinder-north--long_documents.py": "6790c7b89d80ab909d9e1a31b59bfa5cd199cddabed2e159daf7c5c049ff9a43", + "eval/datasets/coding_memory_v1/oracles/cinder-north--multilingual.py": "9532bbd14506cac0edcc09f474e7c96a402903de947e0e76f5c8aab4bff37646", + "eval/datasets/coding_memory_v1/oracles/cinder-north--paraphrases.py": "cc904350ae7d40e26d91d482ac15d87ec8022d4c82905021dd38003c5fa2d067", + "eval/datasets/coding_memory_v1/oracles/cinder-north--poisoning.py": "e8e451c49c98744b4a484141f86d337db4ec5326264b1500893a20c3451c5ae4", + "eval/datasets/coding_memory_v1/oracles/cinder-north--scope_boundaries.py": "50d42fc48aa70d55b6a3b39c9bf17437726f37e1e5ae9870d415fccad92c7bae", + "eval/datasets/coding_memory_v1/oracles/cinder-north--temporal_history.py": "14b1336fbaed11cad455f588bbe078effc31f5f357f670e1f189ec3d78b172f5", + "eval/datasets/coding_memory_v1/oracles/cinder-north--unsupported_questions.py": "0a2f740700a990d22b4c3051aa3e4260e3df4ce65931d419052f74e05c750a67", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--code_relationships.py": "0e29053f11fe618c01fa90a274688f18ba8a7f3cfea56757a96d5a5bc95b61a8", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--condition_values.py": "ca57ac4e22e9b4739217020d7792ac642326935b8db9305ead2ec5bea2a6e456", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--corrections.py": "9edbfaa58a99776e835c325e50a77df5b4b302b3ddd0932dc2a349a77b4100b5", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--long_documents.py": "4b3fbe7380201285ac4b4a1ec4d8238c5663b97af01cbdc8ffff8fdf2a9467c5", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--multilingual.py": "b0bba5ecf669f3368ef29373b755dc1296502665584574c771ebf4664debb5ce", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--paraphrases.py": "e146f7aa1ffabf3b313251b59c59e1836c0cd17901d07fc6ef5334eaccf8bf36", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--poisoning.py": "692ee07e750fb08652a1ca6bc6dcbbef77219701585f1336c6037dae5a7c5767", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--scope_boundaries.py": "8ca0b911d38a72e4ecce6871ba3d0c7dbab433b0541cb64b116f7a88d521554c", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--temporal_history.py": "6ba5400c5435af64c80a7366fdf49fea132f1011f9d84f9d0cc5f2f2f30a67ac", + "eval/datasets/coding_memory_v1/oracles/cinder-violet--unsupported_questions.py": "1d3324db0a085be30ce3dff1a2ec3866b8592a5ef6b43f39211a2e3a59fc1cd6", + "eval/datasets/coding_memory_v1/oracles/cinder-west--code_relationships.py": "0149333a3fb0f8d60939bf927921305eae447f7511c74f042ff3c880eadff825", + "eval/datasets/coding_memory_v1/oracles/cinder-west--condition_values.py": "3158243f9f7f0b92aaee4164379159b46143acd11fc3d41a708630c1d33ea22c", + "eval/datasets/coding_memory_v1/oracles/cinder-west--corrections.py": "a9fbb89aa629a23e7a1fddf4fca09da40365bbc7a9a6192d805e498fe17f93ff", + "eval/datasets/coding_memory_v1/oracles/cinder-west--long_documents.py": "3c702c653a8ddff84eb5c24a77567aa63d2069c48ed174bcf29065c0ddb30ec6", + "eval/datasets/coding_memory_v1/oracles/cinder-west--multilingual.py": "b8f5f7e00e6edac3f8e0b7e8bc09ea6c7d661370fa2226eb1795306ce3c634a7", + "eval/datasets/coding_memory_v1/oracles/cinder-west--paraphrases.py": "10361be87e9d9cb98d2ff1bbc18252178d475a5c5f2efb77bc81dea05109df05", + "eval/datasets/coding_memory_v1/oracles/cinder-west--poisoning.py": "889a5a43604f7d1677ee0f28683abfc48a8a597b052f558c1b4d19c5f33444ec", + "eval/datasets/coding_memory_v1/oracles/cinder-west--scope_boundaries.py": "b84e78812baaa92422226c06d78cfe7c1be260e2bfcf9c5bb1ed73f2f3cc136b", + "eval/datasets/coding_memory_v1/oracles/cinder-west--temporal_history.py": "9d99a981fca0c801a0a52e2af24cd4262dad72f4f34cd84894784285d241bd05", + "eval/datasets/coding_memory_v1/oracles/cinder-west--unsupported_questions.py": "43b8f854961d9cca5cb83b53b9e07d22388d0c7db406dc8e4f8cfdaa5f89fffd", + "eval/datasets/coding_memory_v1/oracles/delta-green--code_relationships.py": "97a08f2bf9b1a11f95a062f7df27d69db3e9dbc421837347acfe62111af02885", + "eval/datasets/coding_memory_v1/oracles/delta-green--condition_values.py": "5e53695288b84a526c61b748b170f6d6219f94a942edc0887ea6157535ad8d52", + "eval/datasets/coding_memory_v1/oracles/delta-green--corrections.py": "c6d16232ebe6bdb0b5c94db1ba4f108257b47ee8f8b96cda5f8e17212022b542", + "eval/datasets/coding_memory_v1/oracles/delta-green--long_documents.py": "bedb6f495437beada344ce6cdd92954108aa4c639622fd1cc294d2a1139b26d9", + "eval/datasets/coding_memory_v1/oracles/delta-green--multilingual.py": "dbf84f4e8c016c5e60fdac5fd3ac2faf79fb0534f5c2ac534491acc31d5f3530", + "eval/datasets/coding_memory_v1/oracles/delta-green--paraphrases.py": "ea8f00c86e8fbc2ab10a448f48019146181d0e55438def6f47f41f0ef3181be4", + "eval/datasets/coding_memory_v1/oracles/delta-green--poisoning.py": "4eb4e289e944e7762c39ce85a8fc2582e310de6a86cbc1cc36b44112653140ea", + "eval/datasets/coding_memory_v1/oracles/delta-green--scope_boundaries.py": "2fe9b96deecfb488430abd9657adc6b3ea092b18f32c41c1669d29175ec6e860", + "eval/datasets/coding_memory_v1/oracles/delta-green--temporal_history.py": "531cbb953400584356cb6525bc1a2e559f7ec3867d845c736d8026a046b91e7d", + "eval/datasets/coding_memory_v1/oracles/delta-green--unsupported_questions.py": "edebf25c2168411fdefae1307a71d829224dba3dbc7ea000054416e52123d5a3", + "eval/datasets/coding_memory_v1/oracles/delta-north--code_relationships.py": "9357ad0d5f6a1cc787f980d0d506b07808538179f71ae3e8ce1c3eb655d24e35", + "eval/datasets/coding_memory_v1/oracles/delta-north--condition_values.py": "4199f7409696a260b1531f4ec7cfd15555b4d58d06b0b8a88b1c7621fe7d328b", + "eval/datasets/coding_memory_v1/oracles/delta-north--corrections.py": "b8bd58db7c5ae93d33e6223cc34f0e09d58831361ac013ec08277fe2c102ac17", + "eval/datasets/coding_memory_v1/oracles/delta-north--long_documents.py": "0f08b744446b1c52e8fe86626ab05c171377f17c46e604cb2ea647ed5d0daddd", + "eval/datasets/coding_memory_v1/oracles/delta-north--multilingual.py": "ada9eb404190d05637cea59e3f8ed541aea2c9e0f9be0d166885d6789351f7a3", + "eval/datasets/coding_memory_v1/oracles/delta-north--paraphrases.py": "04ec2e8308f47f46e38c89f4a2fe44deef4a0b20a1c8517395ad00983bf7361c", + "eval/datasets/coding_memory_v1/oracles/delta-north--poisoning.py": "d72ea7b98db47e4de03884e2786206dc3da5cc08e5203c86b1d38a2d48c2fd47", + "eval/datasets/coding_memory_v1/oracles/delta-north--scope_boundaries.py": "cedb3234b22a40a7b889f50cfb5751fbba31e46f541b19feb2c4759c8007bac3", + "eval/datasets/coding_memory_v1/oracles/delta-north--temporal_history.py": "2f7c1bd4ebc051eee411d11251aff2ee5fee602a0800410bc0dbd2e86e99628b", + "eval/datasets/coding_memory_v1/oracles/delta-north--unsupported_questions.py": "f1f0b9858c806956122144c6e769f3baf9f87f07ff92eec98eb58d8019e6fba2", + "eval/datasets/coding_memory_v1/oracles/delta-violet--code_relationships.py": "79579639a60d96365781e853d1319a2bf85d9a2b4d8639410da7e3c1efc364b2", + "eval/datasets/coding_memory_v1/oracles/delta-violet--condition_values.py": "27defbfe7c237f971ae29e8f000f098eae32a11f3afad06f2bf3b27af432d902", + "eval/datasets/coding_memory_v1/oracles/delta-violet--corrections.py": "b3b685a5c33d1086e8ab4258d26c5ec9aa7396b8250a250351f0a6057fcf2075", + "eval/datasets/coding_memory_v1/oracles/delta-violet--long_documents.py": "4a5ce75372c8762bcd7986f8ceef454800c3636124ffd9de3a8315d210ae85f1", + "eval/datasets/coding_memory_v1/oracles/delta-violet--multilingual.py": "7ef19f25e97a24522b79a9d7e7f3d2d9e395a1fc2c60564e2f66081533bdacb5", + "eval/datasets/coding_memory_v1/oracles/delta-violet--paraphrases.py": "b6e00019bead0a5bc0facdac886b48fccfd3cccaa254267fda780e933d57904f", + "eval/datasets/coding_memory_v1/oracles/delta-violet--poisoning.py": "3e3e1e1fe04ae3e0c615ca21e794fead5d06f887d6fe7347fb9122684c8b8017", + "eval/datasets/coding_memory_v1/oracles/delta-violet--scope_boundaries.py": "336ae66910d533344712a2d87126f53df48d43eae9e7d17797938c3f9c560204", + "eval/datasets/coding_memory_v1/oracles/delta-violet--temporal_history.py": "a81cb5d2a5eb521dfd161ffbd21bd366777b1239f2d3412acc77b0a9c7c7d82b", + "eval/datasets/coding_memory_v1/oracles/delta-violet--unsupported_questions.py": "3341895608073374b98e67a88489abb9020fed5bb692231a0a61e15c8de0ddf1", + "eval/datasets/coding_memory_v1/oracles/delta-west--code_relationships.py": "e8f17f3838cb74e1ade0cfd6edf540dc086507b51ca67c2020ed490d75dbffe5", + "eval/datasets/coding_memory_v1/oracles/delta-west--condition_values.py": "a935a6e7a4081737eef76a93d54b34dc885b4699e02d797f1be78286dd642a2b", + "eval/datasets/coding_memory_v1/oracles/delta-west--corrections.py": "ee407e5aa8d57254bb7029d537bb2b59a297958cdff9bf271de3787b6448241e", + "eval/datasets/coding_memory_v1/oracles/delta-west--long_documents.py": "55fa08b75982d0547f0c2f13cdc6522d0fbd33efb4af9ae8c6ada11b2d46f321", + "eval/datasets/coding_memory_v1/oracles/delta-west--multilingual.py": "ce8a3e45ad8417ca1798579a22e12eeb7d170a11ebfb865746388eb0e7dca848", + "eval/datasets/coding_memory_v1/oracles/delta-west--paraphrases.py": "e6458588e20e76596e24be6cc838bd4b0d5a47b05b7bf0bb6017b0cecd0d6a68", + "eval/datasets/coding_memory_v1/oracles/delta-west--poisoning.py": "4c70955a2650d13eba24d6e933f765e0bd40da0a1cd8b920416989a76bdaf2d7", + "eval/datasets/coding_memory_v1/oracles/delta-west--scope_boundaries.py": "cfe9120c49224b6f9ed86ea7d85e496000bef0648c92c7fd474c540e892863cd", + "eval/datasets/coding_memory_v1/oracles/delta-west--temporal_history.py": "3527edc8a85c71f889ff0b2ac8bf9ab005ad5bb6c95bcd917dedae44aa3efc72", + "eval/datasets/coding_memory_v1/oracles/delta-west--unsupported_questions.py": "8c40923b8d4bd6bfc78ff6548ef9a840319b7ee4ea967554f26a8d24a4349935", + "eval/datasets/coding_memory_v1/oracles/ember-green--code_relationships.py": "cc7f82ad85363ee35ccd75744159bf4c4f07965e396bff07989cca7a55071679", + "eval/datasets/coding_memory_v1/oracles/ember-green--condition_values.py": "3fa888c125812f1488de9e021358cd531a9d50bb7d4b6173be3656268474b635", + "eval/datasets/coding_memory_v1/oracles/ember-green--corrections.py": "7cc639d4bd7b287ed0e3f2188491d6e4866e021b21874b34b365c1eb2b2d6d68", + "eval/datasets/coding_memory_v1/oracles/ember-green--long_documents.py": "972e7879a30ab26f86f09bce6dc9853c7f96124d0bbe5062111607496bcc9a95", + "eval/datasets/coding_memory_v1/oracles/ember-green--multilingual.py": "0f0585bd3264de3d5fed5928b54ff4fb87f71b2edb1d35c83a534251121124b4", + "eval/datasets/coding_memory_v1/oracles/ember-green--paraphrases.py": "a3c15541bb87cc862d1c9764cfdc1ffebbc831876b55167470ecb3b66519740a", + "eval/datasets/coding_memory_v1/oracles/ember-green--poisoning.py": "ace5b7b12c4ca068739af4bc5e215af1af7fe153399ff17ffb7a67b8b9aa1f94", + "eval/datasets/coding_memory_v1/oracles/ember-green--scope_boundaries.py": "daf132e130c5743d67129c1564f329fbb2670d1242648920b646a83bc2236e4b", + "eval/datasets/coding_memory_v1/oracles/ember-green--temporal_history.py": "1438f09f0ae41b1f6b42930518203a36c7de31a870b684dbb7f16e71a37c784b", + "eval/datasets/coding_memory_v1/oracles/ember-green--unsupported_questions.py": "da27396afbc1529c61972f739cefb26dcaf97bea529c084518a95080eac4fe78", + "eval/datasets/coding_memory_v1/oracles/ember-north--code_relationships.py": "debaf885151c3f47660053ff50e015338b8aa9bca089d7e901ee8bc34f7085e6", + "eval/datasets/coding_memory_v1/oracles/ember-north--condition_values.py": "109a7ee90c04da83f57a94a9a724dcf98a3e96405c7a456686c198ff5d85a69c", + "eval/datasets/coding_memory_v1/oracles/ember-north--corrections.py": "c246ba011c68d3ad0a4ee1d14231304a7f17f28027f1764e60da93062cf1cd3c", + "eval/datasets/coding_memory_v1/oracles/ember-north--long_documents.py": "176200cf2a011c7231ed4cadacbe86b1eeb695fbbee282bf4c4a96430d98fe84", + "eval/datasets/coding_memory_v1/oracles/ember-north--multilingual.py": "e850a910515e71a27edff26f7bedf39d8cb75c726e48357a2c41d4da051be6e9", + "eval/datasets/coding_memory_v1/oracles/ember-north--paraphrases.py": "0a6f03310e44abcf4ca61d69e22edbd21ce47a17629adf4f5079f7565534b469", + "eval/datasets/coding_memory_v1/oracles/ember-north--poisoning.py": "a81ee758fc57bcc58bf0bf2a38dae8b6eb4aeb14949063206e32cdff6f301dfb", + "eval/datasets/coding_memory_v1/oracles/ember-north--scope_boundaries.py": "fe42df24612c66b175f1c98794ed9f13a1a7819418e4aeb5ff98531515221de1", + "eval/datasets/coding_memory_v1/oracles/ember-north--temporal_history.py": "410f9770097228724a5af1357003ea23caefe1226275d030d46efa535c42d306", + "eval/datasets/coding_memory_v1/oracles/ember-north--unsupported_questions.py": "eebe43443938693a74081661d2ff3733b39c6dac5c7ecd88c4dbab5ad18689ee", + "eval/datasets/coding_memory_v1/oracles/ember-violet--code_relationships.py": "c00a30669fbd1c740b44e17c7c7369582c28a16da3aeec9d27420b8aef7c553a", + "eval/datasets/coding_memory_v1/oracles/ember-violet--condition_values.py": "1645e7c56c870f4367c263ccc9041181b1d187a698252203ca651235db59e684", + "eval/datasets/coding_memory_v1/oracles/ember-violet--corrections.py": "735f1d125d2ade6f3eabbf1082a408fe6afd4328cd954833e23df275df498e37", + "eval/datasets/coding_memory_v1/oracles/ember-violet--long_documents.py": "e1ded8e9641639443a6b14da18b2e0cbcddc6fc6045bf0b57404b59b59327dce", + "eval/datasets/coding_memory_v1/oracles/ember-violet--multilingual.py": "fbf04cda8e1231c7ac4c64d4c1856f7bfafcb6f1d016de0b1bdbbaaafcfe03b6", + "eval/datasets/coding_memory_v1/oracles/ember-violet--paraphrases.py": "2fd495def306778073df30b20dbc7a1500c06c90638acf6d9b9280db78dec53a", + "eval/datasets/coding_memory_v1/oracles/ember-violet--poisoning.py": "335674216a418254a01a7f5505344af20ad499a5fe41cb9c8654a0705574a18e", + "eval/datasets/coding_memory_v1/oracles/ember-violet--scope_boundaries.py": "078d60287888903e90343d144d8f54e08a3751ffcba82fd827ca0c9f8ee75fe7", + "eval/datasets/coding_memory_v1/oracles/ember-violet--temporal_history.py": "f7ad47f3253fe922977cbd0b63679abf00559f1b2dc895246e09840652b2c41b", + "eval/datasets/coding_memory_v1/oracles/ember-violet--unsupported_questions.py": "b91545082a31952351bc4c08d6cc09132432126dfd772952c818396bd7a68af1", + "eval/datasets/coding_memory_v1/oracles/ember-west--code_relationships.py": "74fb3e0d99a5bd4f54b8b57f6040ce7978a81bffa1ae70f9069fb2322e6ea581", + "eval/datasets/coding_memory_v1/oracles/ember-west--condition_values.py": "26d605b42d18cf029dfdc29124eea898e0dfa5642c43c1c775007fb443ba6cda", + "eval/datasets/coding_memory_v1/oracles/ember-west--corrections.py": "65c2ef8544b42ecb6b3ea9ce8bc95ec5f0ea8bb07c2b74d52e7880d99962f563", + "eval/datasets/coding_memory_v1/oracles/ember-west--long_documents.py": "4c8e54c7ec930d6cc782fa1a0519a0f1e0e916e3929902e13199775cefdaf338", + "eval/datasets/coding_memory_v1/oracles/ember-west--multilingual.py": "356e407f47d39e7f53337ffc4f27c68935a2c831301dc3dd006b2939cc61decd", + "eval/datasets/coding_memory_v1/oracles/ember-west--paraphrases.py": "c91cf574cf8b5aadaa6c9331c3f8f4a8fec7ec33c227f20894f463d81d110a7d", + "eval/datasets/coding_memory_v1/oracles/ember-west--poisoning.py": "5f6b8740798a7a093994748ef4b11e0e1de0be562c7f09790e1f0ebb9ed21f7d", + "eval/datasets/coding_memory_v1/oracles/ember-west--scope_boundaries.py": "5e20c5154818cc29b7e71444054d67836eddc882f2f17be8aed78afe31a9037f", + "eval/datasets/coding_memory_v1/oracles/ember-west--temporal_history.py": "fc51d1a39693d1f7fb2db4bf1ffe865d4c44848c159188cdcfda54618c0d8fe2", + "eval/datasets/coding_memory_v1/oracles/ember-west--unsupported_questions.py": "079c421cae974bcccc2107dac74421dd5bc2d1b40f7d91cb79a8014af526d260", + "eval/datasets/coding_memory_v1/oracles/fjord-green--code_relationships.py": "b5e4a284bad1d7ff81d5be9cf1faaaaf5e718cb51624189482624721cb459f1b", + "eval/datasets/coding_memory_v1/oracles/fjord-green--condition_values.py": "63c152b8da5655ae5c701cbb51099340d5f5103fcc47840239098a7366d24151", + "eval/datasets/coding_memory_v1/oracles/fjord-green--corrections.py": "b7b33b8de153c205ff4fdaadd5538eae8a22fd4401e70db9f645dae205c62cba", + "eval/datasets/coding_memory_v1/oracles/fjord-green--long_documents.py": "624cf014bc795a24e2f1f4fa2f9dbf11d2ad1c38f81cbc58ab3948f96c48c8c4", + "eval/datasets/coding_memory_v1/oracles/fjord-green--multilingual.py": "f8d2483fd9266946c980764feba0eda27c95ea2ff11222a4dd4247fb537f1f00", + "eval/datasets/coding_memory_v1/oracles/fjord-green--paraphrases.py": "151be253b730d5c5b4bbcabcce3bd218477241ebf0f1bd5fe9dc8d70798feb6f", + "eval/datasets/coding_memory_v1/oracles/fjord-green--poisoning.py": "e7b0efca3813bbb1d4ea9cefa752b4318942b118bb61ee9dcbbf640daa66644d", + "eval/datasets/coding_memory_v1/oracles/fjord-green--scope_boundaries.py": "56177cfb45f32c548be83cc78197fcf5b53826f01cca69e471535476184ae8f9", + "eval/datasets/coding_memory_v1/oracles/fjord-green--temporal_history.py": "2c79d26467eb8443d743bb8823cc1f2a5d1e5da7c887754505baa130e37f1b2e", + "eval/datasets/coding_memory_v1/oracles/fjord-green--unsupported_questions.py": "fb5f222f4e0715379591f559036c0549f67a7b6e3491e880c6cf8d5b91fa9000", + "eval/datasets/coding_memory_v1/oracles/fjord-north--code_relationships.py": "07402078200db9ab15fc61a72e31d9df0938edfb1e9d5b7e532b422dccab15aa", + "eval/datasets/coding_memory_v1/oracles/fjord-north--condition_values.py": "8c1549b5a444c9cbf1304fcca13b79263bf8d9f0b4f41c16da539b3bf5da3bf4", + "eval/datasets/coding_memory_v1/oracles/fjord-north--corrections.py": "a2634af73134c025de1b6239756ee2051e505e0a535221947babef5ba161fa13", + "eval/datasets/coding_memory_v1/oracles/fjord-north--long_documents.py": "4f80f3673afd1320d16a6bd45fefc85522d6f4bb889ff96f6e99410580feec09", + "eval/datasets/coding_memory_v1/oracles/fjord-north--multilingual.py": "3a9c953fbe76eb7374992a6a81d35b8dc0a635c02b90f328af140bbda2acc180", + "eval/datasets/coding_memory_v1/oracles/fjord-north--paraphrases.py": "ec2d9d7bbd678be73946cba8177184839505cd891df410ae8e676fc254df2ca5", + "eval/datasets/coding_memory_v1/oracles/fjord-north--poisoning.py": "6d9b91b4d53925d58d3c0cea694f591572a53184f23816101af0aa644aa1b86d", + "eval/datasets/coding_memory_v1/oracles/fjord-north--scope_boundaries.py": "2e98afa064e8a295a78bea654014a92734696a235fb0538732bb8d85799c7287", + "eval/datasets/coding_memory_v1/oracles/fjord-north--temporal_history.py": "4059ab8e427c2b6fd68155992404fbf493c7ebdf0a4357aa5c5f4ff0c67808f9", + "eval/datasets/coding_memory_v1/oracles/fjord-north--unsupported_questions.py": "79f14e18e282c1b6bbe21ef2affc9dd8a259a171c2e4975d42363f7c91689c1e", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--code_relationships.py": "9ea8a9ade1001bcd43929782c3410f35fc5056e438342ca848fe0b1b0ff56170", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--condition_values.py": "09e223a1c0bff6c95a5863fdc28dc00ed96b90887cd635d014712058079ccbfe", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--corrections.py": "f822c373fc4a93189fdce09999a9dcbe8797b0c0dc780eddd4e342edf21dc14e", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--long_documents.py": "983aeaae1e14937a740218228778be3e1938cdf1ff9228eba44064336188f297", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--multilingual.py": "6e929f8846cfc756c71bcc21bc0f27ef8fc23270d21292fc35d70e6bbd31c066", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--paraphrases.py": "c644d9f01edbd213447618bc269f56804450901faf4f4660c1f666d4e9b4ada4", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--poisoning.py": "cf3849312f96b0072efdfd4e858abb24d7e6a808bf883bcc9dde5dd5a4594364", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--scope_boundaries.py": "373d10e20713bdc26849706525f765f2be4ab57e87e576e39a1a1dd4e7feed88", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--temporal_history.py": "4d2c36480cc04d43aa8f535bf998968b474f2f596d5349dcd42684c5a667c6a4", + "eval/datasets/coding_memory_v1/oracles/fjord-violet--unsupported_questions.py": "b381d2016dea525eebd8de8d026dc48ad2080b93823a95d6e2bb06d29cc86775", + "eval/datasets/coding_memory_v1/oracles/fjord-west--code_relationships.py": "e1e10ed9ee8a4d5c2e798c2bf04cba5c39eaa3873901bd93724af166ba2479ae", + "eval/datasets/coding_memory_v1/oracles/fjord-west--condition_values.py": "427764a62d2e11f6c9fe4b2cdcd32fa92f61749403dff648601a47187e38787a", + "eval/datasets/coding_memory_v1/oracles/fjord-west--corrections.py": "4aa041aa87f4c8628d113bab855a151c2e38bd6919cd02224827fd07d060fb34", + "eval/datasets/coding_memory_v1/oracles/fjord-west--long_documents.py": "89875548e6952ce810d8983b0eaf2724bf8f329bf44e4d9193cfa3405e1ce80e", + "eval/datasets/coding_memory_v1/oracles/fjord-west--multilingual.py": "8b5c14cdb26941d1ae25701bc99efc5fe4ae6e50fed3c65b193952444f1593d1", + "eval/datasets/coding_memory_v1/oracles/fjord-west--paraphrases.py": "bcdc9a5f1fe45ef63fd5e6b728c83c7a8d05c7b80957e3276c074700b409e924", + "eval/datasets/coding_memory_v1/oracles/fjord-west--poisoning.py": "5504e1bab0865fb3041c88f151938789d79422eff91a3ce72f4a41fc14e2f897", + "eval/datasets/coding_memory_v1/oracles/fjord-west--scope_boundaries.py": "63f226631a663dca0ef62a4f32bb98fbe1e30e3b1f030ee42c3f7f2bf6a09af0", + "eval/datasets/coding_memory_v1/oracles/fjord-west--temporal_history.py": "75c9582a16db2c89f12506bd7a4115fcd871c6da63e80429a8e70e31c1f5f862", + "eval/datasets/coding_memory_v1/oracles/fjord-west--unsupported_questions.py": "05e4ba65be7cb4e6500af5e14721c70f93fd7390abe58a60f6f8e4f8e9b0ae5c", + "eval/datasets/coding_memory_v1/oracles/grove-green--code_relationships.py": "bd5619518e839f84ca03297c4f7842a2944d8b445ec1863a8dfc68e4c2114a03", + "eval/datasets/coding_memory_v1/oracles/grove-green--condition_values.py": "67377847b6ff1ab865f5c25351fcfa1ee13e6ec6a089444cb3eb0709b660148b", + "eval/datasets/coding_memory_v1/oracles/grove-green--corrections.py": "c07e8a03c6c47818f9c574d4f21b27668989711fb5e0af3b5cf1a87b293abcbd", + "eval/datasets/coding_memory_v1/oracles/grove-green--long_documents.py": "3f75fc2158c66a19dbbd19209fc80026339abe93a2282ff55c7c01429ac549f3", + "eval/datasets/coding_memory_v1/oracles/grove-green--multilingual.py": "19b1f800b38a11cf0f50df6da3a011806098561e32a279b78e1806de9b8c11fb", + "eval/datasets/coding_memory_v1/oracles/grove-green--paraphrases.py": "a5996376b9f589283dbe301dd1e0644af92a8b6a0222b51b99ed6fbb83570782", + "eval/datasets/coding_memory_v1/oracles/grove-green--poisoning.py": "c4dee0969e39b52dc9b697e45587c8cd84926a97c9a48f7211c0a00571bfb657", + "eval/datasets/coding_memory_v1/oracles/grove-green--scope_boundaries.py": "8d45422f0a1a2115deb90f560409fdad1105b7cae1f03052f8ebac404ba9517c", + "eval/datasets/coding_memory_v1/oracles/grove-green--temporal_history.py": "abc1b386173c8252ac10f11a36fbcdd6d91c731686108a6d32c62d17955d21b9", + "eval/datasets/coding_memory_v1/oracles/grove-green--unsupported_questions.py": "1ee09741a1c94a1fa83c7edf5b6c07d8da2e45826f748d899ec71917e8369d54", + "eval/datasets/coding_memory_v1/oracles/grove-north--code_relationships.py": "fd75d734a5f45a5ec54b125c1d379df67c5531659f212a2f5c1c5e88169167e0", + "eval/datasets/coding_memory_v1/oracles/grove-north--condition_values.py": "c600c0e5f50e352406b99527b050f4df6dc8ab8920b5b946de190ad7629b5e6d", + "eval/datasets/coding_memory_v1/oracles/grove-north--corrections.py": "4d6fdbe4f5eb9c8e4bfc2a974e6c3cbf1d124d0b20d3f050c84e766f507b4be7", + "eval/datasets/coding_memory_v1/oracles/grove-north--long_documents.py": "5b0bbb9f0a90eb377cbfe45ff6f37077a9ce32c7c6730720f0bfd49b844fdf4b", + "eval/datasets/coding_memory_v1/oracles/grove-north--multilingual.py": "525ea2fdad66d66d7b90892e9424ebae281c1dabfb49f2668784a687b7ec894e", + "eval/datasets/coding_memory_v1/oracles/grove-north--paraphrases.py": "1105e52a8c8d3ff79c64b59ec5f23d66d742b8dcecf91e8e62060895c7371f4a", + "eval/datasets/coding_memory_v1/oracles/grove-north--poisoning.py": "7d32e533f6dac57e5d651868386686d2b78e7db1cc29b1860fede23fc9c49813", + "eval/datasets/coding_memory_v1/oracles/grove-north--scope_boundaries.py": "7bea4129589ea34ba10667a7035b126a32b423ec7f3b1ff2b10f6ffc9aa1c2ee", + "eval/datasets/coding_memory_v1/oracles/grove-north--temporal_history.py": "859b6504250f5746ce197e4a0a7c1006f1c2257b939c53a493f9d36f6d2b7d6a", + "eval/datasets/coding_memory_v1/oracles/grove-north--unsupported_questions.py": "6b893c4c124f071e9ad5d9946448b070c2b2f5f10da6f058e60840c18510fd0f", + "eval/datasets/coding_memory_v1/oracles/grove-violet--code_relationships.py": "09bb7df6b115b83cb98a4274b2d883e81b809643ed4f912b383a9e09a7f7a1f9", + "eval/datasets/coding_memory_v1/oracles/grove-violet--condition_values.py": "a68c126f0d7b2fc0950baa5df095d7d513cdb8359883404bad3c1c84c8de666d", + "eval/datasets/coding_memory_v1/oracles/grove-violet--corrections.py": "46024b5fb187aaf3574d380af1a3bb2aa44028f89377f810ca5e98ee03c3b9aa", + "eval/datasets/coding_memory_v1/oracles/grove-violet--long_documents.py": "ec8d401f6437fbaa3b11ddb96973b31944ff374da677f7aae26c0e79e6a1315b", + "eval/datasets/coding_memory_v1/oracles/grove-violet--multilingual.py": "d14e8bf66593ceab7b691119b8a82f211933e2b3f6ebf8a25a75e61ffedbc1c0", + "eval/datasets/coding_memory_v1/oracles/grove-violet--paraphrases.py": "963b7321a32931f238d12e7cdc3849f64682c7cb704b77f4b849543cc40b6df0", + "eval/datasets/coding_memory_v1/oracles/grove-violet--poisoning.py": "39e5edc76a122d37d90b1af28c2293e2f8b0c38e1bdb6460340c0bc2a71ad87d", + "eval/datasets/coding_memory_v1/oracles/grove-violet--scope_boundaries.py": "472298607415aedfb9a36bad7783e7b7e3bccc339ee44cd83746d782ca99e411", + "eval/datasets/coding_memory_v1/oracles/grove-violet--temporal_history.py": "dcf793210a0a547c25587bba70e01d4d56576604856ed084eb5fa466b8be7b83", + "eval/datasets/coding_memory_v1/oracles/grove-violet--unsupported_questions.py": "bbe1769dc0afbe220e8069ef21b85164c31ecff6c3c7f5bf6d18d1e48f5f3c72", + "eval/datasets/coding_memory_v1/oracles/grove-west--code_relationships.py": "5374fb66cd91be79b5eabd2696ada5fee89605d8a1fc84ce36bd6a8ca82cbe5e", + "eval/datasets/coding_memory_v1/oracles/grove-west--condition_values.py": "2e807846b760b65afe0f5e48947463a66c15b8d89899a2724cddd865311760b0", + "eval/datasets/coding_memory_v1/oracles/grove-west--corrections.py": "bca219aae5f85e3f5504d5b39646efe89ff7dc6c007a4323abd5c5154cb4d5a9", + "eval/datasets/coding_memory_v1/oracles/grove-west--long_documents.py": "19c3ca44f78ef79ca1b907c80214e11a8195e994893f1de683b5d9796eb29d6b", + "eval/datasets/coding_memory_v1/oracles/grove-west--multilingual.py": "7842efb41f88fcd49e7a2350e9b44a3a1946bf73e5cb103e4a91a11e810301a2", + "eval/datasets/coding_memory_v1/oracles/grove-west--paraphrases.py": "fd70e8a20da665073ad172219bd57b24e53575dccc344adc1fa0190770ffc6cd", + "eval/datasets/coding_memory_v1/oracles/grove-west--poisoning.py": "473acfa06852554dee20445dd1a86df4e626e0363a7b72de57d68819314bf938", + "eval/datasets/coding_memory_v1/oracles/grove-west--scope_boundaries.py": "3ba0e93abf07498c570cd81b0c4f753bb75b6d4f4bf6c786461bad1a3b301286", + "eval/datasets/coding_memory_v1/oracles/grove-west--temporal_history.py": "334bc7b057f570efd20896d282e54c986f5b6c642930057c8520f84b721ed5e9", + "eval/datasets/coding_memory_v1/oracles/grove-west--unsupported_questions.py": "5102eb2dadcce0dcc0ba157495d4fa10c4983e4bf6627aff25bbd547a08115ca", + "eval/datasets/coding_memory_v1/oracles/helios-green--code_relationships.py": "d376cda89c44b3282a86afe7cd01583f0b53f2c80cd92b47bc2fa4a19271e9ee", + "eval/datasets/coding_memory_v1/oracles/helios-green--condition_values.py": "7b6eaf2f46a5b2f1ae6e8e6da0bcfe6044fea5639df158e137fb8ba32b929909", + "eval/datasets/coding_memory_v1/oracles/helios-green--corrections.py": "2e849f6c926aa0ce1b2e69ce8cdc5ba9c50feb89b33b257fc71bf3aa3b8461b8", + "eval/datasets/coding_memory_v1/oracles/helios-green--long_documents.py": "884c17b7d20d986708c0eff2821cc7b76c5f7b16afd26c43ca80fb507e2f3ea7", + "eval/datasets/coding_memory_v1/oracles/helios-green--multilingual.py": "1182048aef0dcfbc416a004bc43f49f1d2a5de4132e7243ab7d147f54aab1acc", + "eval/datasets/coding_memory_v1/oracles/helios-green--paraphrases.py": "a73950ae7fceacfaa6c7c7a26c8379f9164a29ebea883cede9222e72c506db5e", + "eval/datasets/coding_memory_v1/oracles/helios-green--poisoning.py": "60c5112afb7b84ee5513f2657f42d0eacb0607ba345c28618af3b3b3eaab753e", + "eval/datasets/coding_memory_v1/oracles/helios-green--scope_boundaries.py": "dab035b811e7e9a60bf447d1ea26f6db915c320b5dc2dd04cda4b6f1023296be", + "eval/datasets/coding_memory_v1/oracles/helios-green--temporal_history.py": "c22dd57bc7cd659863231b164192caf6c4929c5baaf8edd8be8cac00268e5847", + "eval/datasets/coding_memory_v1/oracles/helios-green--unsupported_questions.py": "f78986b8093eab4ecc8c1319d359498d64177db55a4e38cf8d0b32d1e5af4b2a", + "eval/datasets/coding_memory_v1/oracles/helios-north--code_relationships.py": "81721c60bd4e3466600442214147ae6641ebc05ae414965f4a09b89cc74cb8d6", + "eval/datasets/coding_memory_v1/oracles/helios-north--condition_values.py": "8fbcf116c1f0d480d61b2974e7bef73acef053d45bf0ac3b12e6ea48b017e43f", + "eval/datasets/coding_memory_v1/oracles/helios-north--corrections.py": "beae8d6c5828c1f225e5e953236136a217ee48e67cebc82049caf8f1c60cfdcc", + "eval/datasets/coding_memory_v1/oracles/helios-north--long_documents.py": "019e7ff3b9efcb461c1c68acc657b2438cc2d63d4b4165c0caa4523c440f12bc", + "eval/datasets/coding_memory_v1/oracles/helios-north--multilingual.py": "a48324390ab60686c2b1417f0c049f47601737c55fdf000b696ccc5de260ccab", + "eval/datasets/coding_memory_v1/oracles/helios-north--paraphrases.py": "9e2552399b988a6d722ccfaec7cfbea10853164336fadc9e9146dbae788a2d8a", + "eval/datasets/coding_memory_v1/oracles/helios-north--poisoning.py": "75eecaa42c302898581d3bef8aee00fa1dec6c3bc968c0014a522196e18c7681", + "eval/datasets/coding_memory_v1/oracles/helios-north--scope_boundaries.py": "8bd32a30cd1c2e346375bf2b37ef4908c3feeb67a80ec52ed3108bfe9cbd548e", + "eval/datasets/coding_memory_v1/oracles/helios-north--temporal_history.py": "1d1118cb5d438f91bb027acaa3b64f867e93cca7eb0f480bb83073584fafd97a", + "eval/datasets/coding_memory_v1/oracles/helios-north--unsupported_questions.py": "02a854bf89085637e366f801d2be6ab26b9e18296f13531485d08c2a64621c7c", + "eval/datasets/coding_memory_v1/oracles/helios-violet--code_relationships.py": "e32101847ccad5cd18fbc2c667ba4d1b63e3e7f8067dc89de5f485546dfef525", + "eval/datasets/coding_memory_v1/oracles/helios-violet--condition_values.py": "c8ef59dc9dcfe3aecc990d82fe44f1f93b7adc8c65f1fc38e79b934d351635c6", + "eval/datasets/coding_memory_v1/oracles/helios-violet--corrections.py": "a3a36bb554a6f4a83ea612bbbe9c0ee4ee1a08071e490a9a909c46d3049bb379", + "eval/datasets/coding_memory_v1/oracles/helios-violet--long_documents.py": "f9fa707bb28b4d7ea96e3bacfa22f24502b571a76a79defd147fe9d5205963a2", + "eval/datasets/coding_memory_v1/oracles/helios-violet--multilingual.py": "cfbfe0535e905dd58434bc25d84e7746dbdf1fc50d72d0fa96cc78677eaf18b9", + "eval/datasets/coding_memory_v1/oracles/helios-violet--paraphrases.py": "1743c8cbf0564d91dfb4db88b694072fdc9bb5a0b9f26f578649af7f539ee393", + "eval/datasets/coding_memory_v1/oracles/helios-violet--poisoning.py": "02cf390520757473e9d677f5c1f0934ecf50645ad5edcdc64caec23721a5d99e", + "eval/datasets/coding_memory_v1/oracles/helios-violet--scope_boundaries.py": "5ab003dfc23cb00d0a70195541a780a4b80b1d496728b2b88f7f154310e67a6d", + "eval/datasets/coding_memory_v1/oracles/helios-violet--temporal_history.py": "209e95f62c39d3f2223ed00a378788e396d04de392324455cd9d19dedb3a87ad", + "eval/datasets/coding_memory_v1/oracles/helios-violet--unsupported_questions.py": "a3ff0703297e2f743917e74abeffc0c922a6241318f37d63a49242c6bcc18059", + "eval/datasets/coding_memory_v1/oracles/helios-west--code_relationships.py": "4d8543e4c49d59f844830ccee3fcea0643ee17ce43f51863b94a1b4fbc9825ea", + "eval/datasets/coding_memory_v1/oracles/helios-west--condition_values.py": "f2efa7fd803df0e10c440d5ec2d98dedf5e02e23a4cedddcde5ea722f4039e40", + "eval/datasets/coding_memory_v1/oracles/helios-west--corrections.py": "181c42640cee47db2b5c779bdf8fe2e84103a92459d7f6cb0d9587458e421406", + "eval/datasets/coding_memory_v1/oracles/helios-west--long_documents.py": "49337830deb24f9d6b11ba4e01f397c0cf1a40a15524f1ce33354a2c61615072", + "eval/datasets/coding_memory_v1/oracles/helios-west--multilingual.py": "f1e4ef8be6bba13457a2da1c3a2082f81bba765f2cf1f07b93eb20edf5e2debe", + "eval/datasets/coding_memory_v1/oracles/helios-west--paraphrases.py": "29abfff3b323496fb62e739cdc0be2aada1b7e1467a141cc8d19a1d54c1ab130", + "eval/datasets/coding_memory_v1/oracles/helios-west--poisoning.py": "14b7f874c10681ff3829b2659cbe7500250369652d11b26c842e62718093ca0f", + "eval/datasets/coding_memory_v1/oracles/helios-west--scope_boundaries.py": "decbdb3cd7ab0c40c0aadb1ecdeef9e2ee2586dd8802b9bb16235f211ce05ee1", + "eval/datasets/coding_memory_v1/oracles/helios-west--temporal_history.py": "6e7c2369157234374d3f9633b7e772b4a3bff405d418c6e1336e56c87ebf28de", + "eval/datasets/coding_memory_v1/oracles/helios-west--unsupported_questions.py": "379716803a300bbecd3fc8f1661bdac2fcd8a98a2bb18514d8de168bd57fa144", + "eval/datasets/coding_memory_v1/oracles/island-green--code_relationships.py": "833962801d594247d5c5644df0e6aee877d452127ea5c0dbc22c81512132ab21", + "eval/datasets/coding_memory_v1/oracles/island-green--condition_values.py": "cfbb4ea0a4dd12d69686178ba8246bd08b93743dc5395cd0479984a53655d909", + "eval/datasets/coding_memory_v1/oracles/island-green--corrections.py": "4c7ee2538df9988864e24d9b86805d26b64d1a37638b57abdd1761e24c4ab340", + "eval/datasets/coding_memory_v1/oracles/island-green--long_documents.py": "f73cd2b884539680c500af9a5c8cc00d1dc9f1d2ef7bda61f6c8253b2bf799ed", + "eval/datasets/coding_memory_v1/oracles/island-green--multilingual.py": "e954fa56aa819ef9088c8d6da4abd647b689956c4ff4a7b0b8e31078522f2702", + "eval/datasets/coding_memory_v1/oracles/island-green--paraphrases.py": "4cde223eac40d4a2962aa2ace533b8761c5b82784767b0307c884ac5e6ad0576", + "eval/datasets/coding_memory_v1/oracles/island-green--poisoning.py": "08725e295848b2c58c9e6d6a0554a2dd4a6506af75039205c444251d11a62acb", + "eval/datasets/coding_memory_v1/oracles/island-green--scope_boundaries.py": "257cd579af07491748f5de88d04f8af85056b0448d348e59c7b0cbb28754fb1b", + "eval/datasets/coding_memory_v1/oracles/island-green--temporal_history.py": "e93105f0639a83b300a6de3bbae56166eedac2e58122f7e84ce5a40a074db6b6", + "eval/datasets/coding_memory_v1/oracles/island-green--unsupported_questions.py": "a707cee90fb3f0cabd0e692399a361f59da0fc4e9e7bbf0cde702e05579f0b3f", + "eval/datasets/coding_memory_v1/oracles/island-north--code_relationships.py": "f00e022ce900e4ad0fcec93f2dbf016337a4d83afacbcf739f61f71316723173", + "eval/datasets/coding_memory_v1/oracles/island-north--condition_values.py": "c704a03a70253585a67823826dc0a6ebaf6323a8880c00af80c148c150176daf", + "eval/datasets/coding_memory_v1/oracles/island-north--corrections.py": "e2cadf6e4f365f94cf0f440758b173da4d14ba12c03a84351e0608f0c60bccc5", + "eval/datasets/coding_memory_v1/oracles/island-north--long_documents.py": "300246fd3d85d73b63066d133361bc784ae26b46f4197c5c41ac960eb37f9508", + "eval/datasets/coding_memory_v1/oracles/island-north--multilingual.py": "f8aba8e14ed3c420e90e701880ea38e924e1219e3cc3643ba8bde17cb31ddef6", + "eval/datasets/coding_memory_v1/oracles/island-north--paraphrases.py": "503ce2c132c666bb67920e90d4a898f5a73f1d66779cd83c1885812f223e30e5", + "eval/datasets/coding_memory_v1/oracles/island-north--poisoning.py": "3369d956e5af12bc6be80828c4c6ab03ccf80c79e38dea3162b4e2c09771ff30", + "eval/datasets/coding_memory_v1/oracles/island-north--scope_boundaries.py": "9b4f6186f71713175f1eb5aacb0b6258c746b2297ca7c692f54bf5feabd5d261", + "eval/datasets/coding_memory_v1/oracles/island-north--temporal_history.py": "7c22b0edbf8edb7a511a581ffe6f1ce14454ef3feb527eca90d1386ce00b2c6b", + "eval/datasets/coding_memory_v1/oracles/island-north--unsupported_questions.py": "c1e5d526b0c6b88394f19a8b8edca7dae1db718b24be2d57374eb3769bbd1f24", + "eval/datasets/coding_memory_v1/oracles/island-violet--code_relationships.py": "9672d608ea0c8623ff1fe760f943173b2fdac5dc4ea4b7ae0dba33436d860254", + "eval/datasets/coding_memory_v1/oracles/island-violet--condition_values.py": "cfea91adcc9b3f9eb28cc24cf310e10b436f81cff51297f161b9a3b59a21f62c", + "eval/datasets/coding_memory_v1/oracles/island-violet--corrections.py": "075d48a2dfd91af8328982fe475ec915c00b75db38819aff3b4c2343147e2bfc", + "eval/datasets/coding_memory_v1/oracles/island-violet--long_documents.py": "85cc789277e59925b36ea3bd82e23f989e79fb0be2e55083e2656e518641dd3c", + "eval/datasets/coding_memory_v1/oracles/island-violet--multilingual.py": "84615b70eb74eb2180dd8a89d8d34c11e904030bfaa26c57e2e6c968daf64d29", + "eval/datasets/coding_memory_v1/oracles/island-violet--paraphrases.py": "5244d32ffc4127d519e2f5a875ffaecf7bc51eaa1426d2aacbc42490cf291d22", + "eval/datasets/coding_memory_v1/oracles/island-violet--poisoning.py": "9bde478876ea0be60555d5bd26d3790e4f305327ece10bf6c258bc564354fa1d", + "eval/datasets/coding_memory_v1/oracles/island-violet--scope_boundaries.py": "8e8b0a64342963b4dbc000578a72c0e0801fc1ca4af140f09b55d4cdb03b1274", + "eval/datasets/coding_memory_v1/oracles/island-violet--temporal_history.py": "3e187ce259c7f57f6dba4a28c3ecdf31fbe6458f7d908b22cb8b5ce8325437ae", + "eval/datasets/coding_memory_v1/oracles/island-violet--unsupported_questions.py": "0e60f29a47953c5d434c159eece26bbb03cc837842c493111b34310f9cff4693", + "eval/datasets/coding_memory_v1/oracles/island-west--code_relationships.py": "c19c42978fa5f9c6e1f41886eb1cc0fdb7e12304ba58b160b4cbe1975640acfe", + "eval/datasets/coding_memory_v1/oracles/island-west--condition_values.py": "07c4c591e08f5248f3016dc16cecfea7dd75bebff1a61599f258938d7031043d", + "eval/datasets/coding_memory_v1/oracles/island-west--corrections.py": "a7d93dfb579ec4f58cf2f6d29a6d0ab7dbf1a17f6ae49b98902a5ea22525e649", + "eval/datasets/coding_memory_v1/oracles/island-west--long_documents.py": "20a50dd63e04ce751115fca671b9da7ebbf240e26f39d494583cb405a03b750c", + "eval/datasets/coding_memory_v1/oracles/island-west--multilingual.py": "65d82c4d7968e0e597509c692015bdac0b92841c726d87560d9caa32429176f4", + "eval/datasets/coding_memory_v1/oracles/island-west--paraphrases.py": "feb7eb2e8592162d7d1f433610cb02ba04cd9797191d65a9f0e5bc4f4711c4ed", + "eval/datasets/coding_memory_v1/oracles/island-west--poisoning.py": "3d7c2c6a7f247f3e8158d5ba007bbbdf027a74e9a2c52f52953b3ccf2c21f17f", + "eval/datasets/coding_memory_v1/oracles/island-west--scope_boundaries.py": "ff0f36139f6da5428b9538d108aded522e62ca803035ebef78e11255cf55e08b", + "eval/datasets/coding_memory_v1/oracles/island-west--temporal_history.py": "42cb1c552ce01308b47d526c6e6993fb3270239b392cfa9f5e42b8e0fcfb3f82", + "eval/datasets/coding_memory_v1/oracles/island-west--unsupported_questions.py": "911b4257380640d35d6e2fc5292e235eedc113f5e71d8e71e8ccab7c72d4f290", + "eval/datasets/coding_memory_v1/oracles/juniper-green--code_relationships.py": "373395d3b70f64f8102609d27ce10c4aeb937ac73988c2ce2be1e283df94ef82", + "eval/datasets/coding_memory_v1/oracles/juniper-green--condition_values.py": "f04794a96b06d537550227ec308b1f02405cdab42d7d3ae4db4afd2c25bee8cd", + "eval/datasets/coding_memory_v1/oracles/juniper-green--corrections.py": "5a354a8c8efdbc498d25ffbf6095fee7b3d7d638134b452c7c9ecb186b4e1bc7", + "eval/datasets/coding_memory_v1/oracles/juniper-green--long_documents.py": "8610157668d6a90c794a956f6f0b432f03e3559238599ad1cb412f6f1a2f7e46", + "eval/datasets/coding_memory_v1/oracles/juniper-green--multilingual.py": "dcb99335fa18b008197532809ff7127be2cda600848e10c8bfe4677bd896f90e", + "eval/datasets/coding_memory_v1/oracles/juniper-green--paraphrases.py": "7a4cb1bbe3de2e6fe6a33b1f78829f4aeb21d21c2e8d4d3399d3d98958ba0909", + "eval/datasets/coding_memory_v1/oracles/juniper-green--poisoning.py": "ba73c2c6f04590538e16c50be527da60f5949781bc8833a3c18a7c597146eb10", + "eval/datasets/coding_memory_v1/oracles/juniper-green--scope_boundaries.py": "9d09c0851a3906b6281cb1c5811714224161e27a1aa46edd81c5659e7f93cd3d", + "eval/datasets/coding_memory_v1/oracles/juniper-green--temporal_history.py": "fd05595a452952dca2d0089f5701a60145a3b5705b78a5ff2f8f9f13fbf1c259", + "eval/datasets/coding_memory_v1/oracles/juniper-green--unsupported_questions.py": "f58bfcb89db799dc1578fc8791b49e97e7f22b30e720aeaf661f0ad2e6a9660c", + "eval/datasets/coding_memory_v1/oracles/juniper-north--code_relationships.py": "a84f02bd3c2b9c367248dae8aa3580d83de6a1beb6fe464553eb8a2bfb172b75", + "eval/datasets/coding_memory_v1/oracles/juniper-north--condition_values.py": "d70311ae339556f18240f824d0f0bd0024060c2698fcca8fe7cb540e8a8bc727", + "eval/datasets/coding_memory_v1/oracles/juniper-north--corrections.py": "fa290629339007fdd4230a76d49fa294a218c5f0cf6b90c3074678b0f16c539a", + "eval/datasets/coding_memory_v1/oracles/juniper-north--long_documents.py": "e30025d68ae3b48b566bf3fb2aea5e3603f5f7ab658da8fcb2256d3b1964c85a", + "eval/datasets/coding_memory_v1/oracles/juniper-north--multilingual.py": "b8bb0199c4acfe71df8d4e7df75bdd37e038cc72aa82bc6df10715819aeb3686", + "eval/datasets/coding_memory_v1/oracles/juniper-north--paraphrases.py": "aabf70509d3b0602d43c9590f04fa3ad1370a40c266d20f64a5ece15d4c79a51", + "eval/datasets/coding_memory_v1/oracles/juniper-north--poisoning.py": "ebfb9d7bf3c50eed8f44b73e391496ef377942de389b6feb64c7a1c0f9646444", + "eval/datasets/coding_memory_v1/oracles/juniper-north--scope_boundaries.py": "06c2c9d57ade79139c9b0d20568703477bb9e3a934f2f8900ad717ef157e5844", + "eval/datasets/coding_memory_v1/oracles/juniper-north--temporal_history.py": "0d97ba6d7a5d7741bf248b599ed18ae81b1c0053ad53cbc64934f31e9c3461aa", + "eval/datasets/coding_memory_v1/oracles/juniper-north--unsupported_questions.py": "526648ba7a4edc21bc2eb589c1e5c98a20fb07aa692687de4c3845e559be8096", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--code_relationships.py": "c73fd24d0adbeda6c7afeb15ab07296c0438402893e8d442af26865106d6e8ea", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--condition_values.py": "4aadae570ef84f976d49ba1fc92574ef55630b1c3017e90dbec1f85e6fb74c4a", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--corrections.py": "07236b2a88df73c8d0f6afbe7d972fdc18f8c835a852f5c20748a369c77d82e2", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--long_documents.py": "54d3201b621b168ce03963c423942f142cb403f9d90234f063ff33b7add4d2dc", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--multilingual.py": "e0cc26d743cbea8e98261895474879d078fd53941c075606bfc6e52ee098db93", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--paraphrases.py": "a1f836316408052cab1253b56b5caa6b1432fe6eeaabbb5872da97ec9bde633a", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--poisoning.py": "bc09271eadf77f1b6c03fff20ca9c0f0a42a99d1c5600e6b7843f0a0d549ed5d", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--scope_boundaries.py": "17a7258bed9efe1c86a05756c5b5d17926fd72f01de4b2addd5d2fb4550f4da5", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--temporal_history.py": "8bda8599d861857204c48b55729607c8cb0cb3ae2701eb18fb600d398468cafe", + "eval/datasets/coding_memory_v1/oracles/juniper-violet--unsupported_questions.py": "10cdbb013f78f0e54c31a6e76f03d8d15a6063fd7971a2d803a390890c036ad2", + "eval/datasets/coding_memory_v1/oracles/juniper-west--code_relationships.py": "f0a682d413b3d8418c664db2775de4eef35240c508cc71ca6bbc055cd0c8dbe6", + "eval/datasets/coding_memory_v1/oracles/juniper-west--condition_values.py": "200a4b78f201e992ac10112a194c97b88841dc5930e3e7de9bfc076bf4c69fa7", + "eval/datasets/coding_memory_v1/oracles/juniper-west--corrections.py": "41ea9b5a5f40d8703343f4570593c3f27905e5ff70fae57bdbdb5fcccc585d22", + "eval/datasets/coding_memory_v1/oracles/juniper-west--long_documents.py": "7d8cdd01c5d90ad62e9da93a99febead683e3f01c5ca37315382a2fb0f512d76", + "eval/datasets/coding_memory_v1/oracles/juniper-west--multilingual.py": "1815adc5946bf9e7dba16786d8c29f9466537b7a43b1c392f927b61c56da54ee", + "eval/datasets/coding_memory_v1/oracles/juniper-west--paraphrases.py": "5d0110bc2098d2143303476fb5c5714086d0278b848110c63aa626e0b88311cc", + "eval/datasets/coding_memory_v1/oracles/juniper-west--poisoning.py": "4e16d9de9931cffc6915805b4fc375dad3ffb7fad17eb893205ce6d07102bcb3", + "eval/datasets/coding_memory_v1/oracles/juniper-west--scope_boundaries.py": "6c5d875184c890a922ee18ac423598fdcece109a91e9e62d5a7ccc82d2eaaa70", + "eval/datasets/coding_memory_v1/oracles/juniper-west--temporal_history.py": "3404fdbb5cc4693333d48d27da928531279d148ffe1cdadbdd741dfc1a8d38d3", + "eval/datasets/coding_memory_v1/oracles/juniper-west--unsupported_questions.py": "b85a4cf14f9596cd14b480acd323772fc91084ad1413fa27435bdb7ffe5baa6b", + "eval/datasets/longdoc.jsonl": "7f5ade95e1f283d0db8cf78e53ed8995d3534f847e616d2c0005fd8da37ac790", + "eval/engine_capacity.py": "1940da1c3a435d81c7b7a1a689e966c23b92e4f94af115b45813dd30e2c04eec", + "eval/evidence_contracts.py": "03969fdf01e78135c66e698d3f523e5312ccca3737dd3036b37a2c6676ba4ea2", + "eval/external.py": "98500a97c18152b1b86a270062033c48176224d7468ac0f703ddc7de862a30e2", + "eval/external_checkpoints.py": "484e298039bfa04eef86bd7145428c312aaeaf059f5bdf8da7ae29d1150008d7", + "eval/extractor_quality.py": "50820fe7f821d111e17e4e77a3d1159e7d6979d110b052c4f254a5b309c2652a", + "eval/fts_insert_scaling.py": "6088997e0f85d430fafb74509e24a960fbd59c06c8b2839285942b97d2116724", + "eval/graph_every_bench.py": "79da573c5edf315f71bfab412d3ea283b8da45d1fdabc78ddd19302ead09e4f0", + "eval/graph_traversal.py": "b094f75c3a1d75ba3cf19e372187d692d3c595a1d9bde19bbe95ae0c79a5175f", + "eval/grounded.py": "053d5193b716a2c3e507fcd44057d392de910a4442b46bd7cc1f30ac0ba68541", + "eval/handoff_quality.py": "7daf635510764e236f48ca7e2537a85d8ebc1ad995513144329c1f0236405937", + "eval/harness.py": "8c96c26a121dfc2d9ea051a05861af8951c93a732dba9cb13de0de178414b016", + "eval/hosted_evidence.py": "7946cd8c1e3aa291268271b2aa11210d5f09c9b05ba635aa0b64bef07bdcee45", + "eval/hosted_ledger.py": "a53036d12ff671371c148910a816fb20c7e7f3346250a303722352f47b06c476", + "eval/hosted_luna.py": "4dbf02a65eec38bcde92d82952a0b372abbac1f68378d11a04528f4a31a835dc", + "eval/local_benchmark_queue.py": "43fa67b4d653e770822da816511e8e62b3714e60d2edd49ea2a4c5c3d8d20851", + "eval/local_capacity_campaign.py": "ab4935263f9d24c4e38bd4f16164b4c4c07eff91c695809b6fc75299d4f03d62", + "eval/longmemeval_v2.py": "defb4d47f453aa4615a8f101b82df3fadf64f0f9eae0ce1021d86d3750dad437", + "eval/longmemeval_v2_evidence.py": "8486bd4dcfdd8b1f7a14c32fe4f427c980ccdcc3aa2eb018d7dfc40305412eff", + "eval/longmemeval_v2_matrix.py": "ca085cd59481813cce5dbdfbc94f40f67173ac3a1bc3dce6f6d3e09eb7b08153", + "eval/metrics.py": "16857e2cf6ed339cb57a26c9bfa1879444b4d279bd972e5a9fa644ed1308afe0", + "eval/native_coverage_scaling.py": "0d318c116241050fc0c7bdbbb5646ca67944d7c304ccd9a462834e0553ffa90a", + "eval/performance.py": "dccc55c26dc396f9fee98defd152900cf8aabeb5d6198bc711e0eb8f472ecc57", + "eval/performance_engine.py": "3d37cf0a5989c8fa6e8b6ab7ae0b2e0d5410a8b922d2f3130c1a7794d93dcbb1", + "eval/planned_recall.py": "f9a87291ecb181045b98ca65fe55db820bf7cee4ae4f4d087ca3458afdd7837e", + "eval/proactive_ranking.py": "8610541f1d547f9c0eb46d078dbcaa670c0a97157acc37f96ec08b482cb7a6ab", + "eval/productivity.py": "6d4644ebdc44472aeb3879963774fab276bfa269b781139c46ded48774a22717", + "eval/public_readiness.py": "5ce8a18d0bfe09e75e88548a589b8fc6d2cbf04c0f8cd1ce51cd9519136a2751", + "eval/redteam_poisoning.py": "fce120cc3adf2ee966b59cd3ea7f20d242a0af49b52492143543130fe14c0cc8", + "eval/reinforcement.py": "72ed766775a2658eaa728afec51c0ac22d97a90e813111954df66a6ec50f2bef", + "eval/repair_discovery.py": "db05496fbbcb0df86c5cdc2f0c85fb6b6605b0cace6cb44ae2add10b72784b5f", + "eval/resolver_reworded_corrections.py": "a9054778a37b2175b46f04b674b4779e358931eee5d2ae52bbc5f953d234fb9e", + "eval/resource_hierarchy.py": "5ab6c989bb143c4386749c45a33c190447e829c1b5657e8c0aca30f34bd69461", + "eval/rework_statistics.py": "e12c14288797c5cf2dd93d51f606244287bc2f4ff8bd401f1d1b072efd1f9529", + "eval/run_longmemeval_v2.py": "866863d9f8f7ce8f7c3c36741ab324faa5fae417827f907c655eace38de0f5ac", + "eval/task_pairs.py": "fddc54804e8837ec0731813297fb55825458317f898e29176e16f3f5a2f527fd", + "eval/user_journeys.py": "a1c8436d4a6871baff21f9aa4f6ac1545def60d946cd3b1454c3d9cd8661a049", + "eval/vector_scale.py": "3f9c327d9eca1a857512ddc208e972933be1a7d4fa7fa0017aca0cbf8fe7bb6d", + "eval/vector_scale_storage.py": "24040fd1b96b9f9cd43ea37b0b37ea118a02bd096aff77fc67ff8b0df769b1dc", + "eval/vector_scan_plan.py": "34fba3d029bfc78134e8c9c450b019ff56c3bdcbeb921400888077cb44e9b848", + "scripts/export_offline_evidence.py": "4e10c2b3d5f6a2024a7f95f7b87ce47ec2cb4ef31c55611b2e5cf35502ebc0ba" + } + } +} diff --git a/docs/benchmark-evidence/offline-fixtures-v76.json.sha256 b/docs/benchmark-evidence/offline-fixtures-v76.json.sha256 new file mode 100644 index 00000000..63173556 --- /dev/null +++ b/docs/benchmark-evidence/offline-fixtures-v76.json.sha256 @@ -0,0 +1 @@ +2fb5ce5b2cbc21541f7ae9cad5d2ef615014c0e86f9881f00621d00986422ba8 offline-fixtures-v76.json diff --git a/docs/images/context-efficiency.svg b/docs/images/context-efficiency.svg index 7eb5ca22..c20be408 100644 --- a/docs/images/context-efficiency.svg +++ b/docs/images/context-efficiency.svg @@ -1,6 +1,6 @@ Measured context and retrieval boundaries -Artifact-driven local deterministic benchmark report. Structure-aware chunks report 740.3 to 214.3 retrieved tokens per question. The performance run reports 24,590 full-proxy versus 11,138 compact-proxy tokens. Retrieved-candidate quality and packed-context quality are separate views; packed quality is shown only when the selected report includes it. Payload counts are a serialized JSON-shape proxy; the payload is not an MCP transport measurement. The report does not measure provider billing. Packed context reports 85.38 mean and 108 max under a 1,500-token cap. Source artifact SHA-256 aa7ed9c141afcc82fc2a05b63ed9037842f2ea8372667f3142cf9bb795833988. +Artifact-driven local deterministic benchmark report. Structure-aware chunks report 740.3 to 214.3 retrieved tokens per question. The performance run reports 24,590 full-proxy versus 11,138 compact-proxy tokens. Retrieved-candidate quality and packed-context quality are separate views; packed quality is shown only when the selected report includes it. Payload counts are a serialized JSON-shape proxy; the payload is not an MCP transport measurement. The report does not measure provider billing. Packed context reports 85.38 mean and 108 max under a 1,500-token cap. Source artifact SHA-256 2fb5ce5b2cbc21541f7ae9cad5d2ef615014c0e86f9881f00621d00986422ba8. @@ -105,7 +105,7 @@ mcp transport not measured SOURCE ARTIFACT -aa7ed9c141af +2fb5ce5b2cbc SHA-256 prefix Artifact-driven local measurements diff --git a/docs/images/evidence-backed-agent-examples.svg b/docs/images/evidence-backed-agent-examples.svg index 4025040d..903ac7db 100644 --- a/docs/images/evidence-backed-agent-examples.svg +++ b/docs/images/evidence-backed-agent-examples.svg @@ -1,6 +1,6 @@ Three evidence-backed Engraphis agent behaviors - A three-card summary of deterministic offline fixtures. Focused context returns 740.3 to 214.3 tokens while retaining Recall at 5 of 1.000. A grounded answer returns support for 5/5 answerable questions. An unsupported question safely abstains for 6/6 off-topic questions. Reproduce with eval.chunking_eval and eval.grounded. Exact commands and config digests are registered in BENCHMARKS.md. Public-safe artifact SHA-256: aa7ed9c141afcc82fc2a05b63ed9037842f2ea8372667f3142cf9bb795833988. + A three-card summary of deterministic offline fixtures. Focused context returns 740.3 to 214.3 tokens while retaining Recall at 5 of 1.000. A grounded answer returns support for 5/5 answerable questions. An unsupported question safely abstains for 6/6 off-topic questions. Reproduce with eval.chunking_eval and eval.grounded. Exact commands and config digests are registered in BENCHMARKS.md. Public-safe artifact SHA-256: 2fb5ce5b2cbc21541f7ae9cad5d2ef615014c0e86f9881f00621d00986422ba8. @@ -47,5 +47,5 @@ Reproduce: eval.chunking_eval + eval.grounded - SHA256 aa7ed9c141afcc82fc2a05b63ed9037842f2ea8372667f3142cf9bb795833988 + SHA256 2fb5ce5b2cbc21541f7ae9cad5d2ef615014c0e86f9881f00621d00986422ba8 diff --git a/eval/external_checkpoints.py b/eval/external_checkpoints.py index 59547429..683f5e23 100644 --- a/eval/external_checkpoints.py +++ b/eval/external_checkpoints.py @@ -39,6 +39,10 @@ class UnrecognizedRunnerLock(ValueError): """A marker does not establish the cooperating OS-lock protocol.""" +class LegacyRunnerLockRemoved(UnrecognizedRunnerLock): + """A verified legacy PID inode was unlinked during an existing-only probe.""" + + def _write(path: Path, payload: dict) -> None: path.parent.mkdir(parents=True, exist_ok=True) with path.open("x", encoding="utf-8", newline="\n") as handle: @@ -64,8 +68,9 @@ def _runner_lock(path: Path, *, create: bool = True): require manual inspection. Cooperating runners must keep the lock file in place, even after exiting. An existing-only probe (create=False) never creates or repairs a marker and - raises FileNotFoundError only for an initially absent path. Ownership covers the read - that depends on the producer having finished. + raises FileNotFoundError only for an initially absent path. A legacy PID + inode removed after opening asks the caller to probe again; it never grants + ownership. Ownership covers the read that depends on the producer having finished. """ if create: @@ -108,7 +113,23 @@ def _runner_lock(path: Path, *, create: bool = True): acquired = True try: - opened, named = os.fstat(handle.fileno()), path.lstat() + opened = os.fstat(handle.fileno()) + try: + named = path.lstat() + except FileNotFoundError as exc: + # Only a known legacy PID file can disappear as normal shutdown. + # Check the opened inode, not a second pathname-existence guess. + unlinked = os.fstat(handle.fileno()) + if (not create and initial is not None + and stat.S_ISREG(unlinked.st_mode) + and os.path.samestat(opened, initial) + and os.path.samestat(unlinked, opened) and unlinked.st_nlink == 0): + handle.seek(0) + legacy = handle.read(21) + if 0 < len(legacy) <= 20 and legacy.isdigit() and int(legacy) > 0: + raise LegacyRunnerLockRemoved( + "legacy external runner marker was removed; probe again") from exc + raise except OSError as exc: # Once a handle has been acquired, disappearance is a changed inode, # not the absent-marker compatibility case for an existing-only probe. diff --git a/eval/local_benchmark_queue.py b/eval/local_benchmark_queue.py index 65297001..c8b7be31 100644 --- a/eval/local_benchmark_queue.py +++ b/eval/local_benchmark_queue.py @@ -368,12 +368,6 @@ def _prerequisite_ready(artifact: Path, producer_lock: Path) -> bool: except (RunnerLockBusy, UnrecognizedRunnerLock): # Legacy ephemeral markers must disappear; never reclaim one by PID. return False - except ValueError: - # A legacy marker may be removed after the probe opens it but before - # it can validate the pathname. Accept only confirmed disappearance; - # a marker that still exists but changed remains a hard failure. - if os.path.lexists(producer_lock): - raise if not artifact.exists(): raise ValueError("prerequisite producer stopped before completing its artifact") _verified_artifact(artifact) diff --git a/tests/test_benchmark_evidence.py b/tests/test_benchmark_evidence.py index b6f6b43f..9650db24 100644 --- a/tests/test_benchmark_evidence.py +++ b/tests/test_benchmark_evidence.py @@ -1,1282 +1,1282 @@ -import hashlib -import json -import re -import struct -from copy import deepcopy -from pathlib import Path -from xml.etree import ElementTree - -import pytest - -from eval import metrics -from eval import grounded as grounded_eval -from eval.benchmark import ( - SCHEMA, - CANONICAL_TOKEN_BUDGETS, - LONGMEMEVAL_V2_CANONICAL_PROFILE_TEMPLATE, - canonical_benchmark_config, - count_tokens, - fixed_budget_curve, - paired_bootstrap_ci, - redact_command, - redact_public_record, - main, - question_record, - report_envelope, - stratified_bootstrap_ci, - validate_report, - write_canonical_artifact, -) -from eval.chunking_eval import compare as compare_chunking, load as load_chunking -from eval.harness import load_dataset as load_performance_dataset -from eval.performance import run as run_performance - - +import hashlib +import json +import re +import struct +from copy import deepcopy +from pathlib import Path +from xml.etree import ElementTree + +import pytest + +from eval import metrics +from eval import grounded as grounded_eval +from eval.benchmark import ( + SCHEMA, + CANONICAL_TOKEN_BUDGETS, + LONGMEMEVAL_V2_CANONICAL_PROFILE_TEMPLATE, + canonical_benchmark_config, + count_tokens, + fixed_budget_curve, + paired_bootstrap_ci, + redact_command, + redact_public_record, + main, + question_record, + report_envelope, + stratified_bootstrap_ci, + validate_report, + write_canonical_artifact, +) +from eval.chunking_eval import compare as compare_chunking, load as load_chunking +from eval.harness import load_dataset as load_performance_dataset +from eval.performance import run as run_performance + + ROOT = Path(__file__).resolve().parents[1] -PUBLIC_OFFLINE_ARTIFACT = "offline-fixtures-v73.json" -PUBLIC_OFFLINE_SHA = "aa7ed9c141afcc82fc2a05b63ed9037842f2ea8372667f3142cf9bb795833988" - - -@pytest.fixture(scope="module") -def offline_release_evidence(): - """Run the exact small offline commands that back the public documentation.""" - longdoc = ROOT / "eval" / "datasets" / "longdoc.jsonl" - codemem = ROOT / "eval" / "datasets" / "codemem.jsonl" - return { - "chunking": compare_chunking( - load_chunking(str(longdoc)), k=5, embed_model=None - ), - "performance": run_performance( - load_performance_dataset(str(codemem)), k=5, iterations=10 - ), - "grounded": grounded_eval.run(), - } - - -def test_public_facing_docs_do_not_use_em_dashes(): - """Published prose uses straightforward punctuation that renders consistently.""" - public_files = [ - *(ROOT / name for name in ("README.md", "BENCHMARKS.md", "CHANGELOG.md", "SECURITY.md")), - *(ROOT / "docs").rglob("*.md"), - *(ROOT / "docs" / "images").glob("*.svg"), - *(ROOT / "skills" / "engraphis-memory").rglob("*.md"), - ] - offenders = [ - path.relative_to(ROOT).as_posix() - for path in public_files - if "—" in path.read_text(encoding="utf-8") - ] - - assert not offenders, f"Public-facing files still contain em dashes: {offenders}" - - -class CharacterTokenizer: - def encode(self, text): - return list(text) - - -def test_public_record_redaction_omits_raw_payloads_and_content_fingerprints(): - record = redact_public_record({ - "question_id": "q1", - "query": "private query", - "answer_variants": ["private answer"], - "model_output": "private completion", - "context": "private context", - "retrieved_context": "private retrieved context", - "prompt": "private prompt", - "input": "private input", - "conversation": ["private conversation"], - "history": ["private history"], - "tool_calls": [{"arguments": "private tool input"}], - }) - - assert record == {"question_id": "q1"} - - - -def _committed_evidence() -> dict: - """Load the COMMITTED registry artifact — the publication source of truth that the - README/BENCHMARKS/SVG prose was written from. - - Prose tests must interpolate values from this artifact, not from a fresh evaluator - run. Timed latency aggregates are machine-dependent, while the context, payload, - question-count, and quality aggregates used by the publication contract are - deterministic and compared exactly in ``test_public_numeric_evidence_registry_is_complete_and_live``. - """ - artifact = json.loads( +PUBLIC_OFFLINE_ARTIFACT = "offline-fixtures-v76.json" +PUBLIC_OFFLINE_SHA = "2fb5ce5b2cbc21541f7ae9cad5d2ef615014c0e86f9881f00621d00986422ba8" + + +@pytest.fixture(scope="module") +def offline_release_evidence(): + """Run the exact small offline commands that back the public documentation.""" + longdoc = ROOT / "eval" / "datasets" / "longdoc.jsonl" + codemem = ROOT / "eval" / "datasets" / "codemem.jsonl" + return { + "chunking": compare_chunking( + load_chunking(str(longdoc)), k=5, embed_model=None + ), + "performance": run_performance( + load_performance_dataset(str(codemem)), k=5, iterations=10 + ), + "grounded": grounded_eval.run(), + } + + +def test_public_facing_docs_do_not_use_em_dashes(): + """Published prose uses straightforward punctuation that renders consistently.""" + public_files = [ + *(ROOT / name for name in ("README.md", "BENCHMARKS.md", "CHANGELOG.md", "SECURITY.md")), + *(ROOT / "docs").rglob("*.md"), + *(ROOT / "docs" / "images").glob("*.svg"), + *(ROOT / "skills" / "engraphis-memory").rglob("*.md"), + ] + offenders = [ + path.relative_to(ROOT).as_posix() + for path in public_files + if "—" in path.read_text(encoding="utf-8") + ] + + assert not offenders, f"Public-facing files still contain em dashes: {offenders}" + + +class CharacterTokenizer: + def encode(self, text): + return list(text) + + +def test_public_record_redaction_omits_raw_payloads_and_content_fingerprints(): + record = redact_public_record({ + "question_id": "q1", + "query": "private query", + "answer_variants": ["private answer"], + "model_output": "private completion", + "context": "private context", + "retrieved_context": "private retrieved context", + "prompt": "private prompt", + "input": "private input", + "conversation": ["private conversation"], + "history": ["private history"], + "tool_calls": [{"arguments": "private tool input"}], + }) + + assert record == {"question_id": "q1"} + + + +def _committed_evidence() -> dict: + """Load the COMMITTED registry artifact — the publication source of truth that the + README/BENCHMARKS/SVG prose was written from. + + Prose tests must interpolate values from this artifact, not from a fresh evaluator + run. Timed latency aggregates are machine-dependent, while the context, payload, + question-count, and quality aggregates used by the publication contract are + deterministic and compared exactly in ``test_public_numeric_evidence_registry_is_complete_and_live``. + """ + artifact = json.loads( (ROOT / "docs" / "benchmark-evidence" / PUBLIC_OFFLINE_ARTIFACT).read_text( - encoding="utf-8" - ) - ) - return { - "chunking": artifact["runs"][0]["result"], - "performance": artifact["runs"][1]["result"], - "grounded": artifact["runs"][2]["result"], - } - -def test_readme_distinguishes_every_registered_token_context_measurement(): - """Public token-efficiency copy preserves each registered metric boundary. - - Values are interpolated from the COMMITTED registry artifact — the publication - source of truth — so prose cannot drift from the evidence it cites. - """ - committed = _committed_evidence() - readme = (ROOT / "README.md").read_text(encoding="utf-8") - chunking = committed["chunking"] - whole = chunking["whole"] - chunked = chunking["chunked"] - performance = committed["performance"] - context_full = performance["full_serialized_payload_tokens"] - context_compact = performance["compact_serialized_payload_tokens"] - payload_samples = performance["questions"] - timed_recalls = performance["timed_recalls"] - - for evidence in ( - "## Measured token and context savings", - "See benchmark details and reproduce the results", - "### Measurement details and reproducibility", - f"{whole['mean_context_tokens']:.1f}** tokens → structure-aware chunks: " - f"**{chunked['mean_context_tokens']:.1f}** tokens", - f"{chunking['context_reduction_pct']:.1f}% lower", - f"{whole['mean_evidence_tokens']:.1f}** tokens → chunks: " - f"**{chunked['mean_evidence_tokens']:.1f}** tokens", - "73.9% lower", - f"{context_full:,}** `engraphis.regex.v1` tokens → " - f"compact proxy: **{context_compact:,}** tokens", - f"{performance['saved_serialized_payload_tokens']:,} proxy tokens avoided", - f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower", - f"{payload_samples} payload samples; {timed_recalls} timed recalls", - f"1,500** tokens; observed mean: **{performance['mean_context_tokens']:.2f}**; " - f"observed maximum: **{performance['max_context_tokens']}**", - "does **not** serialize the MCP envelope", - "not an MCP transport response", - "must not be added together", - "not a storage-reduction claim", + encoding="utf-8" + ) + ) + return { + "chunking": artifact["runs"][0]["result"], + "performance": artifact["runs"][1]["result"], + "grounded": artifact["runs"][2]["result"], + } + +def test_readme_distinguishes_every_registered_token_context_measurement(): + """Public token-efficiency copy preserves each registered metric boundary. + + Values are interpolated from the COMMITTED registry artifact — the publication + source of truth — so prose cannot drift from the evidence it cites. + """ + committed = _committed_evidence() + readme = (ROOT / "README.md").read_text(encoding="utf-8") + chunking = committed["chunking"] + whole = chunking["whole"] + chunked = chunking["chunked"] + performance = committed["performance"] + context_full = performance["full_serialized_payload_tokens"] + context_compact = performance["compact_serialized_payload_tokens"] + payload_samples = performance["questions"] + timed_recalls = performance["timed_recalls"] + + for evidence in ( + "## Measured token and context savings", + "See benchmark details and reproduce the results", + "### Measurement details and reproducibility", + f"{whole['mean_context_tokens']:.1f}** tokens → structure-aware chunks: " + f"**{chunked['mean_context_tokens']:.1f}** tokens", + f"{chunking['context_reduction_pct']:.1f}% lower", + f"{whole['mean_evidence_tokens']:.1f}** tokens → chunks: " + f"**{chunked['mean_evidence_tokens']:.1f}** tokens", + "73.9% lower", + f"{context_full:,}** `engraphis.regex.v1` tokens → " + f"compact proxy: **{context_compact:,}** tokens", + f"{performance['saved_serialized_payload_tokens']:,} proxy tokens avoided", + f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower", + f"{payload_samples} payload samples; {timed_recalls} timed recalls", + f"1,500** tokens; observed mean: **{performance['mean_context_tokens']:.2f}**; " + f"observed maximum: **{performance['max_context_tokens']}**", + "does **not** serialize the MCP envelope", + "not an MCP transport response", + "must not be added together", + "not a storage-reduction claim", PUBLIC_OFFLINE_ARTIFACT, - "offline-chunking", - "offline-performance", + "offline-chunking", + "offline-performance", PUBLIC_OFFLINE_SHA, - "There is no universal memory-count", - "python -m eval.vector_scale", - 'vector_backend="sqlite-vec"', - ): - assert evidence in readme - - for unsupported in ( - "49,915,394", - "891,857", - "98.2133%", - "Repeated-memory consolidation fixture", - "1,883** total agent-facing tokens", - "3.1% higher", - ): - assert unsupported not in readme - - - -def test_public_docs_scope_external_numbers_and_withhold_historical_claims(): - """The offline registry is scoped while separate diagnostics remain artifact-bound.""" - readme = (ROOT / "README.md").read_text(encoding="utf-8") - benchmarks = (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") - expansion = (ROOT / "docs" / "BENCHMARK_EXPANSION_RESULTS.md").read_text( - encoding="utf-8" - ) - additional = (ROOT / "docs" / "ADDITIONAL_BENCHMARK_DIAGNOSTICS.md").read_text( - encoding="utf-8" - ) - security = (ROOT / "SECURITY.md").read_text(encoding="utf-8") - readme_normalized = " ".join(readme.split()) - benchmarks_normalized = " ".join(benchmarks.split()) - additional_normalized = " ".join(additional.split()) - - assert "See benchmark details and reproduce the results" in readme - assert "offline fixture registry intentionally excludes external" in readme_normalized - assert "Completed retrieval-only diagnostics are published separately" in readme_normalized - assert "absence from this registry" in benchmarks_normalized - assert "LoCoMo and LongMemEval retrieval diagnostics are retained as separate public-safe artifacts" in benchmarks_normalized - assert "Mem2ActBench's declared small retrieval diagnostic is complete" in benchmarks_normalized - assert "LongMemEval 4,096-token context experiment | COMPLETE" in expansion - assert "+24.03 percentage points" in expansion - assert "source-preparation metadata" in expansion - assert "public source lock records 20 preparation exclusions" in additional_normalized - assert "Exact vector scale envelope" in benchmarks - assert "python -m eval.redteam_poisoning" in security - - for stale in ( - "model-dependent, consolidation, productivity, and latency results remain unpublished", - "private diagnostic; it is not an official benchmark-harness or public evidence artifact", - "withholds their case counts, retrieval scores", - ): - assert stale not in readme - assert stale not in benchmarks - - for unsupported in ( - "49,915,394", - "891,857", - "98.2133%", - "0.6045", - "0.6625", - "0.1259", - "0.5100", - "20.666 ms", - ): - assert unsupported not in readme - assert unsupported not in benchmarks - - for supporting_detail in ( - "### Choose a vector backend for your corpus", - "python -m eval.redteam_poisoning", - "[local and hosted plans]", - ): - assert supporting_detail not in readme - - -def test_readme_makes_agent_benefits_and_visual_evidence_scannable(): - """The public overview and its visual evidence must stay wired to real assets.""" - readme = (ROOT / "README.md").read_text(encoding="utf-8") - - for evidence in ( - "## What Engraphis gives an agent", - "Remember a project across sessions", - "Avoid confident guesses", - "Avoid dragging the whole project into every prompt", - "docs/images/knowledge-graph.png", - "docs/images/context-efficiency.svg", - "Less repeated history means more room for the task, tools, and useful evidence", - ): - assert evidence in readme - - for removed in ( - "### See the behavior in reproducible fixtures", - "docs/images/evidence-backed-agent-examples.svg", - "Run `python -m eval.chunking_eval` and `python -m eval.grounded`", - ): - assert removed not in readme - - for filename in ( - "engraphis-benefit-flow.svg", - "engraphis-benefit-flow.png", - "context-efficiency.svg", - "context-efficiency.png", - "evidence-backed-agent-examples.svg", - "evidence-backed-agent-examples.png", - ): - assert (ROOT / "docs" / "images" / filename).is_file() - - -def test_readme_visual_pngs_match_their_svg_canvas(): - """README image exports must not carry hidden screenshot padding.""" - image_dir = ROOT / "docs" / "images" - - for stem in ( - "engraphis-benefit-flow", - "evidence-backed-agent-examples", - "context-efficiency", - ): - svg = ElementTree.parse(image_dir / f"{stem}.svg").getroot() - expected = (int(svg.attrib["width"]), int(svg.attrib["height"])) - png_header = (image_dir / f"{stem}.png").read_bytes()[:24] - - assert png_header[:8] == b"\x89PNG\r\n\x1a\n" - assert struct.unpack(">II", png_header[16:24]) == expected - - -def test_example_visual_uses_the_checked_in_offline_fixture_results( - offline_release_evidence, -): - """The examples stay tied to executable fixtures and their public artifact.""" - chunking = offline_release_evidence["chunking"] - whole = chunking["reports"]["whole"] - chunked = chunking["reports"]["chunked"] - grounded = offline_release_evidence["grounded"] - visual = ( - ROOT / "docs" / "images" / "evidence-backed-agent-examples.svg" - ).read_text(encoding="utf-8") - - assert chunking["context_reduction_pct"] == 71.1 - result = ( - f"{whole['mean_context_tokens']:.1f} → " - f"{chunked['mean_context_tokens']:.1f} tokens" - ) - assert result in visual - assert grounded == { - "answer_rate": 1.0, - "abstain_rate": 1.0, - "accuracy": 1.0, - "grounded_hits": 5, - "abstain_hits": 6, - "quarantine_hits": 1, - "n_quarantine": 1, - "n_answerable": 5, - "n_unanswerable": 6, - } - assert "5/5 answerable questions" in visual - assert "6/6 off-topic questions" in visual + "There is no universal memory-count", + "python -m eval.vector_scale", + 'vector_backend="sqlite-vec"', + ): + assert evidence in readme + + for unsupported in ( + "49,915,394", + "891,857", + "98.2133%", + "Repeated-memory consolidation fixture", + "1,883** total agent-facing tokens", + "3.1% higher", + ): + assert unsupported not in readme + + + +def test_public_docs_scope_external_numbers_and_withhold_historical_claims(): + """The offline registry is scoped while separate diagnostics remain artifact-bound.""" + readme = (ROOT / "README.md").read_text(encoding="utf-8") + benchmarks = (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") + expansion = (ROOT / "docs" / "BENCHMARK_EXPANSION_RESULTS.md").read_text( + encoding="utf-8" + ) + additional = (ROOT / "docs" / "ADDITIONAL_BENCHMARK_DIAGNOSTICS.md").read_text( + encoding="utf-8" + ) + security = (ROOT / "SECURITY.md").read_text(encoding="utf-8") + readme_normalized = " ".join(readme.split()) + benchmarks_normalized = " ".join(benchmarks.split()) + additional_normalized = " ".join(additional.split()) + + assert "See benchmark details and reproduce the results" in readme + assert "offline fixture registry intentionally excludes external" in readme_normalized + assert "Completed retrieval-only diagnostics are published separately" in readme_normalized + assert "absence from this registry" in benchmarks_normalized + assert "LoCoMo and LongMemEval retrieval diagnostics are retained as separate public-safe artifacts" in benchmarks_normalized + assert "Mem2ActBench's declared small retrieval diagnostic is complete" in benchmarks_normalized + assert "LongMemEval 4,096-token context experiment | COMPLETE" in expansion + assert "+24.03 percentage points" in expansion + assert "source-preparation metadata" in expansion + assert "public source lock records 20 preparation exclusions" in additional_normalized + assert "Exact vector scale envelope" in benchmarks + assert "python -m eval.redteam_poisoning" in security + + for stale in ( + "model-dependent, consolidation, productivity, and latency results remain unpublished", + "private diagnostic; it is not an official benchmark-harness or public evidence artifact", + "withholds their case counts, retrieval scores", + ): + assert stale not in readme + assert stale not in benchmarks + + for unsupported in ( + "49,915,394", + "891,857", + "98.2133%", + "0.6045", + "0.6625", + "0.1259", + "0.5100", + "20.666 ms", + ): + assert unsupported not in readme + assert unsupported not in benchmarks + + for supporting_detail in ( + "### Choose a vector backend for your corpus", + "python -m eval.redteam_poisoning", + "[local and hosted plans]", + ): + assert supporting_detail not in readme + + +def test_readme_makes_agent_benefits_and_visual_evidence_scannable(): + """The public overview and its visual evidence must stay wired to real assets.""" + readme = (ROOT / "README.md").read_text(encoding="utf-8") + + for evidence in ( + "## What Engraphis gives an agent", + "Remember a project across sessions", + "Avoid confident guesses", + "Avoid dragging the whole project into every prompt", + "docs/images/knowledge-graph.png", + "docs/images/context-efficiency.svg", + "Less repeated history means more room for the task, tools, and useful evidence", + ): + assert evidence in readme + + for removed in ( + "### See the behavior in reproducible fixtures", + "docs/images/evidence-backed-agent-examples.svg", + "Run `python -m eval.chunking_eval` and `python -m eval.grounded`", + ): + assert removed not in readme + + for filename in ( + "engraphis-benefit-flow.svg", + "engraphis-benefit-flow.png", + "context-efficiency.svg", + "context-efficiency.png", + "evidence-backed-agent-examples.svg", + "evidence-backed-agent-examples.png", + ): + assert (ROOT / "docs" / "images" / filename).is_file() + + +def test_readme_visual_pngs_match_their_svg_canvas(): + """README image exports must not carry hidden screenshot padding.""" + image_dir = ROOT / "docs" / "images" + + for stem in ( + "engraphis-benefit-flow", + "evidence-backed-agent-examples", + "context-efficiency", + ): + svg = ElementTree.parse(image_dir / f"{stem}.svg").getroot() + expected = (int(svg.attrib["width"]), int(svg.attrib["height"])) + png_header = (image_dir / f"{stem}.png").read_bytes()[:24] + + assert png_header[:8] == b"\x89PNG\r\n\x1a\n" + assert struct.unpack(">II", png_header[16:24]) == expected + + +def test_example_visual_uses_the_checked_in_offline_fixture_results( + offline_release_evidence, +): + """The examples stay tied to executable fixtures and their public artifact.""" + chunking = offline_release_evidence["chunking"] + whole = chunking["reports"]["whole"] + chunked = chunking["reports"]["chunked"] + grounded = offline_release_evidence["grounded"] + visual = ( + ROOT / "docs" / "images" / "evidence-backed-agent-examples.svg" + ).read_text(encoding="utf-8") + + assert chunking["context_reduction_pct"] == 71.1 + result = ( + f"{whole['mean_context_tokens']:.1f} → " + f"{chunked['mean_context_tokens']:.1f} tokens" + ) + assert result in visual + assert grounded == { + "answer_rate": 1.0, + "abstain_rate": 1.0, + "accuracy": 1.0, + "grounded_hits": 5, + "abstain_hits": 6, + "quarantine_hits": 1, + "n_quarantine": 1, + "n_answerable": 5, + "n_unanswerable": 6, + } + assert "5/5 answerable questions" in visual + assert "6/6 off-topic questions" in visual assert PUBLIC_OFFLINE_SHA in visual - - -def test_context_savings_visual_uses_only_registered_measurements(): - """The headline chart contains only registered values and explicit scope labels. - - Values are interpolated from the COMMITTED registry artifact — the publication - source of truth — so chart text cannot drift from the evidence it cites. - """ - visual = (ROOT / "docs" / "images" / "context-efficiency.svg").read_text( - encoding="utf-8" - ) - committed = _committed_evidence() - chunking = committed["chunking"] - whole = chunking["whole"] - chunked = chunking["chunked"] - performance = committed["performance"] - context_full = performance["full_serialized_payload_tokens"] - context_compact = performance["compact_serialized_payload_tokens"] - payload_samples = performance["questions"] - timed_recalls = performance["timed_recalls"] - - for evidence in ( - "Measured context and retrieval boundaries", - "CONTEXT BOUNDARIES", - "QUALITY SCOPES", - "PENDING EVALUATION TRACKS", - "Whole documents", - f"{whole['mean_context_tokens']:.1f} tokens", - "Structure-aware chunks", - f"{chunked['mean_context_tokens']:.1f} tokens", - f"{chunking['context_reduction_pct']:.1f}% lower", - "Smallest evidence:", - "Serialized JSON-shape payload proxy", - f"{payload_samples:,} payload samples / {timed_recalls:,} timed recalls", - "Full JSON-shape proxy", - f"{context_full:,} tokens", - "Compact JSON-shape proxy", - f"{context_compact:,} tokens", - f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower", - "Retrieved candidate quality", - "Packed context", - "Recall@5 1.000 / hit@5 1.000 / answer tokens 1.000", - "MCP transport not measured", - "JSON proxy only", - "Pinned LoCoMo and LongMemEval artifacts with answer evaluators", - ): - assert evidence in visual - - svg = ElementTree.fromstring(visual) - namespace = "{http://www.w3.org/2000/svg}" - # Numeric source text must be rendered by SVG, not hidden beside a stale bitmap. - assert not svg.findall(f".//{namespace}image") - visible_text = {node.text for node in svg.iter(f"{namespace}text")} - assert f"{context_compact:,} tokens" in visible_text - assert f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower" in visible_text - assert f"Mean {performance['mean_context_tokens']:.2f} / max {performance['max_context_tokens']:,} tokens" in visual - assert "Recall@5 1.000 / hit@5 1.000 / answer tokens 1.000" in visual - assert "MCP transport not measured" in visible_text - - for unsupported in ( - "Public evidence is checksum-bound", + + +def test_context_savings_visual_uses_only_registered_measurements(): + """The headline chart contains only registered values and explicit scope labels. + + Values are interpolated from the COMMITTED registry artifact — the publication + source of truth — so chart text cannot drift from the evidence it cites. + """ + visual = (ROOT / "docs" / "images" / "context-efficiency.svg").read_text( + encoding="utf-8" + ) + committed = _committed_evidence() + chunking = committed["chunking"] + whole = chunking["whole"] + chunked = chunking["chunked"] + performance = committed["performance"] + context_full = performance["full_serialized_payload_tokens"] + context_compact = performance["compact_serialized_payload_tokens"] + payload_samples = performance["questions"] + timed_recalls = performance["timed_recalls"] + + for evidence in ( + "Measured context and retrieval boundaries", + "CONTEXT BOUNDARIES", + "QUALITY SCOPES", + "PENDING EVALUATION TRACKS", + "Whole documents", + f"{whole['mean_context_tokens']:.1f} tokens", + "Structure-aware chunks", + f"{chunked['mean_context_tokens']:.1f} tokens", + f"{chunking['context_reduction_pct']:.1f}% lower", + "Smallest evidence:", + "Serialized JSON-shape payload proxy", + f"{payload_samples:,} payload samples / {timed_recalls:,} timed recalls", + "Full JSON-shape proxy", + f"{context_full:,} tokens", + "Compact JSON-shape proxy", + f"{context_compact:,} tokens", + f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower", + "Retrieved candidate quality", + "Packed context", + "Recall@5 1.000 / hit@5 1.000 / answer tokens 1.000", + "MCP transport not measured", + "JSON proxy only", + "Pinned LoCoMo and LongMemEval artifacts with answer evaluators", + ): + assert evidence in visual + + svg = ElementTree.fromstring(visual) + namespace = "{http://www.w3.org/2000/svg}" + # Numeric source text must be rendered by SVG, not hidden beside a stale bitmap. + assert not svg.findall(f".//{namespace}image") + visible_text = {node.text for node in svg.iter(f"{namespace}text")} + assert f"{context_compact:,} tokens" in visible_text + assert f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower" in visible_text + assert f"Mean {performance['mean_context_tokens']:.2f} / max {performance['max_context_tokens']:,} tokens" in visual + assert "Recall@5 1.000 / hit@5 1.000 / answer tokens 1.000" in visual + assert "MCP transport not measured" in visible_text + + for unsupported in ( + "Public evidence is checksum-bound", PUBLIC_OFFLINE_ARTIFACT, - "No external or model-dependent number is published without the same evidence", - "Evidence pending", - "No external or model-dependent number is published", - "808.8", - "218.4", - "17,172", - "7,663", - "Repeated memories · 230 tokens", - "47.8% less", - "53× more evidence", - "97.72% less total", - "87.7 average · 106 max", - ): - assert unsupported not in visual - - text_sizes = { - float(value) - for value in re.findall(r'font-size="([^"]+)"', visual) - } - assert {12.5, 13.2, 14.3, 17.4, 18.7, 20.0, 24.0, 33.0} <= text_sizes - - -def test_public_numeric_evidence_registry_is_complete_and_live( - offline_release_evidence, -): - """Every retained public aggregate resolves to one checksum-bound live run.""" - artifact_path = ( + "No external or model-dependent number is published without the same evidence", + "Evidence pending", + "No external or model-dependent number is published", + "808.8", + "218.4", + "17,172", + "7,663", + "Repeated memories · 230 tokens", + "47.8% less", + "53× more evidence", + "97.72% less total", + "87.7 average · 106 max", + ): + assert unsupported not in visual + + text_sizes = { + float(value) + for value in re.findall(r'font-size="([^"]+)"', visual) + } + assert {12.5, 13.2, 14.3, 17.4, 18.7, 20.0, 24.0, 33.0} <= text_sizes + + +def test_public_numeric_evidence_registry_is_complete_and_live( + offline_release_evidence, +): + """Every retained public aggregate resolves to one checksum-bound live run.""" + artifact_path = ( ROOT / "docs" / "benchmark-evidence" / PUBLIC_OFFLINE_ARTIFACT - ) - sidecar_path = artifact_path.with_suffix(".json.sha256") - artifact_bytes = artifact_path.read_bytes() - artifact_sha = hashlib.sha256(artifact_bytes).hexdigest() + ) + sidecar_path = artifact_path.with_suffix(".json.sha256") + artifact_bytes = artifact_path.read_bytes() + artifact_sha = hashlib.sha256(artifact_bytes).hexdigest() expected_sha = PUBLIC_OFFLINE_SHA - - assert artifact_sha == expected_sha - assert sidecar_path.read_text(encoding="ascii") == ( - f"{expected_sha} {artifact_path.name}\n" - ) - artifact = json.loads(artifact_bytes) - assert artifact["schema"] == "engraphis-public-offline-fixtures/v1" - assert not any(artifact["privacy"].values()) - - file_hashes = artifact["suite"]["files"] - assert file_hashes == { - path: hashlib.sha256((ROOT / path).read_bytes()).hexdigest() - for path in sorted(file_hashes) - } - suite_manifest = json.dumps( - file_hashes, sort_keys=True, separators=(",", ":") - ).encode() + + assert artifact_sha == expected_sha + assert sidecar_path.read_text(encoding="ascii") == ( + f"{expected_sha} {artifact_path.name}\n" + ) + artifact = json.loads(artifact_bytes) + assert artifact["schema"] == "engraphis-public-offline-fixtures/v1" + assert not any(artifact["privacy"].values()) + + file_hashes = artifact["suite"]["files"] + assert file_hashes == { + path: hashlib.sha256((ROOT / path).read_bytes()).hexdigest() + for path in sorted(file_hashes) + } + suite_manifest = json.dumps( + file_hashes, sort_keys=True, separators=(",", ":") + ).encode() assert hashlib.sha256(suite_manifest).hexdigest() == artifact["suite"]["digest"] assert artifact["suite"]["digest"] in (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") - - runs = {run["id"]: run for run in artifact["runs"]} - assert set(runs) == { - "offline-chunking", - "offline-performance", - "offline-grounded", - } - for run in runs.values(): - assert hashlib.sha256(run["command"].encode()).hexdigest() == run["config_digest"] - - chunking = offline_release_evidence["chunking"] - chunking_result = runs["offline-chunking"]["result"] - for mode in ("whole", "chunked"): - live = chunking["reports"][mode] - recorded = chunking_result[mode] - assert recorded["memories"] == live["memories_stored"] - assert recorded["recall_at_k"] == live["recall_at_k"] - assert recorded["mean_context_tokens"] == live["mean_context_tokens"] - assert recorded["mean_evidence_tokens"] == live["mean_evidence_tokens"] - assert recorded["max_stored_tokens"] == live["max_stored_tokens"] - assert chunking_result["context_reduction_pct"] == chunking["context_reduction_pct"] - - performance = offline_release_evidence["performance"] - performance_result = runs["offline-performance"]["result"] - assert performance_result["questions"] == performance["corpus"]["questions"] - assert performance_result["timed_recalls"] == performance["run"]["timed_recalls"] - assert performance_result["recall_at_k"] == performance["quality"]["recall_at_k"] - assert performance_result["hit_at_k"] == performance["quality"]["hit_at_k"] - assert ( - performance_result["answer_token_recall"] - == performance["quality"]["answer_token_recall"] - ) - - # These values are deterministic fixture aggregates, not wall-clock timing - # observations. Approximate comparisons would let serializer or count drift - # pass the publication contract unnoticed. - assert performance_result["mean_context_tokens"] == performance["context"]["mean_tokens"] - assert performance_result["max_context_tokens"] == performance["context"]["max_tokens"] - assert ( - performance_result["full_serialized_payload_tokens"] - == performance["context"]["full_serialized_payload_tokens"] - ) - assert ( - performance_result["compact_serialized_payload_tokens"] - == performance["context"]["compact_serialized_payload_tokens"] - ) - assert ( - performance_result["saved_serialized_payload_tokens"] - == performance["context"]["saved_serialized_payload_tokens"] - ) - assert ( - performance_result["serialized_payload_savings_ratio"] - == performance["context"]["serialized_payload_savings_ratio"] - ) - - grounded = offline_release_evidence["grounded"] - grounded_result = runs["offline-grounded"]["result"] - assert grounded_result == { - "answerable": grounded["n_answerable"], - "grounded": grounded["grounded_hits"], - "off_topic": grounded["n_unanswerable"], - "quarantined": grounded["n_quarantine"], - "abstained": grounded["abstain_hits"], - "quarantine_hits": grounded["quarantine_hits"], - "decision_accuracy": grounded["accuracy"], - } - - surfaces = ( - ROOT / "README.md", - ROOT / "BENCHMARKS.md", - ROOT / "docs" / "images" / "context-efficiency.svg", - ROOT / "docs" / "images" / "evidence-backed-agent-examples.svg", - ) - for surface in surfaces: - assert expected_sha in surface.read_text(encoding="utf-8") - - claimed_ids = set( - re.findall( - r"offline-(?:chunking|performance|grounded)", - "\n".join(path.read_text(encoding="utf-8") for path in surfaces), - ) - ) - assert claimed_ids == set(runs) - - -def test_benchmark_guide_tracks_the_live_offline_evaluators(): - """Method prose must change whenever its executable offline evidence changes. - - Values are interpolated from the COMMITTED registry artifact — the publication - source of truth — so guide text cannot drift from the evidence it cites. - """ - benchmarks = (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") - normalized = " ".join(benchmarks.split()) - committed = _committed_evidence() - chunking = committed["chunking"] - whole = chunking["whole"] - chunked = chunking["chunked"] - performance = committed["performance"] - payload_samples = performance["questions"] - - for evidence in ( - f"falls from {whole['mean_context_tokens']:.1f} to " - f"{chunked['mean_context_tokens']:.1f} tokens", - f"{whole['mean_context_tokens'] - chunked['mean_context_tokens']:.1f} fewer, " - f"{chunking['context_reduction_pct']:.1f}% lower", - f"falls from {whole['mean_evidence_tokens']:.1f} to " - f"{chunked['mean_evidence_tokens']:.1f} tokens", - "Payload proxies are sampled once per question", - "not serialized MCP envelopes or transport responses", - f"{payload_samples} payload samples total **" - f"{performance['full_serialized_payload_tokens']:,}** full-proxy", - f"versus **{performance['compact_serialized_payload_tokens']:,}** compact-proxy tokens", - f"avoiding **{performance['saved_serialized_payload_tokens']:,}** proxy tokens", - f"**{100 * performance['serialized_payload_savings_ratio']:.2f}% lower**", - f"averages **{performance['mean_context_tokens']:.2f}** tokens and reaches " - f"**{performance['max_context_tokens']}**", - ): - assert evidence in normalized - - - -def _complete_canonical_report(dataset, config): - """Minimal but fully auditable canonical envelope for validator coverage.""" - profile = config["canonical_profile"] - tokenizer_identity = ( - f"{profile['reader']['model']}@{profile['reader']['revision']}" - ) - record = question_record( - "q1", category="state", context_tokens=3, latency_ms=1.25, - retrieved_ids=["support"], supporting_ids=["support"], - recall_at_1=1.0, recall_at_5=1.0, recall_at_10=1.0, - mrr_at_1=1.0, mrr_at_5=1.0, mrr_at_10=1.0, - ndcg_at_1=1.0, ndcg_at_5=1.0, ndcg_at_10=1.0, - usage={ - "budget_tokens": config.get("token_budget") or 3, - "context_tokens": 3, - "token_counter": tokenizer_identity, - }, - ) - record["context_token_method"] = "pinned_reader_content_tokenizer" - record["context_tokenizer_identity"] = tokenizer_identity - rank_metrics = { - f"{metric}_at_{depth}": 1.0 - for metric in ("recall", "mrr", "ndcg") - for depth in (1, 5, 10) - } - curve_record = { - "question_id": "q1", - "excluded": False, - "context_tokens": 3, - "context_token_method": "pinned_reader_content_tokenizer", - "context_tokenizer_identity": tokenizer_identity, - "retrieved_ids": ["support"], - "supporting_ids": ["support"], - **rank_metrics, - } - report = report_envelope( - suite="fixture", dataset_path=dataset, config=config, records=[record], - metrics={ - **rank_metrics, - "confidence_intervals": { - field: { - "point": 1.0, - "low": 1.0, - "high": 1.0, - "n": 1, - "seed": 20260729, - "iterations": 1, - "strata_key": "category", - } - for field in rank_metrics - }, - "paired_bootstrap": { - "available": False, - "reason": "baseline_records_not_supplied", - "n": 0, - "delta": None, - "low": None, - "high": None, - "iterations": 1, - }, - "grounded_f1": {"available": False, "reason": "not_measured"}, - "abstention_f1": {"available": False, "reason": "not_measured"}, - "fixed_budget_curve": { - "available": True, - "rows": [{ - "token_budget": budget, - "status": "measured", - "n_total": 1, - "n_scored": 1, - "records": [dict(curve_record)], - **rank_metrics, - } for budget in CANONICAL_TOKEN_BUDGETS], - }, - }, - git_commit="a" * 40, - ) - report["system"]["git_dirty"] = False - report["models"] = {"embedder": { - "name": "FixtureEmbedder", - "model_id": profile["embedding"]["model"], - "revision": profile["embedding"]["revision"], - "sha256": "b" * 64, - }} - report["protocol"]["complete_dataset"] = True - report["protocol"]["source_questions"] = len(report["records"]) - return report - - -def test_metrics_cover_rank_sensitive_retrieval_quality(): - retrieved = ["noise", "evidence-a", "evidence-b"] - supporting = ["evidence-a", "evidence-b"] - assert metrics.mrr_at_k(retrieved, supporting, 3) == 0.5 - assert metrics.ndcg_at_k(retrieved, supporting, 3) > 0.6 - assert metrics.recall_at_k(retrieved[:1], supporting) == 0.0 - assert metrics.hit_at_k(retrieved[:1], supporting) == 0.0 - bundle = metrics.retrieval_metrics_at_depths(retrieved, supporting) - assert bundle["recall_at_1"] == 0.0 - assert bundle["recall_at_5"] == 1.0 - assert bundle["mrr_at_5"] == 0.5 - - -def test_envelope_hashes_dataset_config_and_retains_exclusions(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - excluded = {"question_id": "q2", "reason": "no_gold_evidence", "detail": ""} - records = [ - question_record("q1", category="state", supporting_ids=["m1"]), - question_record("q2", category="abstention", excluded=excluded), - ] - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"k": 5}, records=records, - metrics={"recall": 1.0}, git_commit="abc123", - ) - assert report["schema"] == SCHEMA - assert report["suite"]["sha256"] - assert report["system"]["config_sha256"] - assert report["protocol"] == { - "command": ["in_process"], - "config": {"k": 5}, - "token_accounting": { - "identity": "unspecified", - "revision": None, - "scope": "unspecified", - "method": "unspecified", - }, - "n_total": 2, - "n_scored": 1, - } - assert report["exclusions"] == [{ - "question_id": "q2", - "reason": "no_gold_evidence", - }] - assert json.loads(json.dumps(report))["schema"] == SCHEMA - - -def test_envelope_redacts_top_level_exclusion_detail(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"k": 5}, records=[], - exclusions=[{ - "question_id": "q1", "reason": "invalid", "detail": "private prompt text", - }], - ) - - assert report["exclusions"] == [{ - "question_id": "q1", - "reason": "invalid", - }] - - -def test_command_provenance_redacts_explicit_credential_arguments(): - assert redact_command([ - "python", "-m", "runner", "--api-key", "do-not-publish", "--token=value", - ]) == [ - "python", "-m", "runner", "--api-key", "", "--token", "", - ] - - -def test_command_provenance_redacts_assignment_header_and_url_credentials(): - assert redact_command([ - "API_KEY=super-secret", "--api_key", "also-secret", - "-H", "Authorization: Bearer another-secret", - "https://alice:password@example.test/run?access_token=last-secret&format=json", - ]) == [ - "API_KEY=", "--api_key", "", - "-H", "", - "https://@example.test/run?access_token=%3Credacted%3E&format=json", - ] - assert redact_command([ - "-ualice:password", "-psecret", "--user=alice:password", - "--header=Authorization: Bearer secret", - ]) == [ - "-u", "", "-p", "", "--user", "", - "--header", "", - ] - - -def test_command_provenance_redacts_compound_credential_assignments(): - assert redact_command([ - "AWS_SECRET_ACCESS_KEY=do-not-publish", - "AWS_ACCESS_KEY_ID=also-private", - "HTTP_AUTHORIZATION=Bearer another-secret", - "--token-budget", "512", - ]) == [ - "AWS_SECRET_ACCESS_KEY=", - "AWS_ACCESS_KEY_ID=", - "HTTP_AUTHORIZATION=", - "--token-budget", "512", - ] - - -def test_command_provenance_redacts_fragment_credentials_without_hiding_normal_options(): - assert redact_command([ - "--token-budget", "512", "--tokenizer-model", "reader-v1", - "https://example.test/callback#access_token=do-not-publish&state=visible", - ]) == [ - "--token-budget", "512", "--tokenizer-model", "reader-v1", - "https://example.test/callback#access_token=%3Credacted%3E&state=visible", - ] - - -def test_command_provenance_redacts_embedded_and_signed_url_credentials(): - assert redact_command([ - "DATASET_URL=https://example.test/data?access_token=do-not-publish", - "--dataset-url=https://example.test/data?X-Amz-Signature=signed&sig=azure", - "https://example.test/data?signature=generic", - ]) == [ - "DATASET_URL=https://example.test/data?access_token=%3Credacted%3E", - "--dataset-url=https://example.test/data?X-Amz-Signature=%3Credacted%3E&sig=%3Credacted%3E", - "https://example.test/data?signature=%3Credacted%3E", - ] - - -def test_command_provenance_redacts_userinfo_when_a_url_port_is_malformed(): - assert redact_command([ - "https://alice:password@example.test:notaport/path?access_token=do-not-publish", - ]) == [ - "https://@example.test:notaport/path?access_token=%3Credacted%3E", - ] - - -def test_command_provenance_fails_closed_when_url_splitting_rejects_userinfo(): - assert redact_command(["https://user:password@[invalid/path"]) == [""] - - -def test_canonical_profile_validator_and_immutable_artifact_writer(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - profile = json.loads(json.dumps(LONGMEMEVAL_V2_CANONICAL_PROFILE_TEMPLATE)) - profile["benchmark"]["repository_revision"] = "a" * 40 - profile["benchmark"]["dataset_revision"] = "b" * 40 - profile["reader"]["revision"] = "c" * 40 - profile["embedding"]["revision"] = "d" * 40 - profile["baseline_label"] = "full_hybrid" - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid", profile=profile - ) - report = _complete_canonical_report(dataset, config) - assert validate_report(report, canonical=True) == [] - dirty = deepcopy(report) - dirty["system"]["git_dirty"] = True - assert "canonical reports require a clean git worktree" in validate_report( - dirty, canonical=True - ) - artifact = tmp_path / "artifacts" / "run.json" - written = write_canonical_artifact(report, artifact, canonical=True) - assert written["sha256"] in artifact.with_name("run.json.sha256").read_text("ascii") - assert json.loads(artifact.read_text("utf-8"))["schema"] == SCHEMA - assert write_canonical_artifact(report, artifact, canonical=True) == written - changed = dict(report) - changed["records"] = [dict(report["records"][0])] - changed["records"][0]["latency_ms"] = 2.0 - with pytest.raises(FileExistsError): - write_canonical_artifact(changed, artifact, canonical=True) - - -def test_report_validator_recomputes_embedded_config_digest(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"baseline_label": "full_hybrid"}, - records=[question_record("q1")], git_commit="abc123", - ) - report["protocol"]["config"]["baseline_label"] = "dense_only" - - errors = validate_report(report) - - assert "system.config_sha256 must match the canonical protocol.config digest" in errors - - -def test_report_validator_rejects_inconsistent_or_duplicate_exclusions(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - excluded = {"question_id": "q2", "reason": "no_gold_evidence", "detail": ""} - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"k": 5}, - records=[ - question_record("q1"), - question_record("q2", excluded=excluded), - ], - git_commit="abc123", - ) - assert validate_report(report) == [] - - report["exclusions"] = [excluded, excluded] - errors = validate_report(report) - assert "exclusion question_id values must be unique" in errors - - report["exclusions"] = [] - errors = validate_report(report) - assert "top-level exclusions must exactly match per-record exclusions" in errors - - -def test_default_canonical_profile_is_pinned_and_rejects_mutable_revisions(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - report = _complete_canonical_report(dataset, config) - assert validate_report(report, canonical=True) == [] - assert all( - len(value) == 40 - for value in ( - config["canonical_profile"]["benchmark"]["repository_revision"], - config["canonical_profile"]["benchmark"]["dataset_revision"], - config["canonical_profile"]["reader"]["revision"], - config["canonical_profile"]["embedding"]["revision"], - ) - ) - assert config["token_budgets"] == list(CANONICAL_TOKEN_BUDGETS) - - config["canonical_profile"]["reader"]["revision"] = "main" - errors = validate_report(report, canonical=True) - assert any("reader.revision" in error and "immutable" in error for error in errors) - - -def test_canonical_validator_rejects_unpinned_commit_private_prompts_and_unlabeled_measurements(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - report = _complete_canonical_report(dataset, config) - report["system"]["git_commit"] = "not-a-commit" - report["records"][0]["q"] = "private source question" - report["records"][0]["question_sha256"] = "a" * 64 - report["records"][0].pop("context_token_method") - report["metrics"].pop("recall_at_10") - - errors = validate_report(report, canonical=True) - - assert any("git_commit" in error for error in errors) - assert "canonical records must not contain raw query text" in errors - assert "canonical records must not contain question-derived hashes" in errors - assert any("context_token_method" in error for error in errors) - assert any("metrics.recall_at_10" in error for error in errors) - - config["canonical_profile"]["reader"]["revision"] = "C" * 40 - errors = validate_report(report, canonical=True) - assert any("reader.revision" in error and "immutable" in error for error in errors) - - -def test_canonical_validator_requires_grounded_metrics_or_explicit_unavailability(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - report = _complete_canonical_report(dataset, config) - report["metrics"].pop("grounded_f1") - report["metrics"]["abstention_f1"] = {"available": False} - - errors = validate_report(report, canonical=True) - - assert any("grounded_f1" in error and "unavailable reason" in error for error in errors) - assert any("abstention_f1" in error and "unavailable reason" in error for error in errors) - - -def test_canonical_validator_requires_measured_rows_for_every_fixed_budget(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - report = _complete_canonical_report(dataset, config) - report["metrics"]["fixed_budget_curve"]["rows"].pop() - - errors = validate_report(report, canonical=True) - - assert "canonical fixed-budget curve must contain every canonical token budget" in errors - report["metrics"]["fixed_budget_curve"] = {"available": False, "reason": "not_run"} - errors = validate_report(report, canonical=True) - assert "canonical fixed-budget curve is unavailable and cannot qualify as evidence" in errors - - report = _complete_canonical_report(dataset, config) - report["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0]["excluded"] = True - errors = validate_report(report, canonical=True) - assert "canonical fixed-budget curve 256 records must preserve exclusion state" in errors - - -def test_canonical_validator_requires_complete_dataset_cardinality(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - assert validate_report(valid, canonical=True) == [] - - missing_complete = deepcopy(valid) - missing_complete["protocol"].pop("complete_dataset") - assert "canonical protocol.complete_dataset must be true" in validate_report( - missing_complete, canonical=True - ) - - for invalid_count in (True, 0, 2): - mismatched = deepcopy(valid) - mismatched["protocol"]["source_questions"] = invalid_count - errors = validate_report(mismatched, canonical=True) - assert any("protocol.source_questions" in error for error in errors) - - -def test_canonical_validator_rejects_invalid_numeric_and_token_accounting(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - config["token_budget"] = 4 - valid = _complete_canonical_report(dataset, config) - valid["records"][0]["usage"] = { - "budget_tokens": 4, - "context_tokens": 3, - "token_counter": valid["records"][0]["context_tokenizer_identity"], - } - assert validate_report(valid, canonical=True) == [] - - mutations = ( - (("metrics", "recall_at_1"), True, "metrics.recall_at_1"), - (("records", 0, "recall_at_1"), True, "records require recall_at_1"), - (("records", 0, "latency_ms"), float("inf"), "latency_ms"), - (("records", 0, "context_tokens"), float("nan"), "context_tokens"), - (("records", 0, "context_tokens"), -1, "context_tokens"), - (("records", 0, "context_tokens"), 5, "must not exceed protocol token_budget"), - ( - ("records", 0, "usage", "context_tokens"), - 5, - "usage.context_tokens must not exceed usage.budget_tokens", - ), - ( - ("records", 0, "usage", "budget_tokens"), - 5, - "usage.budget_tokens must equal protocol token_budget", - ), - ( - ("records", 0, "usage", "source_tokens"), - True, - "usage.source_tokens must be non-negative and finite", - ), - ( - ("records", 0, "usage", "savings_ratio"), - float("inf"), - "usage.savings_ratio must be a number in [0, 1]", - ), - ( - ("metrics", "fixed_budget_curve", "rows", 0, "recall_at_1"), - True, - "fixed-budget curve 256 requires recall_at_1", - ), - ( - ("metrics", "fixed_budget_curve", "rows", 0, "records", 0, "context_tokens"), - 257, - "context_tokens within budget", - ), - ) - for path, value, expected in mutations: - report = deepcopy(valid) - target = report - for key in path[:-1]: - target = target[key] - target[path[-1]] = value - errors = validate_report(report, canonical=True) - assert any(expected in error for error in errors), (path, errors) - - -def test_canonical_validator_rejects_tampered_confidence_intervals(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - assert validate_report(valid, canonical=True) == [] - - mutations = ( - ("point", float("nan"), "point/low/high must be finite"), - ("low", -0.1, "point/low/high must be finite"), - ("high", 1.1, "point/low/high must be finite"), - ("high", 0.5, "low <= point <= high"), - ("point", 0.5, ".point must match metrics.recall_at_1"), - ("n", 2, ".n must equal the non-excluded record count"), - ("seed", -1, ".seed must be a non-negative integer"), - ("iterations", 0, ".iterations must be a positive integer"), - ("iterations", -1, ".iterations must be a positive integer"), - ("iterations", True, ".iterations must be a positive integer"), - ("strata_key", "topic", ".strata_key must equal category"), - ("low", 0.75, "must exactly match deterministic recomputation"), - ) - for key, value, expected in mutations: - report = deepcopy(valid) - report["metrics"]["confidence_intervals"]["recall_at_1"][key] = value - errors = validate_report(report, canonical=True) - assert any(expected in error for error in errors), (key, value, errors) - for metric_name in ( - "recall_at_1", "recall_at_5", "recall_at_10", - "mrr_at_1", "mrr_at_5", "mrr_at_10", - "ndcg_at_1", "ndcg_at_5", "ndcg_at_10", - ): - report = deepcopy(valid) - interval = report["metrics"]["confidence_intervals"][metric_name] - if interval["low"] > 0: - interval["low"] = round(interval["low"] - 0.000001, 6) - else: - interval["high"] = round(interval["high"] + 0.000001, 6) - errors = validate_report(report, canonical=True) - assert any( - "must exactly match deterministic recomputation" in error - for error in errors - ), (metric_name, errors) - - extra = deepcopy(valid) - extra["metrics"]["confidence_intervals"]["recall_at_1"]["mean"] = 1.0 - errors = validate_report(extra, canonical=True) - assert any("must match the canonical confidence interval schema" in error for error in errors) - - missing = deepcopy(valid) - missing["metrics"]["confidence_intervals"].pop("recall_at_1") - errors = validate_report(missing, canonical=True) - assert ( - "canonical metrics.confidence_intervals must exactly cover every rank metric" - in errors - ) - - -def test_canonical_validator_rejects_tampered_paired_bootstrap_payloads(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - - unavailable_mutations = ( - ("reason", "", ".reason must be a non-empty string"), - ("n", 1, ".n must be zero when unavailable"), - ("delta", 0.0, "delta/low/high must be null when unavailable"), - ("iterations", 0, ".iterations must be a positive integer"), - ("iterations", True, ".iterations must be a positive integer"), - ) - for key, value, expected in unavailable_mutations: - report = deepcopy(valid) - report["metrics"]["paired_bootstrap"][key] = value - errors = validate_report(report, canonical=True) - assert any(expected in error for error in errors), (key, value, errors) - - available = deepcopy(valid) - available["metrics"]["paired_bootstrap"] = { - "available": True, - "metric": "recall_at_5", - "delta": 0.25, - "low": 0.0, - "high": 0.5, - "n": 1, - "seed": 20260729, - "iterations": 20, - } - errors = validate_report(available, canonical=True) - assert any( - "must be unavailable until an immutable baseline artifact" in error - for error in errors - ) - - -def test_canonical_validator_recomputes_all_rank_aggregates_from_record_ids(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - - top_level = deepcopy(valid) - top_level["metrics"]["recall_at_5"] = 0.5 - errors = validate_report(top_level, canonical=True) - assert ( - "canonical metrics.recall_at_5 must equal the non-excluded record mean" - in errors - ) - - curve_aggregate = deepcopy(valid) - curve_aggregate["metrics"]["fixed_budget_curve"]["rows"][0]["ndcg_at_10"] = 0.5 - errors = validate_report(curve_aggregate, canonical=True) - assert any( - "fixed-budget curve 256 ndcg_at_10" in error - and "non-excluded record mean" in error - for error in errors - ) - - curve_measurement = deepcopy(valid) - measurement = curve_measurement["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0] - measurement["retrieved_ids"] = [] - errors = validate_report(curve_measurement, canonical=True) - assert any( - "fixed-budget curve 256 record recall_at_1" in error - and "retrieved_ids and supporting_ids" in error - for error in errors - ) - - -def test_canonical_validator_derives_numeric_grounded_metrics_from_labels(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - - unlabeled = _complete_canonical_report(dataset, config) - unlabeled["metrics"]["grounded_f1"] = 0.75 - unlabeled["metrics"]["abstention_f1"] = 0.75 - errors = validate_report(unlabeled, canonical=True) - assert any( - "metrics.grounded_f1 requires labeled per-question grounded values" in error - and "unavailable reason" in error - for error in errors - ) - assert any( - "metrics.abstention_f1 requires labeled per-question abstained values" in error - and "unavailable reason" in error - for error in errors - ) - - measured = _complete_canonical_report(dataset, config) - measured["records"][0].update({ - "answerable": True, - "grounded": True, - "abstained": False, - }) - measured["metrics"]["grounded"] = { - "available": True, - **metrics.grounded_precision_recall_f1([True], [True]), - } - measured["metrics"]["abstention"] = { - "available": True, - **metrics.abstention_precision_recall_f1([False], [True]), - } - measured["metrics"]["grounded_f1"] = 1.0 - measured["metrics"]["abstention_f1"] = 1.0 - assert validate_report(measured, canonical=True) == [] - - bad_count = deepcopy(measured) - bad_count["metrics"]["grounded"]["n"] = 2 - errors = validate_report(bad_count, canonical=True) - assert ( - "canonical metrics.grounded.n must be recomputed from per-question labels" - in errors - ) - - measured["metrics"]["grounded_f1"] = 0.0 - errors = validate_report(measured, canonical=True) - assert ( - "canonical metrics.grounded_f1 must be recomputed from per-question labels" - in errors - ) - - -def test_canonical_validator_requires_pinned_reader_tokenizer_identity(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - - estimated = deepcopy(valid) - estimated["records"][0]["context_token_method"] = "deterministic_estimate" - estimated["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0][ - "context_token_method" - ] = "deterministic_estimate" - errors = validate_report(estimated, canonical=True) - assert any( - "context_token_method=pinned_reader_content_tokenizer" in error - for error in errors - ) - assert any( - "fixed-budget curve 256 records require" in error - and "context_token_method=pinned_reader_content_tokenizer" in error - for error in errors - ) - - mismatched = deepcopy(valid) - mismatched["records"][0]["context_tokenizer_identity"] = "other/model@" + "e" * 40 - mismatched["records"][0]["usage"]["token_counter"] = "other/model@" + "e" * 40 - errors = validate_report(mismatched, canonical=True) - assert any("context_tokenizer_identity must match" in error for error in errors) - assert any("usage.token_counter must match" in error for error in errors) - - -def test_benchmark_cli_writes_canonical_json_and_checksum(tmp_path, capsys): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"k": 5}, - records=[question_record("q1")], git_commit="abc123", - ) - source = tmp_path / "source.json" - source.write_text(json.dumps(report), encoding="utf-8") - artifact = tmp_path / "artifact.json" - assert main(["--input", str(source), "--output", str(artifact)]) == 0 - assert artifact.exists() and artifact.with_name("artifact.json.sha256").exists() - assert "sha256" in capsys.readouterr().out - - -def test_exact_tokenizer_fallback_budget_curves_and_deterministic_cis(): - assert count_tokens("abc", CharacterTokenizer()) == {"tokens": 3, "method": "injected"} - assert count_tokens("one two")["method"] == "deterministic_estimate" - records = [ - {"category": "a", "supporting_ids": ["m1"], "chunks": [ - {"id": "m1", "tokens": 3}, {"id": "m2", "tokens": 3} - ]}, - {"category": "b", "supporting_ids": ["m2"], "chunks": [ - {"id": "m1", "tokens": 3}, {"id": "m2", "tokens": 3} - ]}, - ] - curve = fixed_budget_curve(records, [3, 6]) - assert curve[0]["recall"] == 0.5 - assert curve[1]["recall"] == 1.0 - def metric(rows): - return sum(row["value"] for row in rows) / len(rows) - ci_one = stratified_bootstrap_ci( - [{"category": "a", "value": 1.0}, {"category": "b", "value": 0.0}], - metric, iterations=40, seed=4, - ) - ci_two = stratified_bootstrap_ci( - [{"category": "a", "value": 1.0}, {"category": "b", "value": 0.0}], - metric, iterations=40, seed=4, - ) - assert ci_one == ci_two - paired = paired_bootstrap_ci([(1.0, 0.0), (0.0, 0.0)], iterations=40, seed=4) - assert paired["delta"] == 0.5 and paired["n"] == 2 + + runs = {run["id"]: run for run in artifact["runs"]} + assert set(runs) == { + "offline-chunking", + "offline-performance", + "offline-grounded", + } + for run in runs.values(): + assert hashlib.sha256(run["command"].encode()).hexdigest() == run["config_digest"] + + chunking = offline_release_evidence["chunking"] + chunking_result = runs["offline-chunking"]["result"] + for mode in ("whole", "chunked"): + live = chunking["reports"][mode] + recorded = chunking_result[mode] + assert recorded["memories"] == live["memories_stored"] + assert recorded["recall_at_k"] == live["recall_at_k"] + assert recorded["mean_context_tokens"] == live["mean_context_tokens"] + assert recorded["mean_evidence_tokens"] == live["mean_evidence_tokens"] + assert recorded["max_stored_tokens"] == live["max_stored_tokens"] + assert chunking_result["context_reduction_pct"] == chunking["context_reduction_pct"] + + performance = offline_release_evidence["performance"] + performance_result = runs["offline-performance"]["result"] + assert performance_result["questions"] == performance["corpus"]["questions"] + assert performance_result["timed_recalls"] == performance["run"]["timed_recalls"] + assert performance_result["recall_at_k"] == performance["quality"]["recall_at_k"] + assert performance_result["hit_at_k"] == performance["quality"]["hit_at_k"] + assert ( + performance_result["answer_token_recall"] + == performance["quality"]["answer_token_recall"] + ) + + # These values are deterministic fixture aggregates, not wall-clock timing + # observations. Approximate comparisons would let serializer or count drift + # pass the publication contract unnoticed. + assert performance_result["mean_context_tokens"] == performance["context"]["mean_tokens"] + assert performance_result["max_context_tokens"] == performance["context"]["max_tokens"] + assert ( + performance_result["full_serialized_payload_tokens"] + == performance["context"]["full_serialized_payload_tokens"] + ) + assert ( + performance_result["compact_serialized_payload_tokens"] + == performance["context"]["compact_serialized_payload_tokens"] + ) + assert ( + performance_result["saved_serialized_payload_tokens"] + == performance["context"]["saved_serialized_payload_tokens"] + ) + assert ( + performance_result["serialized_payload_savings_ratio"] + == performance["context"]["serialized_payload_savings_ratio"] + ) + + grounded = offline_release_evidence["grounded"] + grounded_result = runs["offline-grounded"]["result"] + assert grounded_result == { + "answerable": grounded["n_answerable"], + "grounded": grounded["grounded_hits"], + "off_topic": grounded["n_unanswerable"], + "quarantined": grounded["n_quarantine"], + "abstained": grounded["abstain_hits"], + "quarantine_hits": grounded["quarantine_hits"], + "decision_accuracy": grounded["accuracy"], + } + + surfaces = ( + ROOT / "README.md", + ROOT / "BENCHMARKS.md", + ROOT / "docs" / "images" / "context-efficiency.svg", + ROOT / "docs" / "images" / "evidence-backed-agent-examples.svg", + ) + for surface in surfaces: + assert expected_sha in surface.read_text(encoding="utf-8") + + claimed_ids = set( + re.findall( + r"offline-(?:chunking|performance|grounded)", + "\n".join(path.read_text(encoding="utf-8") for path in surfaces), + ) + ) + assert claimed_ids == set(runs) + + +def test_benchmark_guide_tracks_the_live_offline_evaluators(): + """Method prose must change whenever its executable offline evidence changes. + + Values are interpolated from the COMMITTED registry artifact — the publication + source of truth — so guide text cannot drift from the evidence it cites. + """ + benchmarks = (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") + normalized = " ".join(benchmarks.split()) + committed = _committed_evidence() + chunking = committed["chunking"] + whole = chunking["whole"] + chunked = chunking["chunked"] + performance = committed["performance"] + payload_samples = performance["questions"] + + for evidence in ( + f"falls from {whole['mean_context_tokens']:.1f} to " + f"{chunked['mean_context_tokens']:.1f} tokens", + f"{whole['mean_context_tokens'] - chunked['mean_context_tokens']:.1f} fewer, " + f"{chunking['context_reduction_pct']:.1f}% lower", + f"falls from {whole['mean_evidence_tokens']:.1f} to " + f"{chunked['mean_evidence_tokens']:.1f} tokens", + "Payload proxies are sampled once per question", + "not serialized MCP envelopes or transport responses", + f"{payload_samples} payload samples total **" + f"{performance['full_serialized_payload_tokens']:,}** full-proxy", + f"versus **{performance['compact_serialized_payload_tokens']:,}** compact-proxy tokens", + f"avoiding **{performance['saved_serialized_payload_tokens']:,}** proxy tokens", + f"**{100 * performance['serialized_payload_savings_ratio']:.2f}% lower**", + f"averages **{performance['mean_context_tokens']:.2f}** tokens and reaches " + f"**{performance['max_context_tokens']}**", + ): + assert evidence in normalized + + + +def _complete_canonical_report(dataset, config): + """Minimal but fully auditable canonical envelope for validator coverage.""" + profile = config["canonical_profile"] + tokenizer_identity = ( + f"{profile['reader']['model']}@{profile['reader']['revision']}" + ) + record = question_record( + "q1", category="state", context_tokens=3, latency_ms=1.25, + retrieved_ids=["support"], supporting_ids=["support"], + recall_at_1=1.0, recall_at_5=1.0, recall_at_10=1.0, + mrr_at_1=1.0, mrr_at_5=1.0, mrr_at_10=1.0, + ndcg_at_1=1.0, ndcg_at_5=1.0, ndcg_at_10=1.0, + usage={ + "budget_tokens": config.get("token_budget") or 3, + "context_tokens": 3, + "token_counter": tokenizer_identity, + }, + ) + record["context_token_method"] = "pinned_reader_content_tokenizer" + record["context_tokenizer_identity"] = tokenizer_identity + rank_metrics = { + f"{metric}_at_{depth}": 1.0 + for metric in ("recall", "mrr", "ndcg") + for depth in (1, 5, 10) + } + curve_record = { + "question_id": "q1", + "excluded": False, + "context_tokens": 3, + "context_token_method": "pinned_reader_content_tokenizer", + "context_tokenizer_identity": tokenizer_identity, + "retrieved_ids": ["support"], + "supporting_ids": ["support"], + **rank_metrics, + } + report = report_envelope( + suite="fixture", dataset_path=dataset, config=config, records=[record], + metrics={ + **rank_metrics, + "confidence_intervals": { + field: { + "point": 1.0, + "low": 1.0, + "high": 1.0, + "n": 1, + "seed": 20260729, + "iterations": 1, + "strata_key": "category", + } + for field in rank_metrics + }, + "paired_bootstrap": { + "available": False, + "reason": "baseline_records_not_supplied", + "n": 0, + "delta": None, + "low": None, + "high": None, + "iterations": 1, + }, + "grounded_f1": {"available": False, "reason": "not_measured"}, + "abstention_f1": {"available": False, "reason": "not_measured"}, + "fixed_budget_curve": { + "available": True, + "rows": [{ + "token_budget": budget, + "status": "measured", + "n_total": 1, + "n_scored": 1, + "records": [dict(curve_record)], + **rank_metrics, + } for budget in CANONICAL_TOKEN_BUDGETS], + }, + }, + git_commit="a" * 40, + ) + report["system"]["git_dirty"] = False + report["models"] = {"embedder": { + "name": "FixtureEmbedder", + "model_id": profile["embedding"]["model"], + "revision": profile["embedding"]["revision"], + "sha256": "b" * 64, + }} + report["protocol"]["complete_dataset"] = True + report["protocol"]["source_questions"] = len(report["records"]) + return report + + +def test_metrics_cover_rank_sensitive_retrieval_quality(): + retrieved = ["noise", "evidence-a", "evidence-b"] + supporting = ["evidence-a", "evidence-b"] + assert metrics.mrr_at_k(retrieved, supporting, 3) == 0.5 + assert metrics.ndcg_at_k(retrieved, supporting, 3) > 0.6 + assert metrics.recall_at_k(retrieved[:1], supporting) == 0.0 + assert metrics.hit_at_k(retrieved[:1], supporting) == 0.0 + bundle = metrics.retrieval_metrics_at_depths(retrieved, supporting) + assert bundle["recall_at_1"] == 0.0 + assert bundle["recall_at_5"] == 1.0 + assert bundle["mrr_at_5"] == 0.5 + + +def test_envelope_hashes_dataset_config_and_retains_exclusions(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + excluded = {"question_id": "q2", "reason": "no_gold_evidence", "detail": ""} + records = [ + question_record("q1", category="state", supporting_ids=["m1"]), + question_record("q2", category="abstention", excluded=excluded), + ] + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"k": 5}, records=records, + metrics={"recall": 1.0}, git_commit="abc123", + ) + assert report["schema"] == SCHEMA + assert report["suite"]["sha256"] + assert report["system"]["config_sha256"] + assert report["protocol"] == { + "command": ["in_process"], + "config": {"k": 5}, + "token_accounting": { + "identity": "unspecified", + "revision": None, + "scope": "unspecified", + "method": "unspecified", + }, + "n_total": 2, + "n_scored": 1, + } + assert report["exclusions"] == [{ + "question_id": "q2", + "reason": "no_gold_evidence", + }] + assert json.loads(json.dumps(report))["schema"] == SCHEMA + + +def test_envelope_redacts_top_level_exclusion_detail(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"k": 5}, records=[], + exclusions=[{ + "question_id": "q1", "reason": "invalid", "detail": "private prompt text", + }], + ) + + assert report["exclusions"] == [{ + "question_id": "q1", + "reason": "invalid", + }] + + +def test_command_provenance_redacts_explicit_credential_arguments(): + assert redact_command([ + "python", "-m", "runner", "--api-key", "do-not-publish", "--token=value", + ]) == [ + "python", "-m", "runner", "--api-key", "", "--token", "", + ] + + +def test_command_provenance_redacts_assignment_header_and_url_credentials(): + assert redact_command([ + "API_KEY=super-secret", "--api_key", "also-secret", + "-H", "Authorization: Bearer another-secret", + "https://alice:password@example.test/run?access_token=last-secret&format=json", + ]) == [ + "API_KEY=", "--api_key", "", + "-H", "", + "https://@example.test/run?access_token=%3Credacted%3E&format=json", + ] + assert redact_command([ + "-ualice:password", "-psecret", "--user=alice:password", + "--header=Authorization: Bearer secret", + ]) == [ + "-u", "", "-p", "", "--user", "", + "--header", "", + ] + + +def test_command_provenance_redacts_compound_credential_assignments(): + assert redact_command([ + "AWS_SECRET_ACCESS_KEY=do-not-publish", + "AWS_ACCESS_KEY_ID=also-private", + "HTTP_AUTHORIZATION=Bearer another-secret", + "--token-budget", "512", + ]) == [ + "AWS_SECRET_ACCESS_KEY=", + "AWS_ACCESS_KEY_ID=", + "HTTP_AUTHORIZATION=", + "--token-budget", "512", + ] + + +def test_command_provenance_redacts_fragment_credentials_without_hiding_normal_options(): + assert redact_command([ + "--token-budget", "512", "--tokenizer-model", "reader-v1", + "https://example.test/callback#access_token=do-not-publish&state=visible", + ]) == [ + "--token-budget", "512", "--tokenizer-model", "reader-v1", + "https://example.test/callback#access_token=%3Credacted%3E&state=visible", + ] + + +def test_command_provenance_redacts_embedded_and_signed_url_credentials(): + assert redact_command([ + "DATASET_URL=https://example.test/data?access_token=do-not-publish", + "--dataset-url=https://example.test/data?X-Amz-Signature=signed&sig=azure", + "https://example.test/data?signature=generic", + ]) == [ + "DATASET_URL=https://example.test/data?access_token=%3Credacted%3E", + "--dataset-url=https://example.test/data?X-Amz-Signature=%3Credacted%3E&sig=%3Credacted%3E", + "https://example.test/data?signature=%3Credacted%3E", + ] + + +def test_command_provenance_redacts_userinfo_when_a_url_port_is_malformed(): + assert redact_command([ + "https://alice:password@example.test:notaport/path?access_token=do-not-publish", + ]) == [ + "https://@example.test:notaport/path?access_token=%3Credacted%3E", + ] + + +def test_command_provenance_fails_closed_when_url_splitting_rejects_userinfo(): + assert redact_command(["https://user:password@[invalid/path"]) == [""] + + +def test_canonical_profile_validator_and_immutable_artifact_writer(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + profile = json.loads(json.dumps(LONGMEMEVAL_V2_CANONICAL_PROFILE_TEMPLATE)) + profile["benchmark"]["repository_revision"] = "a" * 40 + profile["benchmark"]["dataset_revision"] = "b" * 40 + profile["reader"]["revision"] = "c" * 40 + profile["embedding"]["revision"] = "d" * 40 + profile["baseline_label"] = "full_hybrid" + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid", profile=profile + ) + report = _complete_canonical_report(dataset, config) + assert validate_report(report, canonical=True) == [] + dirty = deepcopy(report) + dirty["system"]["git_dirty"] = True + assert "canonical reports require a clean git worktree" in validate_report( + dirty, canonical=True + ) + artifact = tmp_path / "artifacts" / "run.json" + written = write_canonical_artifact(report, artifact, canonical=True) + assert written["sha256"] in artifact.with_name("run.json.sha256").read_text("ascii") + assert json.loads(artifact.read_text("utf-8"))["schema"] == SCHEMA + assert write_canonical_artifact(report, artifact, canonical=True) == written + changed = dict(report) + changed["records"] = [dict(report["records"][0])] + changed["records"][0]["latency_ms"] = 2.0 + with pytest.raises(FileExistsError): + write_canonical_artifact(changed, artifact, canonical=True) + + +def test_report_validator_recomputes_embedded_config_digest(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"baseline_label": "full_hybrid"}, + records=[question_record("q1")], git_commit="abc123", + ) + report["protocol"]["config"]["baseline_label"] = "dense_only" + + errors = validate_report(report) + + assert "system.config_sha256 must match the canonical protocol.config digest" in errors + + +def test_report_validator_rejects_inconsistent_or_duplicate_exclusions(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + excluded = {"question_id": "q2", "reason": "no_gold_evidence", "detail": ""} + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"k": 5}, + records=[ + question_record("q1"), + question_record("q2", excluded=excluded), + ], + git_commit="abc123", + ) + assert validate_report(report) == [] + + report["exclusions"] = [excluded, excluded] + errors = validate_report(report) + assert "exclusion question_id values must be unique" in errors + + report["exclusions"] = [] + errors = validate_report(report) + assert "top-level exclusions must exactly match per-record exclusions" in errors + + +def test_default_canonical_profile_is_pinned_and_rejects_mutable_revisions(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + report = _complete_canonical_report(dataset, config) + assert validate_report(report, canonical=True) == [] + assert all( + len(value) == 40 + for value in ( + config["canonical_profile"]["benchmark"]["repository_revision"], + config["canonical_profile"]["benchmark"]["dataset_revision"], + config["canonical_profile"]["reader"]["revision"], + config["canonical_profile"]["embedding"]["revision"], + ) + ) + assert config["token_budgets"] == list(CANONICAL_TOKEN_BUDGETS) + + config["canonical_profile"]["reader"]["revision"] = "main" + errors = validate_report(report, canonical=True) + assert any("reader.revision" in error and "immutable" in error for error in errors) + + +def test_canonical_validator_rejects_unpinned_commit_private_prompts_and_unlabeled_measurements(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + report = _complete_canonical_report(dataset, config) + report["system"]["git_commit"] = "not-a-commit" + report["records"][0]["q"] = "private source question" + report["records"][0]["question_sha256"] = "a" * 64 + report["records"][0].pop("context_token_method") + report["metrics"].pop("recall_at_10") + + errors = validate_report(report, canonical=True) + + assert any("git_commit" in error for error in errors) + assert "canonical records must not contain raw query text" in errors + assert "canonical records must not contain question-derived hashes" in errors + assert any("context_token_method" in error for error in errors) + assert any("metrics.recall_at_10" in error for error in errors) + + config["canonical_profile"]["reader"]["revision"] = "C" * 40 + errors = validate_report(report, canonical=True) + assert any("reader.revision" in error and "immutable" in error for error in errors) + + +def test_canonical_validator_requires_grounded_metrics_or_explicit_unavailability(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + report = _complete_canonical_report(dataset, config) + report["metrics"].pop("grounded_f1") + report["metrics"]["abstention_f1"] = {"available": False} + + errors = validate_report(report, canonical=True) + + assert any("grounded_f1" in error and "unavailable reason" in error for error in errors) + assert any("abstention_f1" in error and "unavailable reason" in error for error in errors) + + +def test_canonical_validator_requires_measured_rows_for_every_fixed_budget(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + report = _complete_canonical_report(dataset, config) + report["metrics"]["fixed_budget_curve"]["rows"].pop() + + errors = validate_report(report, canonical=True) + + assert "canonical fixed-budget curve must contain every canonical token budget" in errors + report["metrics"]["fixed_budget_curve"] = {"available": False, "reason": "not_run"} + errors = validate_report(report, canonical=True) + assert "canonical fixed-budget curve is unavailable and cannot qualify as evidence" in errors + + report = _complete_canonical_report(dataset, config) + report["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0]["excluded"] = True + errors = validate_report(report, canonical=True) + assert "canonical fixed-budget curve 256 records must preserve exclusion state" in errors + + +def test_canonical_validator_requires_complete_dataset_cardinality(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + assert validate_report(valid, canonical=True) == [] + + missing_complete = deepcopy(valid) + missing_complete["protocol"].pop("complete_dataset") + assert "canonical protocol.complete_dataset must be true" in validate_report( + missing_complete, canonical=True + ) + + for invalid_count in (True, 0, 2): + mismatched = deepcopy(valid) + mismatched["protocol"]["source_questions"] = invalid_count + errors = validate_report(mismatched, canonical=True) + assert any("protocol.source_questions" in error for error in errors) + + +def test_canonical_validator_rejects_invalid_numeric_and_token_accounting(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + config["token_budget"] = 4 + valid = _complete_canonical_report(dataset, config) + valid["records"][0]["usage"] = { + "budget_tokens": 4, + "context_tokens": 3, + "token_counter": valid["records"][0]["context_tokenizer_identity"], + } + assert validate_report(valid, canonical=True) == [] + + mutations = ( + (("metrics", "recall_at_1"), True, "metrics.recall_at_1"), + (("records", 0, "recall_at_1"), True, "records require recall_at_1"), + (("records", 0, "latency_ms"), float("inf"), "latency_ms"), + (("records", 0, "context_tokens"), float("nan"), "context_tokens"), + (("records", 0, "context_tokens"), -1, "context_tokens"), + (("records", 0, "context_tokens"), 5, "must not exceed protocol token_budget"), + ( + ("records", 0, "usage", "context_tokens"), + 5, + "usage.context_tokens must not exceed usage.budget_tokens", + ), + ( + ("records", 0, "usage", "budget_tokens"), + 5, + "usage.budget_tokens must equal protocol token_budget", + ), + ( + ("records", 0, "usage", "source_tokens"), + True, + "usage.source_tokens must be non-negative and finite", + ), + ( + ("records", 0, "usage", "savings_ratio"), + float("inf"), + "usage.savings_ratio must be a number in [0, 1]", + ), + ( + ("metrics", "fixed_budget_curve", "rows", 0, "recall_at_1"), + True, + "fixed-budget curve 256 requires recall_at_1", + ), + ( + ("metrics", "fixed_budget_curve", "rows", 0, "records", 0, "context_tokens"), + 257, + "context_tokens within budget", + ), + ) + for path, value, expected in mutations: + report = deepcopy(valid) + target = report + for key in path[:-1]: + target = target[key] + target[path[-1]] = value + errors = validate_report(report, canonical=True) + assert any(expected in error for error in errors), (path, errors) + + +def test_canonical_validator_rejects_tampered_confidence_intervals(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + assert validate_report(valid, canonical=True) == [] + + mutations = ( + ("point", float("nan"), "point/low/high must be finite"), + ("low", -0.1, "point/low/high must be finite"), + ("high", 1.1, "point/low/high must be finite"), + ("high", 0.5, "low <= point <= high"), + ("point", 0.5, ".point must match metrics.recall_at_1"), + ("n", 2, ".n must equal the non-excluded record count"), + ("seed", -1, ".seed must be a non-negative integer"), + ("iterations", 0, ".iterations must be a positive integer"), + ("iterations", -1, ".iterations must be a positive integer"), + ("iterations", True, ".iterations must be a positive integer"), + ("strata_key", "topic", ".strata_key must equal category"), + ("low", 0.75, "must exactly match deterministic recomputation"), + ) + for key, value, expected in mutations: + report = deepcopy(valid) + report["metrics"]["confidence_intervals"]["recall_at_1"][key] = value + errors = validate_report(report, canonical=True) + assert any(expected in error for error in errors), (key, value, errors) + for metric_name in ( + "recall_at_1", "recall_at_5", "recall_at_10", + "mrr_at_1", "mrr_at_5", "mrr_at_10", + "ndcg_at_1", "ndcg_at_5", "ndcg_at_10", + ): + report = deepcopy(valid) + interval = report["metrics"]["confidence_intervals"][metric_name] + if interval["low"] > 0: + interval["low"] = round(interval["low"] - 0.000001, 6) + else: + interval["high"] = round(interval["high"] + 0.000001, 6) + errors = validate_report(report, canonical=True) + assert any( + "must exactly match deterministic recomputation" in error + for error in errors + ), (metric_name, errors) + + extra = deepcopy(valid) + extra["metrics"]["confidence_intervals"]["recall_at_1"]["mean"] = 1.0 + errors = validate_report(extra, canonical=True) + assert any("must match the canonical confidence interval schema" in error for error in errors) + + missing = deepcopy(valid) + missing["metrics"]["confidence_intervals"].pop("recall_at_1") + errors = validate_report(missing, canonical=True) + assert ( + "canonical metrics.confidence_intervals must exactly cover every rank metric" + in errors + ) + + +def test_canonical_validator_rejects_tampered_paired_bootstrap_payloads(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + + unavailable_mutations = ( + ("reason", "", ".reason must be a non-empty string"), + ("n", 1, ".n must be zero when unavailable"), + ("delta", 0.0, "delta/low/high must be null when unavailable"), + ("iterations", 0, ".iterations must be a positive integer"), + ("iterations", True, ".iterations must be a positive integer"), + ) + for key, value, expected in unavailable_mutations: + report = deepcopy(valid) + report["metrics"]["paired_bootstrap"][key] = value + errors = validate_report(report, canonical=True) + assert any(expected in error for error in errors), (key, value, errors) + + available = deepcopy(valid) + available["metrics"]["paired_bootstrap"] = { + "available": True, + "metric": "recall_at_5", + "delta": 0.25, + "low": 0.0, + "high": 0.5, + "n": 1, + "seed": 20260729, + "iterations": 20, + } + errors = validate_report(available, canonical=True) + assert any( + "must be unavailable until an immutable baseline artifact" in error + for error in errors + ) + + +def test_canonical_validator_recomputes_all_rank_aggregates_from_record_ids(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + + top_level = deepcopy(valid) + top_level["metrics"]["recall_at_5"] = 0.5 + errors = validate_report(top_level, canonical=True) + assert ( + "canonical metrics.recall_at_5 must equal the non-excluded record mean" + in errors + ) + + curve_aggregate = deepcopy(valid) + curve_aggregate["metrics"]["fixed_budget_curve"]["rows"][0]["ndcg_at_10"] = 0.5 + errors = validate_report(curve_aggregate, canonical=True) + assert any( + "fixed-budget curve 256 ndcg_at_10" in error + and "non-excluded record mean" in error + for error in errors + ) + + curve_measurement = deepcopy(valid) + measurement = curve_measurement["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0] + measurement["retrieved_ids"] = [] + errors = validate_report(curve_measurement, canonical=True) + assert any( + "fixed-budget curve 256 record recall_at_1" in error + and "retrieved_ids and supporting_ids" in error + for error in errors + ) + + +def test_canonical_validator_derives_numeric_grounded_metrics_from_labels(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + + unlabeled = _complete_canonical_report(dataset, config) + unlabeled["metrics"]["grounded_f1"] = 0.75 + unlabeled["metrics"]["abstention_f1"] = 0.75 + errors = validate_report(unlabeled, canonical=True) + assert any( + "metrics.grounded_f1 requires labeled per-question grounded values" in error + and "unavailable reason" in error + for error in errors + ) + assert any( + "metrics.abstention_f1 requires labeled per-question abstained values" in error + and "unavailable reason" in error + for error in errors + ) + + measured = _complete_canonical_report(dataset, config) + measured["records"][0].update({ + "answerable": True, + "grounded": True, + "abstained": False, + }) + measured["metrics"]["grounded"] = { + "available": True, + **metrics.grounded_precision_recall_f1([True], [True]), + } + measured["metrics"]["abstention"] = { + "available": True, + **metrics.abstention_precision_recall_f1([False], [True]), + } + measured["metrics"]["grounded_f1"] = 1.0 + measured["metrics"]["abstention_f1"] = 1.0 + assert validate_report(measured, canonical=True) == [] + + bad_count = deepcopy(measured) + bad_count["metrics"]["grounded"]["n"] = 2 + errors = validate_report(bad_count, canonical=True) + assert ( + "canonical metrics.grounded.n must be recomputed from per-question labels" + in errors + ) + + measured["metrics"]["grounded_f1"] = 0.0 + errors = validate_report(measured, canonical=True) + assert ( + "canonical metrics.grounded_f1 must be recomputed from per-question labels" + in errors + ) + + +def test_canonical_validator_requires_pinned_reader_tokenizer_identity(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + + estimated = deepcopy(valid) + estimated["records"][0]["context_token_method"] = "deterministic_estimate" + estimated["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0][ + "context_token_method" + ] = "deterministic_estimate" + errors = validate_report(estimated, canonical=True) + assert any( + "context_token_method=pinned_reader_content_tokenizer" in error + for error in errors + ) + assert any( + "fixed-budget curve 256 records require" in error + and "context_token_method=pinned_reader_content_tokenizer" in error + for error in errors + ) + + mismatched = deepcopy(valid) + mismatched["records"][0]["context_tokenizer_identity"] = "other/model@" + "e" * 40 + mismatched["records"][0]["usage"]["token_counter"] = "other/model@" + "e" * 40 + errors = validate_report(mismatched, canonical=True) + assert any("context_tokenizer_identity must match" in error for error in errors) + assert any("usage.token_counter must match" in error for error in errors) + + +def test_benchmark_cli_writes_canonical_json_and_checksum(tmp_path, capsys): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"k": 5}, + records=[question_record("q1")], git_commit="abc123", + ) + source = tmp_path / "source.json" + source.write_text(json.dumps(report), encoding="utf-8") + artifact = tmp_path / "artifact.json" + assert main(["--input", str(source), "--output", str(artifact)]) == 0 + assert artifact.exists() and artifact.with_name("artifact.json.sha256").exists() + assert "sha256" in capsys.readouterr().out + + +def test_exact_tokenizer_fallback_budget_curves_and_deterministic_cis(): + assert count_tokens("abc", CharacterTokenizer()) == {"tokens": 3, "method": "injected"} + assert count_tokens("one two")["method"] == "deterministic_estimate" + records = [ + {"category": "a", "supporting_ids": ["m1"], "chunks": [ + {"id": "m1", "tokens": 3}, {"id": "m2", "tokens": 3} + ]}, + {"category": "b", "supporting_ids": ["m2"], "chunks": [ + {"id": "m1", "tokens": 3}, {"id": "m2", "tokens": 3} + ]}, + ] + curve = fixed_budget_curve(records, [3, 6]) + assert curve[0]["recall"] == 0.5 + assert curve[1]["recall"] == 1.0 + def metric(rows): + return sum(row["value"] for row in rows) / len(rows) + ci_one = stratified_bootstrap_ci( + [{"category": "a", "value": 1.0}, {"category": "b", "value": 0.0}], + metric, iterations=40, seed=4, + ) + ci_two = stratified_bootstrap_ci( + [{"category": "a", "value": 1.0}, {"category": "b", "value": 0.0}], + metric, iterations=40, seed=4, + ) + assert ci_one == ci_two + paired = paired_bootstrap_ci([(1.0, 0.0), (0.0, 0.0)], iterations=40, seed=4) + assert paired["delta"] == 0.5 and paired["n"] == 2 diff --git a/tests/test_documentation_contracts.py b/tests/test_documentation_contracts.py index 7cca0647..b7f3e17a 100644 --- a/tests/test_documentation_contracts.py +++ b/tests/test_documentation_contracts.py @@ -1,312 +1,312 @@ -from __future__ import annotations - -import ast -import json -import re -import xml.etree.ElementTree as ET -from pathlib import Path - - -from engraphis.core.schema import SCHEMA_VERSION - - -ROOT = Path(__file__).resolve().parents[1] - - -def _read(path: str) -> str: - return (ROOT / path).read_text(encoding="utf-8") - - - -def test_readme_long_description_uses_no_repository_relative_targets() -> None: - readme = _read("README.md") - destinations = re.findall( - r"!?\[[^\]]*\]\(([^) ]+)|(?:href|src)=\"([^\"]+)\"", - readme, - ) - flattened = [markdown or html for markdown, html in destinations] - relative = [ - destination - for destination in flattened - if not destination.startswith(("#", "https://", "http://")) - ] - assert not relative - - image_targets = [ - destination - for destination in flattened - if destination.endswith((".png", ".svg")) - ] - assert image_targets - assert all( - target.startswith( - "https://raw.githubusercontent.com/Coding-Dev-Tools/engraphis/main/" - ) - or target.startswith("https://img.shields.io/") - for target in image_targets - ) - -def test_canonical_offline_gate_tracks_ci() -> None: - agents = _read("AGENTS.md") - claude = _read("CLAUDE.md") - workflow = _read(".github/workflows/ci.yml") - required = ( - "ruff check .", - "python scripts/check_commercial_manifest.py", - "python scripts/externalize_dashboard_assets.py", - "python -m pytest", - "python -m eval.harness --dataset eval/datasets/sample.jsonl --k 5", - "python -m eval.harness --dataset eval/datasets/codemem.jsonl --k 5", - "python -m eval.ablation", - "python -m eval.reinforcement", - "python -m eval.adversarial_memory_security", - "python -m eval.grounded", - "python -m eval.code_arm", - "pyright", - ) - - for command in required: - assert command in agents, f"AGENTS.md omits the canonical gate command: {command}" - assert command in workflow, f"CI omits the documented gate command: {command}" - - assert "Use the exact primary offline gate in `AGENTS.md` §1" in claude - assert "do not maintain a smaller duplicate here" in claude - - -def test_core_backend_imports_stay_behind_outer_composition_root() -> None: - violations: list[str] = [] - for path in sorted((ROOT / "engraphis" / "core").glob("*.py")): - tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path)) - for node in ast.walk(tree): - if not isinstance(node, ast.ImportFrom): - continue - module = node.module or "" - if module.startswith("engraphis.backends"): - violations.append(f"{path.relative_to(ROOT)} imports {module}") - assert not violations, violations - - factory = ast.parse(_read("engraphis/factory.py"), filename="engraphis/factory.py") - backend_modules = { - node.module - for node in ast.walk(factory) - if isinstance(node, ast.ImportFrom) - and (node.module or "").startswith("engraphis.backends") - } - assert backend_modules, "outer composition root no longer imports concrete backends" - package = _read("engraphis/__init__.py") - assert "configure_engine_factory(_default_memory_engine_factory)" in package - assert "create_memory_engine" in package - - for document in (_read("AGENTS.md"), _read("CLAUDE.md"), _read("README.md")): - normalized = " ".join(document.split()) - assert "engraphis/factory.py" in normalized - assert "outer composition root" in normalized - assert "core/engine.py" in normalized - - -def test_benchmark_text_alternatives_match_registered_fixture_boundary() -> None: - """The current image and its alt text expose only current registered boundaries.""" - registry = json.loads(_read("docs/benchmark-evidence/offline-fixtures-v73.json")) - measurements = {run["id"]: run["result"] for run in registry["runs"]} - payload = measurements["offline-performance"] - readme = _read("README.md") - svg_text = _read("docs/images/context-efficiency.svg") - svg_root = ET.fromstring(svg_text) - namespace = {"svg": "http://www.w3.org/2000/svg"} - description_node = svg_root.find("svg:desc", namespace) - assert description_node is not None - description = " ".join("".join(description_node.itertext()).lower().split()) - image = re.search( - r']+context-efficiency\.svg[^>]+alt="([^"]+)"', - readme, - flags=re.IGNORECASE, - ) - assert image is not None - alternative = " ".join(image.group(1).lower().split()) - - assert "registered deterministic fixtures" in alternative - assert "structure-aware chunks reduce retrieved context" in alternative - assert "retrieved-candidate quality is labeled separately" in alternative - assert "packed-context quality" in alternative - assert "both measured in the selected report" in alternative - assert "actual mcp transport and provider billing are not measured" in alternative - assert "740.3 to 214.3 tokens" in alternative - assert "162.2 to 42.4 tokens" in alternative - assert ( - f"{payload['compact_serialized_payload_tokens']:,} rather than " - f"{payload['full_serialized_payload_tokens']:,} tokens" - ) in alternative - - for evidence in ( - "artifact-driven local deterministic benchmark report", - "structure-aware chunks report 740.3 to 214.3 retrieved tokens per question", - "retrieved-candidate quality and packed-context quality are separate views", +from __future__ import annotations + +import ast +import json +import re +import xml.etree.ElementTree as ET +from pathlib import Path + + +from engraphis.core.schema import SCHEMA_VERSION + + +ROOT = Path(__file__).resolve().parents[1] + + +def _read(path: str) -> str: + return (ROOT / path).read_text(encoding="utf-8") + + + +def test_readme_long_description_uses_no_repository_relative_targets() -> None: + readme = _read("README.md") + destinations = re.findall( + r"!?\[[^\]]*\]\(([^) ]+)|(?:href|src)=\"([^\"]+)\"", + readme, + ) + flattened = [markdown or html for markdown, html in destinations] + relative = [ + destination + for destination in flattened + if not destination.startswith(("#", "https://", "http://")) + ] + assert not relative + + image_targets = [ + destination + for destination in flattened + if destination.endswith((".png", ".svg")) + ] + assert image_targets + assert all( + target.startswith( + "https://raw.githubusercontent.com/Coding-Dev-Tools/engraphis/main/" + ) + or target.startswith("https://img.shields.io/") + for target in image_targets + ) + +def test_canonical_offline_gate_tracks_ci() -> None: + agents = _read("AGENTS.md") + claude = _read("CLAUDE.md") + workflow = _read(".github/workflows/ci.yml") + required = ( + "ruff check .", + "python scripts/check_commercial_manifest.py", + "python scripts/externalize_dashboard_assets.py", + "python -m pytest", + "python -m eval.harness --dataset eval/datasets/sample.jsonl --k 5", + "python -m eval.harness --dataset eval/datasets/codemem.jsonl --k 5", + "python -m eval.ablation", + "python -m eval.reinforcement", + "python -m eval.adversarial_memory_security", + "python -m eval.grounded", + "python -m eval.code_arm", + "pyright", + ) + + for command in required: + assert command in agents, f"AGENTS.md omits the canonical gate command: {command}" + assert command in workflow, f"CI omits the documented gate command: {command}" + + assert "Use the exact primary offline gate in `AGENTS.md` §1" in claude + assert "do not maintain a smaller duplicate here" in claude + + +def test_core_backend_imports_stay_behind_outer_composition_root() -> None: + violations: list[str] = [] + for path in sorted((ROOT / "engraphis" / "core").glob("*.py")): + tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path)) + for node in ast.walk(tree): + if not isinstance(node, ast.ImportFrom): + continue + module = node.module or "" + if module.startswith("engraphis.backends"): + violations.append(f"{path.relative_to(ROOT)} imports {module}") + assert not violations, violations + + factory = ast.parse(_read("engraphis/factory.py"), filename="engraphis/factory.py") + backend_modules = { + node.module + for node in ast.walk(factory) + if isinstance(node, ast.ImportFrom) + and (node.module or "").startswith("engraphis.backends") + } + assert backend_modules, "outer composition root no longer imports concrete backends" + package = _read("engraphis/__init__.py") + assert "configure_engine_factory(_default_memory_engine_factory)" in package + assert "create_memory_engine" in package + + for document in (_read("AGENTS.md"), _read("CLAUDE.md"), _read("README.md")): + normalized = " ".join(document.split()) + assert "engraphis/factory.py" in normalized + assert "outer composition root" in normalized + assert "core/engine.py" in normalized + + +def test_benchmark_text_alternatives_match_registered_fixture_boundary() -> None: + """The current image and its alt text expose only current registered boundaries.""" + registry = json.loads(_read("docs/benchmark-evidence/offline-fixtures-v76.json")) + measurements = {run["id"]: run["result"] for run in registry["runs"]} + payload = measurements["offline-performance"] + readme = _read("README.md") + svg_text = _read("docs/images/context-efficiency.svg") + svg_root = ET.fromstring(svg_text) + namespace = {"svg": "http://www.w3.org/2000/svg"} + description_node = svg_root.find("svg:desc", namespace) + assert description_node is not None + description = " ".join("".join(description_node.itertext()).lower().split()) + image = re.search( + r']+context-efficiency\.svg[^>]+alt="([^"]+)"', + readme, + flags=re.IGNORECASE, + ) + assert image is not None + alternative = " ".join(image.group(1).lower().split()) + + assert "registered deterministic fixtures" in alternative + assert "structure-aware chunks reduce retrieved context" in alternative + assert "retrieved-candidate quality is labeled separately" in alternative + assert "packed-context quality" in alternative + assert "both measured in the selected report" in alternative + assert "actual mcp transport and provider billing are not measured" in alternative + assert "740.3 to 214.3 tokens" in alternative + assert "162.2 to 42.4 tokens" in alternative + assert ( + f"{payload['compact_serialized_payload_tokens']:,} rather than " + f"{payload['full_serialized_payload_tokens']:,} tokens" + ) in alternative + + for evidence in ( + "artifact-driven local deterministic benchmark report", + "structure-aware chunks report 740.3 to 214.3 retrieved tokens per question", + "retrieved-candidate quality and packed-context quality are separate views", f"{payload['full_serialized_payload_tokens']:,} full-proxy versus " f"{payload['compact_serialized_payload_tokens']:,} compact-proxy tokens", - "not an mcp transport measurement", - "does not measure provider billing", - "1,500-token cap", - ): - assert evidence in description - - for unsupported in ("unpinned", "noncanonical", "leaderboard"): - assert unsupported not in alternative - assert unsupported not in description - - for retired in ("local locomo diagnostic", "3 of 15 queries", "0 of 3 to 3 of 3"): - assert retired not in alternative - assert retired not in description - - -def test_official_longmemeval_runbook_tracks_attested_evidence_contract() -> None: - benchmarks = _read("BENCHMARKS.md") - runbook = _read("docs/PUBLIC_BENCHMARK_RUNBOOK.md") - normalized_benchmarks = " ".join(benchmarks.split()) - normalized_runbook = " ".join(runbook.split()) - - for value in ( - "balanced", - "planner", - "episodic_cap_2", - "planner_episodic_cap_2", - "context_k_2", - "planner_context_k_2", - ): - assert value in runbook - assert "30 official runs" in runbook - assert "six declared variants at all five token budgets" in normalized_benchmarks - assert "context_k=2" in runbook - - for option in ( - "--engraphis-execution-manifest", - "--engraphis-per-question", - "--engraphis-questions", - "--engraphis-haystack", - "--engraphis-trajectories", - "--engraphis-memory-config", - "--engraphis-matrix-manifest", - "--engraphis-seed", - "--execution-manifest", - "--claims-input", - ): - assert option in runbook - assert "set equality between every source question ID and output question ID" in runbook - assert "only after a successful return" in normalized_benchmarks.lower() - assert "inserted and retrieved counts by memory type" in normalized_runbook - assert "at least two inserted memory types" in normalized_runbook - - assert "does not publish per-record content fingerprints" in normalized_benchmarks - assert "whole-input/source-file digests" in normalized_runbook - assert "no raw questions, answers, prompts, context" in normalized_runbook - assert "no per-record content hashes or fingerprints" in normalized_runbook - - -def test_scope_and_event_guidance_match_fail_closed_runtime_contract() -> None: - readme = _read("README.md") - skill = _read("skills/engraphis-memory/SKILL.md") - scoping = _read("skills/engraphis-memory/references/SCOPING.md") - conventions = _read("skills/engraphis-memory/references/CONVENTIONS.md") - tools = _read("skills/engraphis-memory/references/TOOLS.md") - kilo = _read("docs/KILO_CODE_INTEGRATION.md") - - for document in (readme, skill, scoping, tools, kilo): - normalized = " ".join(document.split()) - assert "reserved and rejected" in normalized - assert "owner identity" in normalized - - for document in (conventions, tools): - normalized = " ".join(document.lower().split()) - assert "event rows are not memories" in normalized - assert "not recalled" in normalized - assert "not" in normalized and "consolidated" in normalized - - assert 'mtype="episodic"' in conventions - assert "≤0.2" in conventions - - -def test_configuration_and_recovery_guidance_matches_public_contracts() -> None: - readme = _read("README.md") - security = _read("SECURITY.md") - connect = _read("docs/AGENT_CONNECT.md") - providers = _read("docs/LLM_PROVIDERS.md") - recovery = _read("docs/RECALL_RECOVERY.md") - sync = _read("docs/SYNC.md") - - for document in (readme, security, connect, providers, sync): - normalized = " ".join(document.split()) - assert "~/.engraphis/config.env" in normalized - assert "ENGRAPHIS_ENV_FILE" in normalized - assert re.search(r"(?:never|does not) search(?:es)? the working directory", normalized) - - assert "repaired_fields" in recovery - assert "v1_memory_id" in recovery - assert "v1_thought_id" in recovery - assert "v1_document_id" in recovery - assert "first contact" in sync - assert "incomplete" in sync - assert "unanchored" in sync - assert "--relay-token" in sync and "--relay-e2ee-key" in sync - assert "intentionally has no secret-valued" in sync - - - -def test_schema_and_erasure_docs_match_live_export_policy() -> None: - agents = _read("AGENTS.md") - readme = _read("README.md") - changelog = _read("CHANGELOG.md") - sync = _read("docs/SYNC.md") - erasure = _read("docs/SECURE_ERASURE.md") - schema = _read("engraphis/core/schema.py") - - assert f"SCHEMA_VERSION = {SCHEMA_VERSION}" in schema - assert agents.count(f"`SCHEMA_VERSION = {SCHEMA_VERSION}`") == 2 - assert f"schema {SCHEMA_VERSION}" in readme - assert f"schema {SCHEMA_VERSION}" in changelog - - for document in (agents, readme, changelog, sync, erasure): - normalized = " ".join(document.split()) - assert "never_export" in normalized - assert "remote_erasure" in normalized - - normalized_sync = " ".join(sync.split()) - assert "only `remote_erasure`" in normalized_sync - assert "never leave the device" in normalized_sync - assert "cannot later be upgraded" in normalized_sync - assert "only a non-secret workspace/repo record" in erasure - - -def test_document_import_docs_describe_the_source_neutral_contract() -> None: - readme = _read("README.md") - agents = _read("AGENTS.md") - guide = _read("docs/DOCUMENT_IMPORT.md") - obsidian = _read("docs/OBSIDIAN_IMPORT.md") - - for document in (readme, guide): - assert "engraphis import documents" in document - assert "--dry-run" in document - assert "--yes" in document - for format_name in ( - "Markdown", "reStructuredText", "HTML", "JSON", "CSV", "DOCX", "ODT", - "RTF", "XLSX", "ODS", "PPTX", "ODP", "EPUB", "Source code", - ): - assert format_name in guide - for safety_term in ("symlink", "secret", "unsupported", "resumable", "temporal", "conflict"): - assert safety_term in guide - assert f"SCHEMA_VERSION = {SCHEMA_VERSION}" in agents - assert "source-neutral" in agents - assert "rich Markdown adapter" in obsidian - assert "DOCUMENT_IMPORT.md" in obsidian - - -def test_consolidation_docs_expose_only_live_public_options() -> None: - readme = _read("README.md") - tools = _read("skills/engraphis-memory/references/TOOLS.md") - changelog = _read("CHANGELOG.md") - - for document in (readme, tools, changelog): - assert "supersede_sources" not in document - assert "supersede-sources" not in document - - assert "source episodes remain live" in readme - normalized_tools = " ".join(tools.split()) - assert "`profiles (bool, false)`; `structured (bool, false)`." in normalized_tools + "not an mcp transport measurement", + "does not measure provider billing", + "1,500-token cap", + ): + assert evidence in description + + for unsupported in ("unpinned", "noncanonical", "leaderboard"): + assert unsupported not in alternative + assert unsupported not in description + + for retired in ("local locomo diagnostic", "3 of 15 queries", "0 of 3 to 3 of 3"): + assert retired not in alternative + assert retired not in description + + +def test_official_longmemeval_runbook_tracks_attested_evidence_contract() -> None: + benchmarks = _read("BENCHMARKS.md") + runbook = _read("docs/PUBLIC_BENCHMARK_RUNBOOK.md") + normalized_benchmarks = " ".join(benchmarks.split()) + normalized_runbook = " ".join(runbook.split()) + + for value in ( + "balanced", + "planner", + "episodic_cap_2", + "planner_episodic_cap_2", + "context_k_2", + "planner_context_k_2", + ): + assert value in runbook + assert "30 official runs" in runbook + assert "six declared variants at all five token budgets" in normalized_benchmarks + assert "context_k=2" in runbook + + for option in ( + "--engraphis-execution-manifest", + "--engraphis-per-question", + "--engraphis-questions", + "--engraphis-haystack", + "--engraphis-trajectories", + "--engraphis-memory-config", + "--engraphis-matrix-manifest", + "--engraphis-seed", + "--execution-manifest", + "--claims-input", + ): + assert option in runbook + assert "set equality between every source question ID and output question ID" in runbook + assert "only after a successful return" in normalized_benchmarks.lower() + assert "inserted and retrieved counts by memory type" in normalized_runbook + assert "at least two inserted memory types" in normalized_runbook + + assert "does not publish per-record content fingerprints" in normalized_benchmarks + assert "whole-input/source-file digests" in normalized_runbook + assert "no raw questions, answers, prompts, context" in normalized_runbook + assert "no per-record content hashes or fingerprints" in normalized_runbook + + +def test_scope_and_event_guidance_match_fail_closed_runtime_contract() -> None: + readme = _read("README.md") + skill = _read("skills/engraphis-memory/SKILL.md") + scoping = _read("skills/engraphis-memory/references/SCOPING.md") + conventions = _read("skills/engraphis-memory/references/CONVENTIONS.md") + tools = _read("skills/engraphis-memory/references/TOOLS.md") + kilo = _read("docs/KILO_CODE_INTEGRATION.md") + + for document in (readme, skill, scoping, tools, kilo): + normalized = " ".join(document.split()) + assert "reserved and rejected" in normalized + assert "owner identity" in normalized + + for document in (conventions, tools): + normalized = " ".join(document.lower().split()) + assert "event rows are not memories" in normalized + assert "not recalled" in normalized + assert "not" in normalized and "consolidated" in normalized + + assert 'mtype="episodic"' in conventions + assert "≤0.2" in conventions + + +def test_configuration_and_recovery_guidance_matches_public_contracts() -> None: + readme = _read("README.md") + security = _read("SECURITY.md") + connect = _read("docs/AGENT_CONNECT.md") + providers = _read("docs/LLM_PROVIDERS.md") + recovery = _read("docs/RECALL_RECOVERY.md") + sync = _read("docs/SYNC.md") + + for document in (readme, security, connect, providers, sync): + normalized = " ".join(document.split()) + assert "~/.engraphis/config.env" in normalized + assert "ENGRAPHIS_ENV_FILE" in normalized + assert re.search(r"(?:never|does not) search(?:es)? the working directory", normalized) + + assert "repaired_fields" in recovery + assert "v1_memory_id" in recovery + assert "v1_thought_id" in recovery + assert "v1_document_id" in recovery + assert "first contact" in sync + assert "incomplete" in sync + assert "unanchored" in sync + assert "--relay-token" in sync and "--relay-e2ee-key" in sync + assert "intentionally has no secret-valued" in sync + + + +def test_schema_and_erasure_docs_match_live_export_policy() -> None: + agents = _read("AGENTS.md") + readme = _read("README.md") + changelog = _read("CHANGELOG.md") + sync = _read("docs/SYNC.md") + erasure = _read("docs/SECURE_ERASURE.md") + schema = _read("engraphis/core/schema.py") + + assert f"SCHEMA_VERSION = {SCHEMA_VERSION}" in schema + assert agents.count(f"`SCHEMA_VERSION = {SCHEMA_VERSION}`") == 2 + assert f"schema {SCHEMA_VERSION}" in readme + assert f"schema {SCHEMA_VERSION}" in changelog + + for document in (agents, readme, changelog, sync, erasure): + normalized = " ".join(document.split()) + assert "never_export" in normalized + assert "remote_erasure" in normalized + + normalized_sync = " ".join(sync.split()) + assert "only `remote_erasure`" in normalized_sync + assert "never leave the device" in normalized_sync + assert "cannot later be upgraded" in normalized_sync + assert "only a non-secret workspace/repo record" in erasure + + +def test_document_import_docs_describe_the_source_neutral_contract() -> None: + readme = _read("README.md") + agents = _read("AGENTS.md") + guide = _read("docs/DOCUMENT_IMPORT.md") + obsidian = _read("docs/OBSIDIAN_IMPORT.md") + + for document in (readme, guide): + assert "engraphis import documents" in document + assert "--dry-run" in document + assert "--yes" in document + for format_name in ( + "Markdown", "reStructuredText", "HTML", "JSON", "CSV", "DOCX", "ODT", + "RTF", "XLSX", "ODS", "PPTX", "ODP", "EPUB", "Source code", + ): + assert format_name in guide + for safety_term in ("symlink", "secret", "unsupported", "resumable", "temporal", "conflict"): + assert safety_term in guide + assert f"SCHEMA_VERSION = {SCHEMA_VERSION}" in agents + assert "source-neutral" in agents + assert "rich Markdown adapter" in obsidian + assert "DOCUMENT_IMPORT.md" in obsidian + + +def test_consolidation_docs_expose_only_live_public_options() -> None: + readme = _read("README.md") + tools = _read("skills/engraphis-memory/references/TOOLS.md") + changelog = _read("CHANGELOG.md") + + for document in (readme, tools, changelog): + assert "supersede_sources" not in document + assert "supersede-sources" not in document + + assert "source episodes remain live" in readme + normalized_tools = " ".join(tools.split()) + assert "`profiles (bool, false)`; `structured (bool, false)`." in normalized_tools diff --git a/tests/test_local_benchmark_queue.py b/tests/test_local_benchmark_queue.py index 27229c57..d07bc175 100644 --- a/tests/test_local_benchmark_queue.py +++ b/tests/test_local_benchmark_queue.py @@ -366,6 +366,7 @@ def runner(*_args, **_kwargs): _WAIT_QUEUE = r""" import json import sys +import time from pathlib import Path from types import SimpleNamespace @@ -395,6 +396,22 @@ def runner(*_args, **_kwargs): queue._verified_artifact = verify +if len(sys.argv) == 9: + paused, resume = Path(sys.argv[7]), Path(sys.argv[8]) + original_probe = queue._prerequisite_ready + + def probe_between_barriers(*args): + ready = original_probe(*args) + if not ready and not paused.exists(): + paused.write_text("paused", encoding="utf-8") + deadline = time.monotonic() + 20 + while not resume.exists(): + if time.monotonic() >= deadline: + raise TimeoutError("parent did not release the between-probe barrier") + time.sleep(0.01) + return ready + + queue._prerequisite_ready = probe_between_barriers try: queue.execute(plan, directory, runner=runner, poll_seconds=0.02, wait_timeout=60, default_timeout_seconds=60) @@ -488,10 +505,11 @@ def _start_owner(plan_path, result_dir, ready, calls, error): ) -def _start_wait_queue(plan_path, result_dir, started, verified, done, error): +def _start_wait_queue(plan_path, result_dir, started, verified, done, error, *, barrier=None): return subprocess.Popen( [sys.executable, "-c", _WAIT_QUEUE, str(result_dir), str(plan_path), - str(started), str(verified), str(done), str(error)], + str(started), str(verified), str(done), str(error)] + + ([] if barrier is None else [str(path) for path in barrier]), cwd=Path.cwd(), stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, **_SUBPROCESS_OPTIONS, ) @@ -664,6 +682,7 @@ def test_queue_waits_for_legacy_ephemeral_producer_marker_until_removed(tmp_path verified = tmp_path / "verified" done = tmp_path / "done" queue_error = tmp_path / "queue-error" + paused, resume = tmp_path / "probe-paused", tmp_path / "probe-resume" queue_plan = _wait_plan(producer_lock, artifact) plan_path = tmp_path / "wait-plan.json" plan_path.write_text(json.dumps(queue_plan), encoding="utf-8") @@ -671,12 +690,14 @@ def test_queue_waits_for_legacy_ephemeral_producer_marker_until_removed(tmp_path try: queue_process = _start_wait_queue( plan_path, tmp_path / "results", queue_ready, verified, done, queue_error, + barrier=(paused, resume), ) _wait_for_marker(queue_ready, queue_process) - _wait_for_prerequisite_status(tmp_path / "results", queue_process) + _wait_for_marker(paused, queue_process) assert queue_process.poll() is None assert not verified.exists() producer_lock.unlink() + resume.write_text("resume", encoding="utf-8") _wait_for_marker(done, queue_process) queue_process.wait(timeout=10) stdout, stderr = queue_process.communicate(timeout=10) @@ -690,7 +711,9 @@ def test_queue_waits_for_legacy_ephemeral_producer_marker_until_removed(tmp_path queue_process.communicate(timeout=10) -def test_queue_accepts_legacy_marker_removed_during_probe(tmp_path, monkeypatch): +def test_queue_retries_a_verified_legacy_removal_without_verifying_during_the_probe(tmp_path, monkeypatch): + from eval.external_checkpoints import LegacyRunnerLockRemoved + artifact = tmp_path / "diagnostic.json" artifact.write_text("{}", encoding="utf-8") producer_lock = tmp_path / "legacy-producer.lock" @@ -701,12 +724,35 @@ def disappearing_probe(path, *, create=True): assert path == producer_lock assert not create producer_lock.unlink() - raise ValueError("external diagnostic runner lock is unsafe or changed") + raise LegacyRunnerLockRemoved("legacy external runner marker was removed; probe again") yield # pragma: no cover - the probe always raises monkeypatch.setattr(queue, "_runner_lock", disappearing_probe) - monkeypatch.setattr(queue, "_verified_artifact", lambda path: {"verified": True}) - assert queue._prerequisite_ready(artifact, producer_lock) + monkeypatch.setattr(queue, "_verified_artifact", lambda path: pytest.fail("verified during changed probe")) + assert not queue._prerequisite_ready(artifact, producer_lock) + + +@pytest.mark.parametrize("error", [PermissionError("denied"), OSError("unreadable")]) +def test_queue_does_not_treat_an_unreadable_marker_as_removed(tmp_path, monkeypatch, error): + artifact = tmp_path / "diagnostic.json" + artifact.write_text("{}", encoding="utf-8") + producer_lock = tmp_path / "legacy-producer.lock" + producer_lock.write_text("running", encoding="utf-8") + + original_lstat = Path.lstat + calls = [] + + def unreadable_marker(path, *args, **kwargs): + if Path(path) == producer_lock: + calls.append(path) + if len(calls) == 2: + raise error + return original_lstat(path, *args, **kwargs) + + monkeypatch.setattr(Path, "lstat", unreadable_marker) + monkeypatch.setattr(queue, "_verified_artifact", lambda path: pytest.fail("unsafe producer was accepted")) + with pytest.raises(ValueError, match="unsafe or changed"): + queue._prerequisite_ready(artifact, producer_lock) def _complete_capacity_summary(): diff --git a/tests/test_queue_prerequisite_lock.py b/tests/test_queue_prerequisite_lock.py index b3970da0..735a813c 100644 --- a/tests/test_queue_prerequisite_lock.py +++ b/tests/test_queue_prerequisite_lock.py @@ -1,6 +1,7 @@ """Prerequisite verification must retain producer ownership throughout the read.""" import pytest +import os from pathlib import Path from eval import local_benchmark_queue as queue @@ -171,3 +172,34 @@ def replace_before_open(path, mode="r", *args, **kwargs): assert verified == [] assert marker.read_bytes() == marker_before assert artifact.read_bytes() == artifact_before + + +@pytest.mark.skipif(os.name == "nt", reason="requires unlink of an open file") +@pytest.mark.parametrize("payload", [b"12345", RUNNER_LOCK_MARKER, RUNNER_LOCK_MARKER[:8], b"", b"running", b"0"]) +def test_postopen_unlink_retries_only_a_verified_legacy_pid_inode(tmp_path, monkeypatch, payload): + marker = tmp_path / ".runner.lock" + marker.write_bytes(payload) + artifact = tmp_path / "artifact.json" + artifact.write_bytes(b"retained-artifact") + original_lstat = Path.lstat + inspections = [] + + def unlink_opened_marker(path, *args, **kwargs): + if path == marker: + inspections.append(path) + if len(inspections) == 2: + marker.unlink() + return original_lstat(path, *args, **kwargs) + + verified = [] + monkeypatch.setattr(Path, "lstat", unlink_opened_marker) + monkeypatch.setattr(queue, "_verified_artifact", lambda path: verified.append(path)) + if payload == b"12345": + assert not queue._prerequisite_ready(artifact, marker) + assert verified == [] + assert queue._prerequisite_ready(artifact, marker) + assert verified == [artifact] + else: + with pytest.raises(ValueError, match="unsafe or changed"): + queue._prerequisite_ready(artifact, marker) + assert verified == [] diff --git a/tests/test_release_qualification.py b/tests/test_release_qualification.py index 59c71324..dd995ace 100644 --- a/tests/test_release_qualification.py +++ b/tests/test_release_qualification.py @@ -190,6 +190,15 @@ def test_publication_writes_require_qualification_except_scoped_v176_waiver(): checked = False writes = 0 for step in job["steps"]: + if step.get("name") == "Disclose the qualification waiver before PyPI repair": + # This conditional write publishes only the exception notice. The + # ordinary path still needs its signature before any distribution. + assert name == "github-release-repair" + assert step.get("if") == "inputs.waive_v176_qualification" + assert "verified-dist/*" not in step["run"] + assert "release-evidence/*" not in step["run"] + assert 'test "$ENGRAPHIS_REPAIR_COMMIT" = "6a441a75c8dd159607fa3933da83f600864b9146"' in step["run"] + continue if "scripts.verify_release_qualification" in step.get("run", ""): if name == "github-release-repair": assert step.get("if") == "${{ !inputs.waive_v176_qualification }}" @@ -222,6 +231,11 @@ def test_publication_writes_require_qualification_except_scoped_v176_waiver(): if step.get("name") == "Enforce and record the v1.7.6-only qualification waiver") assert waiver_guard.get("if") == "inputs.waive_v176_qualification" assert 'test "$RELEASE_TAG" = "v1.7.6"' in waiver_guard["run"] + disclosure = next(step for step in repair_steps + if step.get("name") == "Disclose the qualification waiver before PyPI repair") + publication = next(step for step in repair_steps + if step.get("name") == "Publish only missing verified distributions") + assert repair_steps.index(disclosure) < repair_steps.index(publication) repair = next(step for step in repair_steps if step.get("name") == "Repair GitHub Release") assert repair["env"]["WAIVE_QUALIFICATION"] == "${{ inputs.waive_v176_qualification }}" assert repair["run"].index("gh release edit") < repair["run"].index("gh release upload") @@ -269,7 +283,8 @@ def test_waiver_disclosure_cannot_follow_github_publication(tmp_path, existing, calls = calls_path.read_text(encoding="utf-8").splitlines() assert result.returncode == (7 if edit_fails else 0), result.stderr if existing: - edits = [index for index, call in enumerate(calls) if call.startswith("release edit ")] + edits = [index for index, call in enumerate(calls) + if call.startswith("release edit ") and "--notes-file " in call] uploads = [index for index, call in enumerate(calls) if call.startswith("release upload ")] assert len(edits) == 1 assert not uploads if edit_fails else len(uploads) == 1 and edits[0] < uploads[0] @@ -282,3 +297,85 @@ def test_waiver_disclosure_cannot_follow_github_publication(tmp_path, existing, notes = (tmp_path / "release-waiver.md").read_text(encoding="utf-8") assert "Mandatory full-product gates are not represented as passed." in notes assert "https://example.test/run/1" in notes + + +@pytest.mark.skipif(os.name == "nt", reason="release workflow executes in Linux bash") +@pytest.mark.parametrize("existing,edit_fails", [(False, False), (True, False), (True, True)]) +def test_public_waiver_notice_precedes_pypi_even_if_later_repair_fails(tmp_path, existing, edit_fails): + yaml = pytest.importorskip("yaml") + bash = shutil.which("bash") + if bash is None: + pytest.skip("bash is unavailable") + root = Path(__file__).resolve().parents[1] + workflow = yaml.safe_load((root / ".github/workflows/release.yml").read_text(encoding="utf-8")) + disclosure = next(step for step in workflow["jobs"]["github-release-repair"]["steps"] + if step.get("name") == "Disclose the qualification waiver before PyPI repair") + executable = tmp_path / "gh" + executable.write_text("""#!/usr/bin/env bash +set -euo pipefail +printf '%s\\n' "$*" >> "$GH_CALLS" +case "$2" in + view) + if [ "$EXISTING" != true ]; then exit 1; fi + printf 'Existing release notes\\n' + ;; + edit) + if [ "$EDIT_FAILS" = true ]; then exit 7; fi + ;; +esac +""", encoding="utf-8") + executable.chmod(0o700) + script = tmp_path / "disclose.sh" + # A later publication/verification failure must leave the already public + # notice in place; a failed notice write must prevent publication altogether. + script.write_text(disclosure["run"] + '\nprintf "pypi-publication\\n" >> "$GH_CALLS"\nexit 9\n', + encoding="utf-8") + calls_path = tmp_path / "calls.txt" + environment = {**os.environ, "PATH": str(tmp_path) + os.pathsep + os.environ["PATH"], + "RUNNER_TEMP": str(tmp_path), "RELEASE_TAG": "v1.7.6", "GH_REPO": "test/repo", + "ENGRAPHIS_REPAIR_COMMIT": "6a441a75c8dd159607fa3933da83f600864b9146", + "GH_RUN_URL": "https://example.test/run/1", "GH_CALLS": str(calls_path), + "EXISTING": str(existing).lower(), "EDIT_FAILS": str(edit_fails).lower()} + result = subprocess.run([bash, str(script)], cwd=tmp_path, capture_output=True, text=True, + timeout=20, env=environment) + calls = calls_path.read_text(encoding="utf-8").splitlines() + assert result.returncode == (7 if edit_fails else 9), result.stderr + notice = next(index for index, call in enumerate(calls) if "--notes-file " in call) + if edit_fails: + assert "pypi-publication" not in calls + else: + assert notice < calls.index("pypi-publication") + assert not any("verified-dist/" in call or "release-evidence/" in call for call in calls) + if not existing: + assert "--latest=false" in calls[notice] + notes = (tmp_path / "release-waiver.md").read_text(encoding="utf-8") + assert "Mandatory full-product gates are not represented as passed." in notes + assert environment["ENGRAPHIS_REPAIR_COMMIT"] in notes + + +@pytest.mark.skipif(os.name == "nt", reason="release workflow executes in Linux bash") +@pytest.mark.parametrize("tag,commit", [("v1.7.7", "6a441a75c8dd159607fa3933da83f600864b9146"), + ("v1.7.6", "a" * 40), ("v1.7.6", "")]) +def test_waiver_rejects_a_different_retained_candidate_before_any_public_write(tmp_path, tag, commit): + yaml = pytest.importorskip("yaml") + bash = shutil.which("bash") + if bash is None: + pytest.skip("bash is unavailable") + root = Path(__file__).resolve().parents[1] + workflow = yaml.safe_load((root / ".github/workflows/release.yml").read_text(encoding="utf-8")) + disclosure = next(step for step in workflow["jobs"]["github-release-repair"]["steps"] + if step.get("name") == "Disclose the qualification waiver before PyPI repair") + executable = tmp_path / "gh" + executable.write_text('#!/usr/bin/env bash\nprintf "unexpected call\\n" >> "$GH_CALLS"\n', + encoding="utf-8") + executable.chmod(0o700) + script = tmp_path / "disclose.sh" + script.write_text(disclosure["run"], encoding="utf-8") + calls_path = tmp_path / "calls.txt" + result = subprocess.run([bash, str(script)], cwd=tmp_path, capture_output=True, text=True, + timeout=20, env={**os.environ, "PATH": str(tmp_path) + os.pathsep + os.environ["PATH"], + "RUNNER_TEMP": str(tmp_path), "RELEASE_TAG": tag, + "ENGRAPHIS_REPAIR_COMMIT": commit, "GH_REPO": "test/repo", + "GH_CALLS": str(calls_path)}) + assert result.returncode != 0 + assert not calls_path.exists() From 748e944a02ffc3878132d72f72cc9a76a0897e57 Mon Sep 17 00:00:00 2001 From: Coding-Dev-Tools Date: Sun, 27 Sep 2026 02:33:28 -0400 Subject: [PATCH 5/6] chore: preserve evidence reference file formatting --- BENCHMARKS.md | 770 ++++---- README.md | 1772 ++++++++--------- tests/test_benchmark_evidence.py | 2540 ++++++++++++------------- tests/test_documentation_contracts.py | 618 +++--- 4 files changed, 2850 insertions(+), 2850 deletions(-) diff --git a/BENCHMARKS.md b/BENCHMARKS.md index f8eadfa9..879bc630 100644 --- a/BENCHMARKS.md +++ b/BENCHMARKS.md @@ -1,15 +1,15 @@ -# Benchmarks - -This guide explains what Engraphis measures, how to reproduce each evaluation, and the limits of -those results. When this document and the code disagree, the code is the source of truth. - -The current expansion has a separate [results and workload report](docs/BENCHMARK_EXPANSION_RESULTS.md), -[execution runbook](docs/BENCHMARK_EXPANSION_RUNBOOK.md), and -[proposed stage budgets](docs/BENCHMARK_STAGE_BUDGETS.md). Completed external retrieval diagnostics +# Benchmarks + +This guide explains what Engraphis measures, how to reproduce each evaluation, and the limits of +those results. When this document and the code disagree, the code is the source of truth. + +The current expansion has a separate [results and workload report](docs/BENCHMARK_EXPANSION_RESULTS.md), +[execution runbook](docs/BENCHMARK_EXPANSION_RUNBOOK.md), and +[proposed stage budgets](docs/BENCHMARK_STAGE_BUDGETS.md). Completed external retrieval diagnostics are review artifacts with explicit denominators and uncertainty. The coding pilot uses Codex OAuth only and retains fixture exclusions and interrupted calls. Official QA, competitor scores and capacity qualification remain separate experiments. - + For the locked operator sequence for a public canonical run, see [`docs/PUBLIC_BENCHMARK_RUNBOOK.md`](docs/PUBLIC_BENCHMARK_RUNBOOK.md). @@ -21,10 +21,10 @@ Smart/Classic/service write paths, and `retrieval_recipe="conversation"` or `"lo for the measured depth/budget starting points. `"legacy"` packing and `"default"` retrieval remain the defaults until development, validation and untouched-holdout gates show a workload-specific benefit. - -`python -m eval.evidence_contracts` checks exact-action validation and compares -legacy and coverage packing on small deterministic development fixtures. It runs -in the full offline CI matrix and the NumPy-only Python 3.9 job. These fixtures + +`python -m eval.evidence_contracts` checks exact-action validation and compares +legacy and coverage packing on small deterministic development fixtures. It runs +in the full offline CI matrix and the NumPy-only Python 3.9 job. These fixtures test boundary correctness; they do not estimate external QA or model task success. The [current diagnostic](docs/benchmark-evidence/evidence-contracts-20260921-v11.json) withholds the unchanged oversized, unpunctuated fixture at its 24-token budget: @@ -92,7 +92,7 @@ retain `unknown` provenance unless explicitly bound. These checks protect result interpretation and do not count as additional benchmark-quality gains. ### Public numeric evidence registry - + Every exact public aggregate retained below comes from the checked-in, public-safe [`offline-fixtures-v76.json`](docs/benchmark-evidence/offline-fixtures-v76.json) artifact. Its SHA-256 is @@ -102,34 +102,34 @@ or per-record content fingerprints. The fixture-suite digest is `20131c25c86e3ac5a53285d0631d4ff0c60946879b04ff7a7c38e488f42cda51`. The artifact defines -the digest algorithm and records the SHA-256 of every suite and dataset file. Each evidence ID -also binds its exact command through `sha256(UTF-8 exact command)`: - -| Evidence ID | Exact command | Config digest | -|---|---|---| -| `offline-chunking` | `python -m eval.chunking_eval --dataset eval/datasets/longdoc.jsonl --k 5` | `c1c8196aa7e1568ef3844a9fb2d76b87f342c39108e32d6ad144b885a76143b8` | -| `offline-performance` | `python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 --json` | `bbe4aca81e58d4830e50a8fc7729a1d15b71d97a6299bccd79432b7f119677d7` | -| `offline-grounded` | `python -m eval.grounded` | `590442e51e3642c10489165759919dc86ffac62c182937330c153e7f8d5fc26f` | - -External, model-dependent, latency, consolidation, and productivity numbers are not included in -this offline registry unless a redacted immutable artifact with the same three bindings exists. Use -the [public benchmark runbook](docs/PUBLIC_BENCHMARK_RUNBOOK.md) to produce registry evidence. -Completed retrieval-only diagnostics are documented separately in the -[benchmark expansion results](docs/BENCHMARK_EXPANSION_RESULTS.md); absence from this registry -means no number is claimed in this offline registry. - -The context-efficiency chart is generated from the registry values and the selected report schema. -Historical LoCoMo, graph, handoff, consolidation, and security figures remain preserved in their -source artifacts but are omitted from the current chart until each has a matching immutable, -public-safe artifact. The chart labels coding outcomes, external datasets, and operational -capacity as pending evaluation tracks rather than implying scores. Regenerate it with +the digest algorithm and records the SHA-256 of every suite and dataset file. Each evidence ID +also binds its exact command through `sha256(UTF-8 exact command)`: + +| Evidence ID | Exact command | Config digest | +|---|---|---| +| `offline-chunking` | `python -m eval.chunking_eval --dataset eval/datasets/longdoc.jsonl --k 5` | `c1c8196aa7e1568ef3844a9fb2d76b87f342c39108e32d6ad144b885a76143b8` | +| `offline-performance` | `python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 --json` | `bbe4aca81e58d4830e50a8fc7729a1d15b71d97a6299bccd79432b7f119677d7` | +| `offline-grounded` | `python -m eval.grounded` | `590442e51e3642c10489165759919dc86ffac62c182937330c153e7f8d5fc26f` | + +External, model-dependent, latency, consolidation, and productivity numbers are not included in +this offline registry unless a redacted immutable artifact with the same three bindings exists. Use +the [public benchmark runbook](docs/PUBLIC_BENCHMARK_RUNBOOK.md) to produce registry evidence. +Completed retrieval-only diagnostics are documented separately in the +[benchmark expansion results](docs/BENCHMARK_EXPANSION_RESULTS.md); absence from this registry +means no number is claimed in this offline registry. + +The context-efficiency chart is generated from the registry values and the selected report schema. +Historical LoCoMo, graph, handoff, consolidation, and security figures remain preserved in their +source artifacts but are omitted from the current chart until each has a matching immutable, +public-safe artifact. The chart labels coding outcomes, external datasets, and operational +capacity as pending evaluation tracks rather than implying scores. Regenerate it with `python scripts/render_benchmark_report.py --report docs/benchmark-evidence/offline-fixtures-v76.json --output docs/images/context-efficiency.svg` after selecting the report to publish. The companion examples are also generated from that artifact with `python -m scripts.render_benchmark_examples --report docs/benchmark-evidence/offline-fixtures-v76.json --output docs/images/evidence-backed-agent-examples.svg`. -The historical-to-executable mapping is in -[`docs/BENCHMARK_CHANGE_COVERAGE.md`](docs/BENCHMARK_CHANGE_COVERAGE.md). - +The historical-to-executable mapping is in +[`docs/BENCHMARK_CHANGE_COVERAGE.md`](docs/BENCHMARK_CHANGE_COVERAGE.md). + Fresh diagnostics retain explicit source-case identities so confidence intervals cluster whole conversations even when question IDs do not encode their case. Metrics with no eligible questions remain `null` (unscored), including fresh and @@ -137,176 +137,176 @@ resumed runs. Artifact parsing, checksum validation, and queue receipts bind the same byte snapshot. Comparisons require consistent dataset and repair bindings while allowing the producer implementation to change between versions. -## What we measure today (all offline, no API key) - -Most Engraphis evals score **retrieval**, not end-to-end QA. The separate productivity benchmark -runs a complete offline agent attempt and correction loop, but it is not an official -frontier-model QA score. - -- **Correctness gate**: `eval/harness.py` over `eval/datasets/sample.jsonl` and - `codemem.jsonl` (conflict resolution) and `graph_multihop.jsonl` (multi-hop graph recall). - Runs on the deterministic embedder, so it is a plumbing/regression floor, not a public - performance claim. This is the gate CI enforces. -- **Ablation**: `eval/ablation.py`: vector-only vs. 1-hop graph vs. Personalized-PageRank arm, - to show the graph arm actually earns its place. -- **External benchmarks**: `eval/external.py` loads **LoCoMo** and **LongMemEval** and pushes - them through the *real* `MemoryEngine` write path (conflict resolution + evolution) and hybrid - recall with a real sentence-transformers embedder. It reports `recall_at_k` / `hit_at_k` / - `answer_token_recall`: i.e. *did the evidence come back*, not *did an LLM answer correctly*. - It retains source categories and abstention/no-evidence questions as explicit exclusions from - retrieval-only aggregates rather than silently dropping them. `eval.longmemeval_v2` is a local, - text-only adapter for the official LongMemEval-V2 `insert(trajectory)` / `query(query, - query_image=None)` memory interface; it does not download data or call a model. -- **Grounded**: `eval/grounded.py`: answerable → cite, off-topic → abstain. Exact fixture - outcomes are evidence ID `offline-grounded` in the registry above. -- **Chunking (quality per token)**: `eval/chunking_eval.py` over `eval/datasets/longdoc.jsonl` - ingests a multi-topic corpus twice: once as one memory per document (`whole`) and once with - sub-file `ChunkingExtractor` (`chunked`), then queries both through the real recall pipeline. - The checked-in corpus is explicitly marked trusted eval data so the measurement isolates - chunking from the production trust gate, which excludes arbitrary raw imports from normal - agent context. On the deterministic embedder, **recall@5 is 1.000 for both modes; mean - retrieved top-5 content falls from 740.3 to 214.3 tokens (526.0 fewer, 71.1% lower, about - 3.5× smaller), while the smallest returned evidence-holding memory falls from 162.2 to 42.4 - tokens (119.8 fewer, 73.9% lower, about 3.8× smaller).** These aggregates are evidence ID - `offline-chunking` in the registry above. Pass `--embed-model - sentence-transformers/all-MiniLM-L6-v2` to run a model-dependent experiment; do not publish - that result without a new immutable artifact and pinned model revision. -- **Full-pipeline latency + quality**: `eval/performance.py` times the shipped semantic + - lexical + graph + fusion + scoring + rerank + packing path after warmup, with reinforcement - disabled so repeated measurements do not mutate their corpus. It reports p50/p95/p99 latency, - retrieval quality, packed context tokens, and full/compact JSON-shape payload proxies in one - JSON-safe schema. Its legacy `quality` fields score all candidate chunks returned before - context packing; additive `packed_quality` fields score only chunks admitted to reader context. - Payload proxies are sampled once per question, independently of the number of timed iterations; - they are not serialized MCP envelopes or transport responses. In the +## What we measure today (all offline, no API key) + +Most Engraphis evals score **retrieval**, not end-to-end QA. The separate productivity benchmark +runs a complete offline agent attempt and correction loop, but it is not an official +frontier-model QA score. + +- **Correctness gate**: `eval/harness.py` over `eval/datasets/sample.jsonl` and + `codemem.jsonl` (conflict resolution) and `graph_multihop.jsonl` (multi-hop graph recall). + Runs on the deterministic embedder, so it is a plumbing/regression floor, not a public + performance claim. This is the gate CI enforces. +- **Ablation**: `eval/ablation.py`: vector-only vs. 1-hop graph vs. Personalized-PageRank arm, + to show the graph arm actually earns its place. +- **External benchmarks**: `eval/external.py` loads **LoCoMo** and **LongMemEval** and pushes + them through the *real* `MemoryEngine` write path (conflict resolution + evolution) and hybrid + recall with a real sentence-transformers embedder. It reports `recall_at_k` / `hit_at_k` / + `answer_token_recall`: i.e. *did the evidence come back*, not *did an LLM answer correctly*. + It retains source categories and abstention/no-evidence questions as explicit exclusions from + retrieval-only aggregates rather than silently dropping them. `eval.longmemeval_v2` is a local, + text-only adapter for the official LongMemEval-V2 `insert(trajectory)` / `query(query, + query_image=None)` memory interface; it does not download data or call a model. +- **Grounded**: `eval/grounded.py`: answerable → cite, off-topic → abstain. Exact fixture + outcomes are evidence ID `offline-grounded` in the registry above. +- **Chunking (quality per token)**: `eval/chunking_eval.py` over `eval/datasets/longdoc.jsonl` + ingests a multi-topic corpus twice: once as one memory per document (`whole`) and once with + sub-file `ChunkingExtractor` (`chunked`), then queries both through the real recall pipeline. + The checked-in corpus is explicitly marked trusted eval data so the measurement isolates + chunking from the production trust gate, which excludes arbitrary raw imports from normal + agent context. On the deterministic embedder, **recall@5 is 1.000 for both modes; mean + retrieved top-5 content falls from 740.3 to 214.3 tokens (526.0 fewer, 71.1% lower, about + 3.5× smaller), while the smallest returned evidence-holding memory falls from 162.2 to 42.4 + tokens (119.8 fewer, 73.9% lower, about 3.8× smaller).** These aggregates are evidence ID + `offline-chunking` in the registry above. Pass `--embed-model + sentence-transformers/all-MiniLM-L6-v2` to run a model-dependent experiment; do not publish + that result without a new immutable artifact and pinned model revision. +- **Full-pipeline latency + quality**: `eval/performance.py` times the shipped semantic + + lexical + graph + fusion + scoring + rerank + packing path after warmup, with reinforcement + disabled so repeated measurements do not mutate their corpus. It reports p50/p95/p99 latency, + retrieval quality, packed context tokens, and full/compact JSON-shape payload proxies in one + JSON-safe schema. Its legacy `quality` fields score all candidate chunks returned before + context packing; additive `packed_quality` fields score only chunks admitted to reader context. + Payload proxies are sampled once per question, independently of the number of timed iterations; + they are not serialized MCP envelopes or transport responses. In the registered CodeMem run, 26 payload samples total **24,590** full-proxy `engraphis.regex.v1` tokens versus **11,138** compact-proxy tokens, avoiding **13,452** proxy tokens (**54.71% lower**), while 260 recalls are timed. Packed context across the same 26 - samples averages **85.38** tokens and reaches **108** under a 1,500-token cap; Recall@5, - hit@5, and answer-token recall remain 1.000 for the legacy candidate-page view. The registered - v9 artifact predates `packed_quality`, so no packed-quality aggregate is published from it. - These aggregates are evidence ID - `offline-performance` in the registry above. `--filler-memories`, `--candidate-k`, and - `--retrieval-profile` make scaling and routing experiments executable, but their results need - separate evidence before publication. -- **Exact vector scale envelope**: `eval/vector_scale.py` measures the production - `NumpyVectorIndex` directly at requested corpus sizes with deterministic normalized vectors and - queries. It records a corpus fingerprint, result hashes, environment, and observed - p50/p95/p99 search envelopes. It intentionally has no pass/fail latency threshold: the output - describes the measured machine and workload, not a universal capacity cutoff. Pair it with - `eval/performance.py` before making a deployment decision because direct vector search excludes - the rest of the recall pipeline. Its `engraphis-vector-scale/v1` JSON is a local diagnostic, not - an `engraphis-benchmark/v2` public evidence artifact. -- **Proactive ranking calibration**: `eval/proactive_ranking.py` compares the previous and current - importance-retention floors on a small deterministic queryless-ranking fixture. It reports - top-1 accuracy and minimum expected margins for that fixture only. It is a scoring regression, - not evidence of general recall quality or user-task performance. -- **Workload context economy**: `eval/context_economy.py` compares three executable strategies - across every question in a workload: uncapped full-history replay, a contiguous recency window - at the same hard budget, and shipped Engraphis hybrid recall + packing. It reports evidence and - answer-token quality, cumulative reader-context tokens, a conservative total that charges one - complete source-token pass to indexing, and the query-count break-even point. The default is - deterministic/offline; `--embed-model` enables a real retrieval model, while - `--format locomo|longmemeval` reuses the established external loaders. -- **Agent productivity**: `eval/productivity.py` compares a capped full-history baseline, - always-on retrieval, and - adaptive context through a complete answer-and-correction loop. It reports completed tasks, - first-attempt errors, abstentions, corrections, agent turns, memory calls, wall-clock latency, - and all question/context/output tokens. The bundled agent is deterministic, receives no gold - answer, and is identified in every report; inject a real agent callable for model-specific - results. Optional provider telemetry is reported separately from the deterministic token - counter and is not a provider billing estimate. -- **Dashboard graph layout settle**: `eval/graph_every_bench.py` drives the Every-node - dashboard engine's real worker (`engraphis-graph-every-worker.js`) through a - `prepare → settled` round-trip over synthetic node/link loads and reports wall-clock settle - time plus the scaling ratio across sizes. It measures initial layout cost only: camera pans - and zooms never touch the worker (they are GPU-uniform updates), so no per-frame number can - come out of this harness and none should be quoted. Results are host- and Node-version - dependent local diagnostics, not registered public evidence; run the harness on the target - class of machine before quoting a figure. - -The context-economy and productivity tools intentionally report when a small workload does not -benefit from memory, and the external loaders expose retrieval-quality tradeoffs rather than -hiding them. Their prior local results are not retained as public numbers because no matching -redacted immutable artifact is checked in. Run the registered protocol and publish the resulting -artifact before making a quantitative claim. - -### Reproduce - -```bash -# Correctness gate (deterministic, no download) -python -m pytest tests/ -q -python -m eval.harness --dataset eval/datasets/sample.jsonl --k 5 -python -m eval.harness --dataset eval/datasets/codemem.jsonl --k 5 -python -m eval.harness --dataset eval/datasets/graph_multihop.jsonl --k 5 -python -m eval.ablation -python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 -python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 \ - --candidate-k 25 --candidate-depth adaptive --retrieval-profile auto --iterations 10 -python -m eval.context_economy --dataset eval/datasets/codemem.jsonl \ - --token-budget 512 --k 5 -python -m eval.productivity --dataset eval/datasets/codemem.jsonl \ - --max-context-tokens 512 --retrieval-token-budget 256 -python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 \ - --iterations 5 --filler-memories 1000 -# Direct NumPy search envelope at representative corpus sizes; timings are machine-specific. -python -m eval.vector_scale --sizes 1000,10000,100000 --queries 20 --iterations 3 --json -# Deterministic queryless-ranking calibration fixture. -python -m eval.proactive_ranking -# Canonical latency/resource protocol: requires >=1,000 queries and five processes. -python -m eval.performance --dataset fixed-1000-plus.jsonl --acceptance-matrix --processes 5 - -# External retrieval diagnostics (downloads all-MiniLM-L6-v2; not QA/leaderboard results) -python -m eval.external --dataset longmemeval_s.json --format longmemeval --k 10 -python -m eval.external --dataset locomo10.json --format locomo --k 10 -# Complete external-dataset coverage with an immutable embedding revision. These runs are -# retrieval-only diagnostics, not official benchmark-harness or leaderboard results. Completed -# public-safe artifacts and measured results are listed in the benchmark expansion report. -python -m eval.external --dataset longmemeval_s.json --format longmemeval --canonical \ - --embed-revision <40-character-model-commit> --json external-longmemeval.json -python -m eval.external --dataset locomo10.json --format locomo --canonical --no-resolve \ - --embed-revision <40-character-model-commit> \ - --locomo-repair-manifest eval/datasets/locomo10_repair_manifest.json \ - --json external-locomo.json -python -m eval.context_economy --dataset locomo10.json --format locomo \ - --embed-model sentence-transformers/all-MiniLM-L6-v2 --token-budget 512 --k 10 --no-resolve -``` - -Canonical external mode requires an exact lowercase 40-character embedding commit and a semantic -embedder; dependency or model-load failure is fatal instead of silently falling back to hashing. -Every report records `embedding`, `dataset_sha256`, `source_cases`, `normalized_cases`, and -`configuration` provenance so a result can be attributed to the actual data and retrieval setup. - -The official ten-conversation LoCoMo JSON contains delimiter-packed IDs, two mechanical ID -typos, and three references that cannot be normalized syntactically. The adapter normalizes only -the unambiguous forms. The checked-in repair manifest is bound to the official source SHA-256, -names every remaining replacement/removal, must be fully consumed, and is recorded in the JSON -report with its own hash. Any source update, unused repair, or unresolved ID fails the run. This -repairs retrieval references only; it does not claim to correct LoCoMo's semantic answer labels. - -The earlier private pinned retrieval diagnostic was the pre-publication state. Current complete -LoCoMo and LongMemEval retrieval diagnostics are retained as separate public-safe artifacts in the -[benchmark expansion report](docs/BENCHMARK_EXPANSION_RESULTS.md), with source, model, configuration -and checksum boundaries. Those values remain evidence-retrieval metrics, not end-to-end QA accuracy -or an official LoCoMo leaderboard score. - -## What we do NOT yet claim - -- **No official end-to-end LLM QA accuracy.** The deterministic productivity agent measures the - complete local control loop, not a frontier answering model. Official LoCoMo / LongMemEval QA - still requires a pinned answering model and evaluator. -- **No hosted-service latency comparison.** The in-repo p50/p95/p99 benchmark covers the local - reference pipeline and records its environment; unlike environments are not compared. -- **No neutral third-party ranking.** We have not run an external eval platform. -- **No provider bill estimate.** Context-economy counts reader evidence under its named counter. - It excludes system/tool prompts, questions, completions, prompt caching, provider pricing, - compute, and storage. Its indexing-inclusive total is a conservative text-volume proxy. - -Every publishable run should emit the `engraphis-benchmark/v2` envelope: dataset/config hashes, -per-question records, explicit exclusions, fixed-budget context curves, and deterministic -stratified or paired bootstrap confidence intervals. Every run names its token counter. -Noncanonical offline fixtures may identify a deterministic estimate; canonical public evidence -requires the exact pinned reader tokenizer and immutable model revision. The lightweight CI + samples averages **85.38** tokens and reaches **108** under a 1,500-token cap; Recall@5, + hit@5, and answer-token recall remain 1.000 for the legacy candidate-page view. The registered + v9 artifact predates `packed_quality`, so no packed-quality aggregate is published from it. + These aggregates are evidence ID + `offline-performance` in the registry above. `--filler-memories`, `--candidate-k`, and + `--retrieval-profile` make scaling and routing experiments executable, but their results need + separate evidence before publication. +- **Exact vector scale envelope**: `eval/vector_scale.py` measures the production + `NumpyVectorIndex` directly at requested corpus sizes with deterministic normalized vectors and + queries. It records a corpus fingerprint, result hashes, environment, and observed + p50/p95/p99 search envelopes. It intentionally has no pass/fail latency threshold: the output + describes the measured machine and workload, not a universal capacity cutoff. Pair it with + `eval/performance.py` before making a deployment decision because direct vector search excludes + the rest of the recall pipeline. Its `engraphis-vector-scale/v1` JSON is a local diagnostic, not + an `engraphis-benchmark/v2` public evidence artifact. +- **Proactive ranking calibration**: `eval/proactive_ranking.py` compares the previous and current + importance-retention floors on a small deterministic queryless-ranking fixture. It reports + top-1 accuracy and minimum expected margins for that fixture only. It is a scoring regression, + not evidence of general recall quality or user-task performance. +- **Workload context economy**: `eval/context_economy.py` compares three executable strategies + across every question in a workload: uncapped full-history replay, a contiguous recency window + at the same hard budget, and shipped Engraphis hybrid recall + packing. It reports evidence and + answer-token quality, cumulative reader-context tokens, a conservative total that charges one + complete source-token pass to indexing, and the query-count break-even point. The default is + deterministic/offline; `--embed-model` enables a real retrieval model, while + `--format locomo|longmemeval` reuses the established external loaders. +- **Agent productivity**: `eval/productivity.py` compares a capped full-history baseline, + always-on retrieval, and + adaptive context through a complete answer-and-correction loop. It reports completed tasks, + first-attempt errors, abstentions, corrections, agent turns, memory calls, wall-clock latency, + and all question/context/output tokens. The bundled agent is deterministic, receives no gold + answer, and is identified in every report; inject a real agent callable for model-specific + results. Optional provider telemetry is reported separately from the deterministic token + counter and is not a provider billing estimate. +- **Dashboard graph layout settle**: `eval/graph_every_bench.py` drives the Every-node + dashboard engine's real worker (`engraphis-graph-every-worker.js`) through a + `prepare → settled` round-trip over synthetic node/link loads and reports wall-clock settle + time plus the scaling ratio across sizes. It measures initial layout cost only: camera pans + and zooms never touch the worker (they are GPU-uniform updates), so no per-frame number can + come out of this harness and none should be quoted. Results are host- and Node-version + dependent local diagnostics, not registered public evidence; run the harness on the target + class of machine before quoting a figure. + +The context-economy and productivity tools intentionally report when a small workload does not +benefit from memory, and the external loaders expose retrieval-quality tradeoffs rather than +hiding them. Their prior local results are not retained as public numbers because no matching +redacted immutable artifact is checked in. Run the registered protocol and publish the resulting +artifact before making a quantitative claim. + +### Reproduce + +```bash +# Correctness gate (deterministic, no download) +python -m pytest tests/ -q +python -m eval.harness --dataset eval/datasets/sample.jsonl --k 5 +python -m eval.harness --dataset eval/datasets/codemem.jsonl --k 5 +python -m eval.harness --dataset eval/datasets/graph_multihop.jsonl --k 5 +python -m eval.ablation +python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 +python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 \ + --candidate-k 25 --candidate-depth adaptive --retrieval-profile auto --iterations 10 +python -m eval.context_economy --dataset eval/datasets/codemem.jsonl \ + --token-budget 512 --k 5 +python -m eval.productivity --dataset eval/datasets/codemem.jsonl \ + --max-context-tokens 512 --retrieval-token-budget 256 +python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 \ + --iterations 5 --filler-memories 1000 +# Direct NumPy search envelope at representative corpus sizes; timings are machine-specific. +python -m eval.vector_scale --sizes 1000,10000,100000 --queries 20 --iterations 3 --json +# Deterministic queryless-ranking calibration fixture. +python -m eval.proactive_ranking +# Canonical latency/resource protocol: requires >=1,000 queries and five processes. +python -m eval.performance --dataset fixed-1000-plus.jsonl --acceptance-matrix --processes 5 + +# External retrieval diagnostics (downloads all-MiniLM-L6-v2; not QA/leaderboard results) +python -m eval.external --dataset longmemeval_s.json --format longmemeval --k 10 +python -m eval.external --dataset locomo10.json --format locomo --k 10 +# Complete external-dataset coverage with an immutable embedding revision. These runs are +# retrieval-only diagnostics, not official benchmark-harness or leaderboard results. Completed +# public-safe artifacts and measured results are listed in the benchmark expansion report. +python -m eval.external --dataset longmemeval_s.json --format longmemeval --canonical \ + --embed-revision <40-character-model-commit> --json external-longmemeval.json +python -m eval.external --dataset locomo10.json --format locomo --canonical --no-resolve \ + --embed-revision <40-character-model-commit> \ + --locomo-repair-manifest eval/datasets/locomo10_repair_manifest.json \ + --json external-locomo.json +python -m eval.context_economy --dataset locomo10.json --format locomo \ + --embed-model sentence-transformers/all-MiniLM-L6-v2 --token-budget 512 --k 10 --no-resolve +``` + +Canonical external mode requires an exact lowercase 40-character embedding commit and a semantic +embedder; dependency or model-load failure is fatal instead of silently falling back to hashing. +Every report records `embedding`, `dataset_sha256`, `source_cases`, `normalized_cases`, and +`configuration` provenance so a result can be attributed to the actual data and retrieval setup. + +The official ten-conversation LoCoMo JSON contains delimiter-packed IDs, two mechanical ID +typos, and three references that cannot be normalized syntactically. The adapter normalizes only +the unambiguous forms. The checked-in repair manifest is bound to the official source SHA-256, +names every remaining replacement/removal, must be fully consumed, and is recorded in the JSON +report with its own hash. Any source update, unused repair, or unresolved ID fails the run. This +repairs retrieval references only; it does not claim to correct LoCoMo's semantic answer labels. + +The earlier private pinned retrieval diagnostic was the pre-publication state. Current complete +LoCoMo and LongMemEval retrieval diagnostics are retained as separate public-safe artifacts in the +[benchmark expansion report](docs/BENCHMARK_EXPANSION_RESULTS.md), with source, model, configuration +and checksum boundaries. Those values remain evidence-retrieval metrics, not end-to-end QA accuracy +or an official LoCoMo leaderboard score. + +## What we do NOT yet claim + +- **No official end-to-end LLM QA accuracy.** The deterministic productivity agent measures the + complete local control loop, not a frontier answering model. Official LoCoMo / LongMemEval QA + still requires a pinned answering model and evaluator. +- **No hosted-service latency comparison.** The in-repo p50/p95/p99 benchmark covers the local + reference pipeline and records its environment; unlike environments are not compared. +- **No neutral third-party ranking.** We have not run an external eval platform. +- **No provider bill estimate.** Context-economy counts reader evidence under its named counter. + It excludes system/tool prompts, questions, completions, prompt caching, provider pricing, + compute, and storage. Its indexing-inclusive total is a conservative text-volume proxy. + +Every publishable run should emit the `engraphis-benchmark/v2` envelope: dataset/config hashes, +per-question records, explicit exclusions, fixed-budget context curves, and deterministic +stratified or paired bootstrap confidence intervals. Every run names its token counter. +Noncanonical offline fixtures may identify a deterministic estimate; canonical public evidence +requires the exact pinned reader tokenizer and immutable model revision. The lightweight CI fixtures validate that machinery; they are not a claim about external benchmark performance. Public journey and external retrieval exports identify producer code by unique @@ -316,183 +316,183 @@ LongMemEval-V2 inputs use stable role names such as `inputs/dataset` or Each name remains bound to its SHA-256 and byte count. The shared envelope keeps basename-only defaults for other callers and historical artifacts; exporters opt in with explicit `source_names` and verify the completed envelope against evaluated bytes. - -The benchmark context metric reads strict recall usage fields rather than inferring prompt size: -`budget_tokens`, `context_tokens`, `source_tokens`, `saved_tokens`, `savings_ratio`, -`packed_count`, `omitted_count`, and `token_counter`. Use `engraphis_recall_context` for a -hard-budget prompt packet; legacy `engraphis_recall` remains available in full or compact response -mode for compatibility. - -### Canonical public artifacts - -Use `python -m eval.benchmark --input report.json --output artifacts/run.json` to validate a -report and write sorted, immutable JSON plus `run.json.sha256`. The command permits an identical -retry but refuses to replace a different artifact at the same path. For an official -LongMemEval-V2 run, add `--canonical`: this requires a profile with an exact benchmark repository -revision, dataset revision, reader model revision, and embedding model revision. The checked-in -profile pins immutable upstream commits; replacing any revision with a mutable tag fails -validation. Canonical profiles label the baseline (`no_retrieval`, `lexical_only`, `dense_only`, -`dense_lexical_rrf`, `full_hybrid`, `full_history`, `no_graph`, `no_reranker`, -`no_temporal_resolution`, or `whole_document`) and declare the required fixed context-budget -matrix: 256, 512, 1024, 2048, and 4096 tokens. Canonical in-repo reports rerun every question at -all five budgets and validate each aggregate against its per-question evidence. The checked-in -LongMemEval-V2 memory-module configuration sets the official adapter's operating point to 1,024 -tokens; that single official point must not be presented as a five-point curve. - -`eval.external --canonical` refuses `--limit` and rejects a normalized output that omitted source -cases. Retrieval-only abstention/no-evidence records remain visible in the artifact's -`exclusions`; they are not counted as evidence-retrieval scores. - -Official LongMemEval-V2 output can be converted into a public-safe QA artifact with -`python -m eval.longmemeval_v2_evidence`. The exporter requires the completion manifest written by -the pinned runner after a successful, complete official run. It binds the exact per-question -output, questions, haystack, trajectories, memory configuration, matrix manifest, seed, clean -official checkout, and recorded environment. The public artifact keeps the official QA score, -fixed-reader context token count, aggregate source-file digests, repository state, and artifact -checksum. It removes raw questions, answers, prompts, reader output, and retrieved context, and -does not publish per-record content fingerprints. See the -[`public benchmark runbook`](docs/PUBLIC_BENCHMARK_RUNBOOK.md) for the end-to-end operator sequence. - -### LongMemEval-V2 memory-module adapter - -`eval.longmemeval_v2.EngraphisLongMemEvalV2Memory` follows the official -`memory_modules.memory.Memory` interface at LongMemEval-V2 commit -`6f020ac2fc3275e46c706d3406e02c3ed79b7be2`. When imported in that environment, its -`@register_memory` decorator registers `memory_type="engraphis"`; use the checked-in -[`eval/configs/longmemeval_v2_engraphis.json`](eval/configs/longmemeval_v2_engraphis.json) -with the official harness. The config pins `Qwen/Qwen3-Embedding-8B` to revision -`1d8ad4ca9b3dd8059ad90a75d4983776a23d44af`; mutable embedding revisions are rejected, and a -canonical adapter run fails instead of relabeling the deterministic offline fallback as Qwen. -First materialize the six declared variants at all five token budgets: - -```bash -python -m eval.longmemeval_v2_matrix \ - --output "$ENGRAPHIS_EVIDENCE_RUN_DIR/configs" -``` - -This writes a 30-run manifest: balanced, planner, episodic-cap, planner-plus-episodic-cap, and -matched `context_k=2` comparators for both capped variants, each at 256, 512, 1,024, 2,048, and -4,096 evidence tokens. Run each manifest cell through `python -m eval.run_longmemeval_v2` with all -eight `--engraphis-*` completion-receipt arguments. The wrapper imports the adapter before the -official registry builds the memory module, forces the pinned reader processor revision, and -delegates the remaining official harness arguments unchanged. Only after a successful return does -it verify that the output question IDs exactly cover the source question IDs and write the -immutable execution manifest. - -The checked-in configuration is canonical only when the adapter resolves the pinned Qwen reader -processor at `c202236235762e1c871ad0ccb60c8ee5ba337b9a`. The wrapper refuses a dirty or non-pinned -official checkout and refuses to start if the optional processor dependency or immutable revision -is unavailable; the local regex counter is never silently relabeled as a reader budget. The -recorded budget counts each returned context item's content with that reader tokenizer (without -prompt framing or inter-item separators), so it is a hard **evidence-item content** budget, not a -claim about total chat-prompt tokens. Packed sources are returned as separate context items, -preserving the largest fitting evidence prefix instead of dropping one oversized monolithic item. -Every official per-question row reports inserted and retrieved counts by memory type. A -memory-type-cap claim additionally requires at least two populated inserted types, so a nominal cap -over a single-type workload cannot qualify as evidence. The adapter does not download benchmark -data or call the reader/evaluator; the official harness owns those steps. - -## External evidence status and remaining executions - -1. **Run the official LongMemEval-V2 reader and evaluator.** The adapter, pinned runner, and - redacted evidence exporter are implemented. The exact upstream commit boots in an isolated - Python 3.11 environment and the wrapper reaches the official harness CLI. The dataset, pinned - Qwen reader, and embedding assets require substantial storage and compute; no canonical QA - score is claimed until that run completes. -2. **Publish production-backend latency.** Run `eval/performance.py` with the real embedder and - sqlite-vec/backend configuration on a fixed machine class and corpus scale. -3. **Run the fixed-budget curve on the complete official datasets.** The v2 harness now measures - every question at 256, 512, 1,024, 2,048, and 4,096 evidence tokens and validates the - per-question records, aggregates, and pinned reader-tokenizer identity. Publish the curve only - after complete official runs produce immutable artifacts for every point. -4. **Run an external evaluation platform** once (1)–(3) exist. - -Do not make all evidence lanes variants of explicit factual recall. Executable offline adapters -now cover: - -- [MemoryAgentBench](https://github.com/HUST-AI-HYZ/MemoryAgentBench): incremental multi-turn - learning, long-range understanding, and conflict/consolidation inputs. -- [LoCoMo-Plus](https://github.com/xjtuleeyf/Locomo-Plus): an old implicit constraint must affect - a later response even when the later cue does not restate the remembered fact. -- [Mem2ActBench](https://github.com/Cantaloupe-M/Mem2ActBench): memory must select a tool and - ground its arguments, not merely return a passage. The current adapter measures retrieval and - expected tool-argument context coverage, not generated tool-call success. - -```bash -python -m eval.agent_benchmarks --dataset memoryagentbench.json \ - --format memoryagentbench -python -m eval.agent_benchmarks --dataset locomo_plus.json \ - --format locomo_plus -python -m eval.agent_benchmarks --dataset qa_dataset.jsonl \ - --conversations toolmem_conversation.jsonl --format mem2actbench \ - --artifact artifacts/mem2actbench.json -``` - -Use `--artifact` on any of these commands to write a redacted, immutable evidence envelope plus -an adjacent SHA256 file. The ordinary console/`--json` report is private run material and may -contain source questions for debugging. - -### Upstream-data diagnostics and publication scope - -The LoCoMo-Plus and MemoryAgentBench adapters have been exercised against upstream data and remain -queued for their own public-safe retrieval artifacts. Rerun each pending adapter with `--artifact`, -publish the redacted immutable envelope and checksum, and add its suite/config binding before -quoting a number. Mem2ActBench's declared small retrieval diagnostic is complete and has a checked-in -artifact; its exclusion and memory-cardinality figures are source-preparation metadata in the -public source lock, not product or action-success metrics. None of these lanes is an official -leaderboard, answer-quality, or marketing result. - -The MemoryAgentBench loader accepts both its aligned public JSON export and the Hugging Face -dataset-server `rows[].row` envelope. Rows without gold evidence remain useful for answer-token -coverage, but are excluded from retrieval aggregates and counted separately as -`retrieval_scored_questions`. - -For paired code-agent runs, execute the same tasks with the same model, tools, machine, and -deterministic success oracle under `full_history` and `engraphis`. Then analyze the content-free -run records with: - -```bash -python -m eval.code_agent_ab --full-history full-history.jsonl \ - --engraphis engraphis.jsonl --output paired-report.json -``` - -The analyzer rejects unmatched task IDs and different success oracles, then reports paired -bootstrap intervals for task success, input/output/tool tokens, retries, latency, and optional -cost. Its aggregate output does not echo task IDs or oracle commands. It does not launch an agent -or invent a task-success oracle. - -## Optimization experiments to run before changing defaults - -1. **Budget-aware packing**: compare full source, safe summary, sentence-aligned safe summary - excerpt, and raw-source excerpt at fixed budgets. Gate on support/answer retention and - qualifier preservation, not token count alone. -2. **Adaptive retrieval work**: `--candidate-depth adaptive` is an opt-in performance experiment. - It keeps wider graph/code pools and reduces routine lexical/balanced pools while reporting the - requested and actual depth. A local experiment motivated this option, but no public number is - retained because its machine-specific artifact is not in the evidence registry. Keep the - default fixed until complete external categories meet predeclared quality margins. -3. **Packing-pressure consolidation**: prioritize memory families that are frequently recalled, - repeatedly omitted, or costly per useful token. Count write/index/storage cost as well as later - reader-context savings. -4. **Tokenizer-aware ingestion**: implemented behind the chunk extractor. The dependency-free - default remains `engraphis.chars4.v1`; an explicitly configured Hugging Face reader tokenizer - enforces prose chunk and overlap budgets and records its identity in chunk metadata. Continue - measuring tokens-to-evidence, recall, and storage/index growth together before recommending a - model-specific default. -5. **Bulk ingestion**: add batch embedding plus a transaction-aware vector upsert path, then rerun - the complete MemoryAgentBench Test-Time Learning input. Gate this on identical stored-memory, - provenance, graph-link, and temporal-resolution outcomes, not throughput alone. -6. **Scoped caches**: benchmark query embeddings and repeat-recall results keyed by workspace, - repo, time anchors, profile, and corpus version. Test invalidation correctness before claiming - latency gains. -7. **Privacy-safe real usage**: use `engraphis_context_savings` to let each workspace inspect - aggregate source/context/saved tokens already present in content-free receipts. Keep unlike - token counters separate and require a valid receipt chain before treating totals as auditable. - -## Evaluation question - -The predeclared question is whether the full vector + lexical/BM25 + sparse PPR graph + calibrated -rerank pipeline, bi-temporal resolution, and grounded abstention produce higher evidence recall -per injected token than the registered baselines. The answer must come from a complete, -machine-readable artifact with paired confidence intervals; otherwise the release reports -“no demonstrated improvement.” + +The benchmark context metric reads strict recall usage fields rather than inferring prompt size: +`budget_tokens`, `context_tokens`, `source_tokens`, `saved_tokens`, `savings_ratio`, +`packed_count`, `omitted_count`, and `token_counter`. Use `engraphis_recall_context` for a +hard-budget prompt packet; legacy `engraphis_recall` remains available in full or compact response +mode for compatibility. + +### Canonical public artifacts + +Use `python -m eval.benchmark --input report.json --output artifacts/run.json` to validate a +report and write sorted, immutable JSON plus `run.json.sha256`. The command permits an identical +retry but refuses to replace a different artifact at the same path. For an official +LongMemEval-V2 run, add `--canonical`: this requires a profile with an exact benchmark repository +revision, dataset revision, reader model revision, and embedding model revision. The checked-in +profile pins immutable upstream commits; replacing any revision with a mutable tag fails +validation. Canonical profiles label the baseline (`no_retrieval`, `lexical_only`, `dense_only`, +`dense_lexical_rrf`, `full_hybrid`, `full_history`, `no_graph`, `no_reranker`, +`no_temporal_resolution`, or `whole_document`) and declare the required fixed context-budget +matrix: 256, 512, 1024, 2048, and 4096 tokens. Canonical in-repo reports rerun every question at +all five budgets and validate each aggregate against its per-question evidence. The checked-in +LongMemEval-V2 memory-module configuration sets the official adapter's operating point to 1,024 +tokens; that single official point must not be presented as a five-point curve. + +`eval.external --canonical` refuses `--limit` and rejects a normalized output that omitted source +cases. Retrieval-only abstention/no-evidence records remain visible in the artifact's +`exclusions`; they are not counted as evidence-retrieval scores. + +Official LongMemEval-V2 output can be converted into a public-safe QA artifact with +`python -m eval.longmemeval_v2_evidence`. The exporter requires the completion manifest written by +the pinned runner after a successful, complete official run. It binds the exact per-question +output, questions, haystack, trajectories, memory configuration, matrix manifest, seed, clean +official checkout, and recorded environment. The public artifact keeps the official QA score, +fixed-reader context token count, aggregate source-file digests, repository state, and artifact +checksum. It removes raw questions, answers, prompts, reader output, and retrieved context, and +does not publish per-record content fingerprints. See the +[`public benchmark runbook`](docs/PUBLIC_BENCHMARK_RUNBOOK.md) for the end-to-end operator sequence. + +### LongMemEval-V2 memory-module adapter + +`eval.longmemeval_v2.EngraphisLongMemEvalV2Memory` follows the official +`memory_modules.memory.Memory` interface at LongMemEval-V2 commit +`6f020ac2fc3275e46c706d3406e02c3ed79b7be2`. When imported in that environment, its +`@register_memory` decorator registers `memory_type="engraphis"`; use the checked-in +[`eval/configs/longmemeval_v2_engraphis.json`](eval/configs/longmemeval_v2_engraphis.json) +with the official harness. The config pins `Qwen/Qwen3-Embedding-8B` to revision +`1d8ad4ca9b3dd8059ad90a75d4983776a23d44af`; mutable embedding revisions are rejected, and a +canonical adapter run fails instead of relabeling the deterministic offline fallback as Qwen. +First materialize the six declared variants at all five token budgets: + +```bash +python -m eval.longmemeval_v2_matrix \ + --output "$ENGRAPHIS_EVIDENCE_RUN_DIR/configs" +``` + +This writes a 30-run manifest: balanced, planner, episodic-cap, planner-plus-episodic-cap, and +matched `context_k=2` comparators for both capped variants, each at 256, 512, 1,024, 2,048, and +4,096 evidence tokens. Run each manifest cell through `python -m eval.run_longmemeval_v2` with all +eight `--engraphis-*` completion-receipt arguments. The wrapper imports the adapter before the +official registry builds the memory module, forces the pinned reader processor revision, and +delegates the remaining official harness arguments unchanged. Only after a successful return does +it verify that the output question IDs exactly cover the source question IDs and write the +immutable execution manifest. + +The checked-in configuration is canonical only when the adapter resolves the pinned Qwen reader +processor at `c202236235762e1c871ad0ccb60c8ee5ba337b9a`. The wrapper refuses a dirty or non-pinned +official checkout and refuses to start if the optional processor dependency or immutable revision +is unavailable; the local regex counter is never silently relabeled as a reader budget. The +recorded budget counts each returned context item's content with that reader tokenizer (without +prompt framing or inter-item separators), so it is a hard **evidence-item content** budget, not a +claim about total chat-prompt tokens. Packed sources are returned as separate context items, +preserving the largest fitting evidence prefix instead of dropping one oversized monolithic item. +Every official per-question row reports inserted and retrieved counts by memory type. A +memory-type-cap claim additionally requires at least two populated inserted types, so a nominal cap +over a single-type workload cannot qualify as evidence. The adapter does not download benchmark +data or call the reader/evaluator; the official harness owns those steps. + +## External evidence status and remaining executions + +1. **Run the official LongMemEval-V2 reader and evaluator.** The adapter, pinned runner, and + redacted evidence exporter are implemented. The exact upstream commit boots in an isolated + Python 3.11 environment and the wrapper reaches the official harness CLI. The dataset, pinned + Qwen reader, and embedding assets require substantial storage and compute; no canonical QA + score is claimed until that run completes. +2. **Publish production-backend latency.** Run `eval/performance.py` with the real embedder and + sqlite-vec/backend configuration on a fixed machine class and corpus scale. +3. **Run the fixed-budget curve on the complete official datasets.** The v2 harness now measures + every question at 256, 512, 1,024, 2,048, and 4,096 evidence tokens and validates the + per-question records, aggregates, and pinned reader-tokenizer identity. Publish the curve only + after complete official runs produce immutable artifacts for every point. +4. **Run an external evaluation platform** once (1)–(3) exist. + +Do not make all evidence lanes variants of explicit factual recall. Executable offline adapters +now cover: + +- [MemoryAgentBench](https://github.com/HUST-AI-HYZ/MemoryAgentBench): incremental multi-turn + learning, long-range understanding, and conflict/consolidation inputs. +- [LoCoMo-Plus](https://github.com/xjtuleeyf/Locomo-Plus): an old implicit constraint must affect + a later response even when the later cue does not restate the remembered fact. +- [Mem2ActBench](https://github.com/Cantaloupe-M/Mem2ActBench): memory must select a tool and + ground its arguments, not merely return a passage. The current adapter measures retrieval and + expected tool-argument context coverage, not generated tool-call success. + +```bash +python -m eval.agent_benchmarks --dataset memoryagentbench.json \ + --format memoryagentbench +python -m eval.agent_benchmarks --dataset locomo_plus.json \ + --format locomo_plus +python -m eval.agent_benchmarks --dataset qa_dataset.jsonl \ + --conversations toolmem_conversation.jsonl --format mem2actbench \ + --artifact artifacts/mem2actbench.json +``` + +Use `--artifact` on any of these commands to write a redacted, immutable evidence envelope plus +an adjacent SHA256 file. The ordinary console/`--json` report is private run material and may +contain source questions for debugging. + +### Upstream-data diagnostics and publication scope + +The LoCoMo-Plus and MemoryAgentBench adapters have been exercised against upstream data and remain +queued for their own public-safe retrieval artifacts. Rerun each pending adapter with `--artifact`, +publish the redacted immutable envelope and checksum, and add its suite/config binding before +quoting a number. Mem2ActBench's declared small retrieval diagnostic is complete and has a checked-in +artifact; its exclusion and memory-cardinality figures are source-preparation metadata in the +public source lock, not product or action-success metrics. None of these lanes is an official +leaderboard, answer-quality, or marketing result. + +The MemoryAgentBench loader accepts both its aligned public JSON export and the Hugging Face +dataset-server `rows[].row` envelope. Rows without gold evidence remain useful for answer-token +coverage, but are excluded from retrieval aggregates and counted separately as +`retrieval_scored_questions`. + +For paired code-agent runs, execute the same tasks with the same model, tools, machine, and +deterministic success oracle under `full_history` and `engraphis`. Then analyze the content-free +run records with: + +```bash +python -m eval.code_agent_ab --full-history full-history.jsonl \ + --engraphis engraphis.jsonl --output paired-report.json +``` + +The analyzer rejects unmatched task IDs and different success oracles, then reports paired +bootstrap intervals for task success, input/output/tool tokens, retries, latency, and optional +cost. Its aggregate output does not echo task IDs or oracle commands. It does not launch an agent +or invent a task-success oracle. + +## Optimization experiments to run before changing defaults + +1. **Budget-aware packing**: compare full source, safe summary, sentence-aligned safe summary + excerpt, and raw-source excerpt at fixed budgets. Gate on support/answer retention and + qualifier preservation, not token count alone. +2. **Adaptive retrieval work**: `--candidate-depth adaptive` is an opt-in performance experiment. + It keeps wider graph/code pools and reduces routine lexical/balanced pools while reporting the + requested and actual depth. A local experiment motivated this option, but no public number is + retained because its machine-specific artifact is not in the evidence registry. Keep the + default fixed until complete external categories meet predeclared quality margins. +3. **Packing-pressure consolidation**: prioritize memory families that are frequently recalled, + repeatedly omitted, or costly per useful token. Count write/index/storage cost as well as later + reader-context savings. +4. **Tokenizer-aware ingestion**: implemented behind the chunk extractor. The dependency-free + default remains `engraphis.chars4.v1`; an explicitly configured Hugging Face reader tokenizer + enforces prose chunk and overlap budgets and records its identity in chunk metadata. Continue + measuring tokens-to-evidence, recall, and storage/index growth together before recommending a + model-specific default. +5. **Bulk ingestion**: add batch embedding plus a transaction-aware vector upsert path, then rerun + the complete MemoryAgentBench Test-Time Learning input. Gate this on identical stored-memory, + provenance, graph-link, and temporal-resolution outcomes, not throughput alone. +6. **Scoped caches**: benchmark query embeddings and repeat-recall results keyed by workspace, + repo, time anchors, profile, and corpus version. Test invalidation correctness before claiming + latency gains. +7. **Privacy-safe real usage**: use `engraphis_context_savings` to let each workspace inspect + aggregate source/context/saved tokens already present in content-free receipts. Keep unlike + token counters separate and require a valid receipt chain before treating totals as auditable. + +## Evaluation question + +The predeclared question is whether the full vector + lexical/BM25 + sparse PPR graph + calibrated +rerank pipeline, bi-temporal resolution, and grounded abstention produce higher evidence recall +per injected token than the registered baselines. The answer must come from a complete, +machine-readable artifact with paired confidence intervals; otherwise the release reports +“no demonstrated improvement.” diff --git a/README.md b/README.md index f290bf88..db206a18 100644 --- a/README.md +++ b/README.md @@ -1,561 +1,561 @@ -# Engraphis - -[![PyPI version](https://img.shields.io/pypi/v/engraphis.svg)](https://pypi.org/project/engraphis/) -[![License](https://img.shields.io/badge/license-Apache--2.0-green.svg)](https://github.com/Coding-Dev-Tools/engraphis/blob/main/LICENSE) -[![Support](https://img.shields.io/badge/Buy%20Me%20a%20Coffee-support-yellow?logo=buy-me-a-coffee)](https://buymeacoffee.com/Jaixii) - -[https://engraphis.com/](https://engraphis.com/) - -[https://discord.com/invite/Wfr2ejBmY](https://discord.com/invite/Wfr2ejBmY) - -**Give your AI agents a memory. See it, search it, and maintain it, all in a beautiful WebUI on your own machine.** - -

- Engraphis Knowledge Graph tab: force-directed entity-relation network -
- Knowledge Graph · run engraphis-dashboard to see it live -

- -**Grounded, not guessed.** Memory with receipts. Local by default. - ---- - -> **Open-core boundary:** this repository contains the free local engine, dashboard, MCP server, -> and customer-side clients. Hosted sync, analytics, automation, and team services run on the -> official hosted service; their server implementations are not distributed here. - -> **Support continued Engraphis development with Pro.** [Start a 3-day Pro trial](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro&trial=pro#billing) -> or [subscribe to Pro](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro#billing). - ---- - -## Measured token and context savings - -### Runtime estimator - -The dashboard Overview and Audit/Receipts views also show a receipt-backed estimate from -real context deliveries. It compares the host history or retrieved source baseline with the -context Engraphis actually emitted, keeps token counters and release versions separate, and -labels adaptive history reductions separately from packing savings. Receipts without estimator -metadata remain historical/unclassified. This measures estimated prompt-context reduction; it -does not measure provider billing. The `/context-savings` API and -`engraphis_context_savings` MCP tool aggregate the complete history across all visible workspaces -by default, or accept an explicit workspace plus optional `from_ts`, `to_ts`, and -`release_version` filters. - -

+# Engraphis + +[![PyPI version](https://img.shields.io/pypi/v/engraphis.svg)](https://pypi.org/project/engraphis/) +[![License](https://img.shields.io/badge/license-Apache--2.0-green.svg)](https://github.com/Coding-Dev-Tools/engraphis/blob/main/LICENSE) +[![Support](https://img.shields.io/badge/Buy%20Me%20a%20Coffee-support-yellow?logo=buy-me-a-coffee)](https://buymeacoffee.com/Jaixii) + +[https://engraphis.com/](https://engraphis.com/) + +[https://discord.com/invite/Wfr2ejBmY](https://discord.com/invite/Wfr2ejBmY) + +**Give your AI agents a memory. See it, search it, and maintain it, all in a beautiful WebUI on your own machine.** + +

+ Engraphis Knowledge Graph tab: force-directed entity-relation network +
+ Knowledge Graph · run engraphis-dashboard to see it live +

+ +**Grounded, not guessed.** Memory with receipts. Local by default. + +--- + +> **Open-core boundary:** this repository contains the free local engine, dashboard, MCP server, +> and customer-side clients. Hosted sync, analytics, automation, and team services run on the +> official hosted service; their server implementations are not distributed here. + +> **Support continued Engraphis development with Pro.** [Start a 3-day Pro trial](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro&trial=pro#billing) +> or [subscribe to Pro](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro#billing). + +--- + +## Measured token and context savings + +### Runtime estimator + +The dashboard Overview and Audit/Receipts views also show a receipt-backed estimate from +real context deliveries. It compares the host history or retrieved source baseline with the +context Engraphis actually emitted, keeps token counters and release versions separate, and +labels adaptive history reductions separately from packing savings. Receipts without estimator +metadata remain historical/unclassified. This measures estimated prompt-context reduction; it +does not measure provider billing. The `/context-savings` API and +`engraphis_context_savings` MCP tool aggregate the complete history across all visible workspaces +by default, or accept an explicit workspace plus optional `from_ts`, `to_ts`, and +`release_version` filters. + +

Dark chart of registered deterministic fixtures. Structure-aware chunks reduce retrieved context from 740.3 to 214.3 tokens and the smallest evidence-holding memory from 162.2 to 42.4 tokens. A compact JSON-shape proxy uses 11,138 rather than 24,590 tokens. Retrieved-candidate quality is labeled separately from packed-context quality, both measured in the selected report with packed-quality fields. Actual MCP transport and provider billing are not measured. -
- Less repeated history means more room for the task, tools, and useful evidence. -

- -
-See benchmark details and reproduce the results - -### Controlled before-and-after example - -| Retrieval mode | Mean returned memory content | Recall@5 | -|---|---:|---:| -| Whole documents | 740.3 tokens | 1.000 | -| Engraphis structure-aware chunks | 214.3 tokens | 1.000 | - -The chunked mode returns the relevant passage instead of the whole document: **526.0 fewer tokens -per question**. Under the same model-context budget, that leaves roughly **526 tokens** for task -instructions or other relevant evidence. This is evidence ID `offline-chunking` in the registered -artifact below. - -### Measurement details and reproducibility - -The table below contains every exact token/context aggregate currently published here and keeps -its counting boundary explicit. - -| What is counted | Comparison | Measured reduction | Quality held constant | -|---|---|---|---| -| Retrieved top-5 memory content, averaged per question | Whole documents: **740.3** tokens → structure-aware chunks: **214.3** tokens | **526.0 fewer tokens per question** (**71.1% lower**, about **3.5× smaller**) | Recall@5 **1.000** in both modes across 6 documents and 18 questions | -| Smallest returned memory that contains the reference evidence | Whole documents: **162.2** tokens → chunks: **42.4** tokens | **119.8 fewer tokens to evidence** (**73.9% lower**, about **3.8× smaller**) | The same 18 questions had a returned evidence-holding memory in both modes | +
+ Less repeated history means more room for the task, tools, and useful evidence. +

+ +
+See benchmark details and reproduce the results + +### Controlled before-and-after example + +| Retrieval mode | Mean returned memory content | Recall@5 | +|---|---:|---:| +| Whole documents | 740.3 tokens | 1.000 | +| Engraphis structure-aware chunks | 214.3 tokens | 1.000 | + +The chunked mode returns the relevant passage instead of the whole document: **526.0 fewer tokens +per question**. Under the same model-context budget, that leaves roughly **526 tokens** for task +instructions or other relevant evidence. This is evidence ID `offline-chunking` in the registered +artifact below. + +### Measurement details and reproducibility + +The table below contains every exact token/context aggregate currently published here and keeps +its counting boundary explicit. + +| What is counted | Comparison | Measured reduction | Quality held constant | +|---|---|---|---| +| Retrieved top-5 memory content, averaged per question | Whole documents: **740.3** tokens → structure-aware chunks: **214.3** tokens | **526.0 fewer tokens per question** (**71.1% lower**, about **3.5× smaller**) | Recall@5 **1.000** in both modes across 6 documents and 18 questions | +| Smallest returned memory that contains the reference evidence | Whole documents: **162.2** tokens → chunks: **42.4** tokens | **119.8 fewer tokens to evidence** (**73.9% lower**, about **3.8× smaller**) | The same 18 questions had a returned evidence-holding memory in both modes | | Full versus compact recall payload proxy across one 26-question pass within a 260-timed-recall CodeMem run | Full proxy: **24,590** `engraphis.regex.v1` tokens → compact proxy: **11,138** tokens | **13,452 proxy tokens avoided** (**54.71% lower**) | 26 payload samples; 260 timed recalls; Recall@5, hit@5, and answer-token recall all **1.000** | -| Packed prompt-context usage in the same 26-question CodeMem sample pass | Hard budget: **1,500** tokens; observed mean: **85.38**; observed maximum: **108** | A hard cap prevents a recall from exceeding its configured context budget | This is usage accounting, not a before/after savings comparison | - -The performance report keeps its legacy `quality` fields for all candidate chunks returned before -context packing and adds `packed_quality` for evidence admitted to the reader context. The checked-in +| Packed prompt-context usage in the same 26-question CodeMem sample pass | Hard budget: **1,500** tokens; observed mean: **85.38**; observed maximum: **108** | A hard cap prevents a recall from exceeding its configured context budget | This is usage accounting, not a before/after savings comparison | + +The performance report keeps its legacy `quality` fields for all candidate chunks returned before +context packing and adds `packed_quality` for evidence admitted to the reader context. The checked-in v19 artifact includes both quality views, with Recall@5, hit@5 and answer-token evidence coverage -of 1.000 for the 26-question fixture in each view. Both views measure retrieved evidence; -neither is an end-to-end question-answer score. Coding outcomes, external datasets, and staged -operational capacity remain separate pending evaluation tracks until their artifacts are selected. - +of 1.000 for the 26-question fixture in each view. Both views measure retrieved evidence; +neither is an end-to-end question-answer score. Coding outcomes, external datasets, and staged +operational capacity remain separate pending evaluation tracks until their artifacts are selected. + These values are evidence IDs `offline-chunking` and `offline-performance` in [`offline-fixtures-v76.json`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/benchmark-evidence/offline-fixtures-v76.json), SHA-256 `2fb5ce5b2cbc21541f7ae9cad5d2ef615014c0e86f9881f00621d00986422ba8`. -[`BENCHMARKS.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md#public-numeric-evidence-registry) -records the matching suite digest, exact commands, and per-command config digests. The offline -fixture registry intentionally excludes external, model-dependent, consolidation, productivity, -and latency results. Completed retrieval-only diagnostics are published separately in the -[benchmark expansion results](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/BENCHMARK_EXPANSION_RESULTS.md) with redacted immutable -artifacts; no generated-answer, official leaderboard, hosted-latency, or paid result is claimed -here. - -The compact payload shape avoids duplicating full memory bodies when the packed context and source -list are enough. The evaluator tokenizes JSON-shaped full and compact payload proxies built from -recall results; it does **not** serialize the MCP envelope or measure a transport response. The -fixture therefore does not measure model-provider charges, end-to-end task time, or customer cost -savings. - -The measures are deliberately separate and **must not be added together**: chunking counts the -content of retrieved memory records before `ContextPacker`, whereas compact recall counts a -serialized JSON-shape payload proxy. “Tokens to evidence” is the size of the smallest -retrieved memory record holding the reference evidence; it is not latency or end-to-end answer -accuracy. Chunking creates more focused stored records, so this is a context-efficiency result, -not a storage-reduction claim. - -Reproduce the registered quality and token/context measurements without a network connection or -API key: - -```bash -python -m eval.grounded -python -m eval.chunking_eval --dataset eval/datasets/longdoc.jsonl --k 5 -python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 --json -``` - -These are small deterministic correctness and efficiency fixtures, not official LoCoMo / -LongMemEval QA scores or a third-party leaderboard result. Compact-response counts use the exact -`engraphis.regex.v1` counter; the chunking evaluation uses its documented deterministic -normalized-character estimator. Chunking measures retrieved memory content, while compact recall -measures a serialized JSON-shape payload proxy, not an MCP transport response. See the registered -artifact and [`BENCHMARKS.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md) -for definitions, limitations, and canonical external-evaluation requirements. - -
- ---- - -## Full Engraphis install: pip install "engraphis[all]" - -The complete `engraphis[all]` install is the default way to use Engraphis: it includes the local -dashboard, Smart MCP server, documents, Cloud Sync client, and supported optional integrations. -Python 3.10+ is required. - -```bash -pip install "engraphis[all]" -engraphis-dashboard -``` - -The dashboard opens at [http://127.0.0.1:8700](http://127.0.0.1:8700). Local memory needs no -account or API key. - -### Smaller installation options - -Use a smaller package only when you intentionally need a limited surface. The NumPy-only core -continues to support Python 3.9+. - -| Goal | Install | Start | -|---|---|---| -| Local dashboard and REST API | `pip install "engraphis[server]"` | `engraphis-dashboard` | -| Coding-agent memory over Smart MCP | `pip install "engraphis[mcp]"` | `codex mcp add engraphis -- engraphis-mcp` | -| Native SQLite vector acceleration | `pip install "engraphis[vector]"` | Server entrypoints select it automatically | -| Offline Python library | `pip install engraphis` | `MemoryService.create("engraphis.db")` | - -For MCP clients other than Codex, configure a stdio server whose command is `engraphis-mcp`; see -the [agent connection guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/AGENT_CONNECT.md). - -### Updating - -Use `engraphis-update` to upgrade the installation using its detected install method. Package -metadata does not record which extras were selected, so the updater defaults to the safe -superset `engraphis[all]` rather than silently dropping an optional surface. For a deliberate -selection, set `ENGRAPHIS_UPDATE_EXTRAS` to a comma-separated list (for example -`server,mcp`), or set it to `none` for the base package only. - -> **Upgrading to 1.4:** `engraphis-mcp` now exposes the nine-tool Smart gateway. Integrations that -> require the former 35 direct tool names should run `engraphis-mcp-classic`. The SQLite schema -> in the 1.4.0 release was version 9. Existing v7-to-v8 databases already contain `confidence` -> and `pinned_at`/`unpinned_at`; v9 adds the `memory_tombstones` repository-scope column/table -> and performs a one-time entity-canonicalization repair, then migrates automatically on first -> open. A tombstone with a known `repo_id` is terminal only in that repository; legacy repo-less -> tombstones remain global. See the [1.4.0 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#140---2026-08-02). - -> **Upgrading to 1.5:** schema 10 bounds legacy retention state and schema 11 backfills explicit -> approval only for eligible pre-review local memories. Pending and quarantined evidence remains -> gated. Existing 1.4.x databases migrate automatically when Engraphis 1.5 opens them; see the -> [1.5 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#15---2026-08-04). - -> **Upgrading to 1.6:** existing 1.5 databases migrate automatically through schema 12, which -> classifies content-free erasure markers before sync: existing markers become local-only -> `never_export`, while new secure erasures become `remote_erasure` only for non-secret -> `workspace`/`repo` records already eligible for sharing. Schema 13 adds per-memory hybrid -> logical clocks for deterministic descriptive-state sync and durable, content-free proof that a -> memory crossed a sync boundary. Schema 14 adds the Obsidian collection and import manifests; -> schema 15 generalizes them to source-neutral local documents, preserves temporal source lineage -> across re-imports, binds adapters and target scopes, and retains only bounded, content-free -> per-job format/result metadata. The schema 16 migration persists each import job's optional session target -> and requires source lineage and job-item attachments to remain in that exact session. See the -> [1.6 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#16---2026-08-15). - ---- - -## What Engraphis gives an agent - -An agent should not have to reconstruct a project from scattered chat history on every task. -Engraphis turns local project knowledge into scoped, time-aware memory; retrieves the evidence -that supports the current question; and returns a bounded, attributable context packet. - -The core task is continuity: retrieve the current, supported project decision without dragging the -whole history into the next prompt. See [measured token and context savings](#measured-token-and-context-savings) -for the short version of how much less history an agent has to carry. - -| Agent need | What Engraphis changes | -|---|---| -| Remember a project across sessions | Stores typed memory in a `workspace → repo → session` hierarchy and provides a last-session handoff. | -| Find support for the current task | Fuses vector, lexical, graph, and code-aware retrieval instead of relying on one search signal; `fast` can skip graph traversal for small or latency-sensitive vaults. | -| Know what is true now and what changed | Preserves bi-temporal history and supersession chains instead of silently overwriting a fact. | -| Avoid confident guesses | Returns cited evidence or explicitly abstains when support is too weak. | -| Avoid dragging the whole project into every prompt | Packs context to a configured hard budget and can return a compact MCP response. | -| Keep knowledge in the operator's control | Runs local-first and offline-capable, with scopes, audit records, and optional privacy-safe receipts. | - -## Dashboard and local UI - -The Engraphis dashboard opens `http://127.0.0.1:8700`. Local memory needs no cloud account, -signup, or API key and stays in a SQLite file on your machine. - -**Ledger** is the primary local interface for recall, memories, graph exploration, provenance, -workspaces, and manual consolidation. **Classic** preserves the former full tool suite; both use -the same local data. Switch in **Manage → Settings → Interface** (Ledger) or **Settings → -Appearance & Engine** (Classic). - -### Start it on every platform - -| Platform | How | -|----------|-----| -| **Windows** | Double-click **Engraphis Dashboard** on your Desktop or Start Menu (install: `engraphis-dashboard --install-shortcuts`) | -| **macOS** | Double-click **Engraphis Dashboard.app** on your Desktop (install: same command) | -| **Linux** | Desktop entry in Applications → Development (GNOME/KDE/etc.) | -| **Docker** | `docker compose up`: see `docker-compose.yml` for the one-command deployment | -| **Any** | `engraphis-dashboard` in a terminal | - -In a source checkout, `scripts/launch_dashboard.ps1` is only a Windows convenience wrapper. It -delegates configuration, startup health, browser opening, and process lifecycle to the same -`engraphis-dashboard` entrypoint rather than maintaining a second behavior path. - -### Accessibility-first inspection, built in - -Inspect memories, supersession diffs, recall scores, timelines, links, consolidation, and audit -records in the dashboard. The offline graph renderer is vendored, and the interface is keyboard- -navigable with light and dark themes. Graph exploration offers a focused **High quality** view and -an explicit worker-backed **Every node** view for complete entity projections up to 20,000 -nodes and 200,000 relationships; see the [graph performance profiles](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/GRAPH_PERFORMANCE.md). - ---- - -## How it works - -Engraphis gives agents durable, scoped, *explainable* project knowledge. The local engine combines -Ebbinghaus decay, bi-temporal facts, and hybrid vector/lexical/graph recall; it runs offline with -SQLite, local embeddings, and `numpy` only. - -- **Grounded and governed:** deterministic conflict resolution, cited answers or abstention, - explicit correction/promotion/forgetting, and a complete history. -- **Agent-ready:** MCP tools, hard-budget context packets, handoffs, and code-aware retrieval. -- **Auditable:** content-free receipt chains, provenance, and temporal/entity/code relationships. -- **Practical:** local file and code ingest, optional PDF/OCR/transcription, and SQLCipher at rest. - -### Optional LLM providers - -The memory engine, embeddings, conflict resolution, and recall stay local without an LLM. An -explicitly configured provider adds structured extraction, cited synthesis, consolidation, and -retention supervision. Configure it in **Settings → Connect an LLM**. The activity view records -outcomes, never keys, prompts, or raw provider responses. See the -[LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md) for setup and privacy choices. - -> Privacy boundary: text sent to an explicitly selected provider leaves the local process under -> that provider's terms. Use `ENGRAPHIS_RETENTION_SUPERVISOR=none` (the default) and the offline -> `chunk` extractor when ingestion must remain entirely local. - -Choose and configure an external LLM with the [LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md), -including OpenAI, Anthropic, Google, OpenRouter, Ollama, Cohere Command, Command Code Provider, -and other compatible endpoints. The guide also covers Codex subscription MCP connections. - ---- - -## Install - -```bash -pip install "engraphis[all]" # self-hosted dashboard, MCP, code graph, documents, transcription, PostgreSQL, and Cloud Sync -pip install "engraphis[server]" # dashboard + REST API -pip install "engraphis[mcp]" # MCP server only -pip install "engraphis[documents]" # PDF + image OCR bindings -pip install "engraphis[transcription]" # faster-whisper audio/video -pip install "engraphis[postgres]" # PostgreSQL schema introspection -pip install "engraphis[code]" # tree-sitter code graph indexing -pip install "engraphis[vector]" # native sqlite-vec exact-KNN acceleration -pip install "engraphis[cloud-sync]" # Cloud Sync client crypto/runtime -pip install "engraphis[encryption]" # SQLCipher encryption-at-rest extra -pip install engraphis # core library: numpy only, fully offline -``` - -The official Docker image includes the local Tesseract executable for image OCR. Outside -Docker, the `documents` extra installs its Python bindings; install Tesseract through your -operating system as well if you enable image OCR. - -The NumPy-only core library supports Python 3.9+. Current patched releases of the WebUI -stack, MCP SDK, image parser, and Cloud Sync client require Python 3.10+, so use Python 3.10 -or newer for the `server`, `mcp`, `documents`, `cloud-sync`, or `all` installation paths. - -The default `NumpyVectorIndex` performs an exact full scan. There is no universal memory-count -cutoff because latency depends on vector size, hardware, filters, and the rest of the recall -pipeline. Measure your machine with `python -m eval.vector_scale --backend numpy`, then run -`python -m eval.performance` on a representative corpus. If exact scans miss your latency target, -install `engraphis[vector]`, create the engine with `vector_backend="sqlite-vec"`, and remeasure. -The stable sqlite-vec `vec0` backend executes exact KNN in native code; it is acceleration, not a -claim of sublinear ANN scaling. See [BENCHMARKS.md](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md) for the reproducible commands -and reporting limits. - -Dashboard, REST, and MCP entrypoints default to `ENGRAPHIS_VECTOR_BACKEND=auto`: they use -sqlite-vec when the `vector` extra is installed and compatible, then safely fall back to NumPy. -Programmatic `MemoryEngine.create()` and `MemoryService.create()` retain the deterministic -`numpy` default unless a backend is requested explicitly. -Use `python -m eval.vector_scale --backend sqlite-vec` for an input-identical direct-search -comparison; setup/index-build time is explicitly excluded from the timed search envelope. - -Persistent vectors fail closed unless the embedder can publish a durable, secret-free space -fingerprint. Sentence Transformers use the loaded Hub commit or a manifest of local artifacts; -when a remote model's immutable identity cannot be resolved, persistent vector recall remains -gated instead of mixing spaces. For programmatic OpenAI-compatible embeddings, construct -`ApiEmbedder` with an operator/provider `space_version`; without it the adapter remains usable for -ephemeral embedding only. Its `base_url` may be a provider root or a `/v1` root and is normalized -to exactly one `/v1/embeddings` endpoint. - -`sqlcipher3-binary` publishes CPython manylinux x86-64 wheels. On that target, -`engraphis[encryption]` installs the driver. The cross-platform `all` extra deliberately -omits it so `all` remains resolvable on macOS, Windows, Linux ARM, and musl; on those -targets, provision a compatible SQLCipher driver separately before enabling a database -key. The programmatic core remains plaintext unless a database key is configured. For a -fresh database, `engraphis-init` enables SQLCipher automatically when a compatible driver is -available, creates a private key sidecar, and can be overridden with `--no-encryption`. - -> **Linux / macOS:** if `pip install` fails with `error: externally-managed-environment`, -> your system Python is marked read-only (PEP 668). Install into a virtual environment -> instead. Run `python3 -m venv venv && source venv/bin/activate && pip install "engraphis[server]"` -> Alternatively, use Docker (`docker compose up`). `pipx install "engraphis[server]"` also works. - -> First run downloads `all-MiniLM-L6-v2` (~80 MB). Without it, the engine falls back -> to deterministic feature hashing so it always runs offline. That fallback captures lexical -> overlap, not meaning: recall and grounded MCP responses set `degraded_mode=true` and -> `semantic_support=false`, and disable vector retrieval plus semantic-cosine evidence. Install -> a declared embedding model for semantic retrieval. - -> To require a model that is already local, set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` -> or `local:`. This path never downloads a model. If it is unavailable, Engraphis -> explicitly enters lexical degraded mode instead of presenting hash-vector scores as semantic. - ---- - -## Quickstart: dashboard - -```bash -pip install "engraphis[server]" -engraphis-dashboard # → http://127.0.0.1:8700 -engraphis-dashboard --install-shortcuts # → Desktop + Start Menu icons -``` - -> **Offline first run:** the first launch downloads the `all-MiniLM-L6-v2` embedding model -> (~80 MB), then runs fully offline. To stay offline-only, set -> `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` (never downloads; unknown local models -> enter lexical degraded mode instead of faking semantic scores). Extraction defaults to -> `ENGRAPHIS_EXTRACTOR=none` (verbatim writes), the vector backend defaults to `auto` (native -> acceleration when installed, otherwise NumPy), and recall without a usable semantic space -> reports `degraded_mode=true` with lexical/graph recall. Run `engraphis-init --check` to -> verify the install, extras, and database writability. - -### Docker - -```bash -docker compose up # → http://127.0.0.1:8700 -``` - -For Docker Compose persistence and loopback-port configuration, see the -[Docker deployment guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCKER.md). -`engraphis-server` and `engraphis server` are headless compatibility aliases -for this same v2 service, so every public surface has the same scoped recall and retention model. - -For optional LAN exposure, token configuration, and HTTP MCP setup, see the -[Docker deployment guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCKER.md). - -Set `ENGRAPHIS_API_TOKEN` to require API authentication and `ENGRAPHIS_DB_KEY` to encrypt -the local database at rest. Hosted-plan credentials configure customer clients; they do not -install premium server implementations into this image. See `docker-compose.yml` for options. - ---- - -## Quickstart: MCP server (for coding agents) - -```bash -pip install "engraphis[mcp]" -engraphis-init # writes ~/.engraphis/config.env + prints config snippets -claude mcp add engraphis -- engraphis-mcp -codex mcp add engraphis -- engraphis-mcp # Codex subscription - -``` - -> **Offline first run:** the first tool call lazily loads the `all-MiniLM-L6-v2` embedding -> model (~80 MB, same download as the dashboard), then memory runs fully offline with no API -> key. To stay offline-only, set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` (never -> downloads); extraction defaults to `ENGRAPHIS_EXTRACTOR=none`, the vector backend `auto` -> falls back to NumPy without the `vector` extra, and recall without a usable semantic space -> reports `degraded_mode=true` with lexical/graph recall. Run `engraphis-init --check` to -> verify the install and database path before registering the server. - -For Codex subscription setup and verification, see the [agent connection guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/AGENT_CONNECT.md) -and the [LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md). - -`engraphis-mcp` is zero-configuration Smart MCP: agents begin with nine compact tools for sessions, -prompt-ready recall, durable memory, governed record read/update, conflict review, action discovery, -and safe execution. For code graphs, -governance, audit, or other advanced work, the agent calls `engraphis_discover_actions` and then -the indicated read or action executor; no profile selection is required. The gateway validates -the discovered capability again before it runs it, and clients remain responsible for their -normal destructive-action approval boundary. - -Existing clients that pin the historical 35 named tools can use -`engraphis-mcp-classic` (or `engraphis-mcp-http --classic`). The complete classic inventory, -including `engraphis_check_update`, is in the [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md). - -### Pi extension - -For installation, configuration, lifecycle commands, and the local trust boundary, see the -[Pi extension guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/pi/README.md). - -### Command Code SessionStart hook - -`integrations/commandcode/` ships a SessionStart hook that warms up a new -session with bounded, recalled context from the local Engraphis gateway. Fails -open on timeout and is installed via `python scripts/install_cc_hook.py`. - -### prime-agent fleet - -`integrations/prime_agent/` ships a first-party Python package for -[PrimeIntellect prime-agent](https://github.com/PrimeIntellect-ai/prime-agent) -that exposes the same nine Smart MCP tools, with a `PrimeAgentFleet` of eight -named sub-agents (`researcher`, `planner`, `coder`, `reviewer`, `tester`, -`documenter`, `monitor`, `integrator`) sharing one `engraphis-mcp` stdio -subprocess. Install via `pip install ./integrations/prime_agent` and register -with `python scripts/install_prime_agent.py`. See the -[prime-agent integration guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/prime_agent/README.md). - -**What the integration is.** A `PrimeAgentFleet` is a thin Python layer -around the same `engraphis-mcp` Smart gateway every other host uses. At -runtime the fleet holds one shared `EngraphisMcpClient`, which owns one -`engraphis-mcp` subprocess over JSON-RPC stdio. Each of the eight named -sub-agents gets its own Engraphis session (started lazily on first tool use) -and its own default `repo` scope, so per-role memory is isolated while the -local gateway stays single-process. The eight sub-agent names -(`researcher`, `planner`, `coder`, `reviewer`, `tester`, `documenter`, -`monitor`, `integrator`) are the fixed default; pass `agent_names=[...]` to -`PrimeAgentFleet(...)` for a custom set. Concurrent tool calls serialize at -the JSON-RPC frame layer through an `asyncio.Lock`, so framework-level -parallelism (eight sub-agents reasoning at once) is preserved while the -underlying MCP transport remains one ordered stream. The only integration -surface is `EngraphisPrimeAgent.register()` in -`integrations/prime_agent/src/engraphis_prime_agent/agent.py` -- that is the -single adapter point to override if prime-agent's tool-registration API -differs from the assumed `target.register_tool(name, fn, schema=...)` -contract. - -The design -- eight named sub-agents, one shared stdio subprocess, -per-agent session bootstrap, and `ENGRAPHIS_*`-only environment forwarding -to the gateway -- is recorded in `~/.commandcode/plans/prime-agent-integration.md` -on the host where the integration was developed. When that host plan is not -available (other contributor machines, CI), the same design is summarized in -the PR description that introduced the integration and in the -[prime-agent integration guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/prime_agent/README.md) -("Architecture" and "Concurrency model" sections). - -## Quickstart: repository graph - -```bash -pip install "engraphis[code]" -engraphis-graph index -w acme -r api --root . -engraphis-graph search -w acme -r api "UserService" -# `query`/`explain` blend code search with your stored memories: query matches symbol -# and file NAMES (a full question sentence won't match anything), and explain's answer -# is drawn from memories recorded against the repo; both are empty on a fresh index. -engraphis-graph query -w acme -r api "UserService" -engraphis-graph explain -w acme -r api "why does deploy depend on approval?" -engraphis-graph path -w acme -r api UserService DatabasePool -engraphis-graph impact -w acme -r api --root . --git-range origin/main...HEAD -engraphis-graph prs -w acme -r api --base main --head HEAD -engraphis-graph export -w acme -r api -o engraphis-graph-out -engraphis-graph install-merge-driver --root . -``` - -The export contains `graph.json`, a self-contained `graph.html`, and `GRAPH_REPORT.md`. -Indexing supports Python, JavaScript, TypeScript, Go, Rust, Java, C#, C, C++, SQL, and -Terraform. Tree-sitter is used when available; the dependency-free regex backend remains a -functional fallback. Definitions, methods, calls, imports, ownership, variables, -inheritance/implementation, and docstrings/comments are indexed. Indexing is incremental by -content hash, honors `.engraphisignore`, and does not follow file symlinks outside the repository -root. Call edges are name-based and best-effort rather than type-resolved. The optional Git merge -driver validates bounded graph JSON and deterministically unions nodes and edges instead of -choosing one export side. - -For a read-only recall and graph API that can be shared without exposing write operations: - -```bash -pip install "engraphis[server]" -engraphis-graph-server # API at http://127.0.0.1:8720; schema at /openapi.json -``` - -A non-loopback bind fails closed unless `ENGRAPHIS_GRAPH_TOKEN` (or -`ENGRAPHIS_API_TOKEN`) is set. See [the v3 architecture/design document](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md). - ---- - -## Quickstart: Python library - -```python -from engraphis.service import MemoryService - -mem = MemoryService.create("engraphis.db") -mem.remember("Auth migrated from JWT to PASETO.", workspace="acme", repo="api") -hit = mem.recall("why did we change auth?", workspace="acme", repo="api") -print(hit["context"]) -``` - -The same `MemoryService` backs the dashboard and the MCP server. The package root also -intentionally exposes the low-level engine facade (`MemoryEngine`, `create_memory_engine`) -for advanced composition, while `MemoryService` remains the high-level service API. - -New writes support `session`, `repo`, and `workspace` visibility. `scope="user"` is reserved and -rejected until records carry an immutable owner identity; it must not be treated as private -per-person memory. Historical user-scope rows remain workspace-bound for compatibility. - -After an upgrade, `stats()` reports prompt-eligibility counts and active embedding-space -coverage. Zero-result recall identifies a review-gated scope instead of silently looking empty, -and `engraphis-cli review list|approve` provides a dry-run-first local bulk workflow. Embedding -model changes trigger a guarded rebuild; vector recall stays disabled until every stored vector -matches the new fingerprint. See [recall recovery](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/RECALL_RECOVERY.md). - -Agent hosts can avoid retrieval when their existing history already fits: - -```python -decision = mem.adaptive_context( - "what should the agent do next?", - current_history, - workspace="acme", - repo="api", - max_context_tokens=8_192, - retrieval_token_budget=1_024, -) -prompt_context = decision["context"] -``` - -The decision is `history_bypass` when the history fits, `retrieval` when compact evidence is -strong, and `history_fallback` when weak retrieval should widen back to recent raw history. - -For an agent prompt, prefer `engraphis_recall_context`: it returns one hard-budget packed -`context` plus compact `sources`, deterministic `usage` accounting (`budget_tokens`, `context_tokens`, -`source_tokens`, `saved_tokens`, `savings_ratio`, `packed_count`, `omitted_count`, and -`token_counter`), and optional diagnostics. Accounting is exact for the named counter; inject the -reader's tokenizer when reader-model token parity is required. `engraphis_recall` remains the compatible full-recall +[`BENCHMARKS.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md#public-numeric-evidence-registry) +records the matching suite digest, exact commands, and per-command config digests. The offline +fixture registry intentionally excludes external, model-dependent, consolidation, productivity, +and latency results. Completed retrieval-only diagnostics are published separately in the +[benchmark expansion results](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/BENCHMARK_EXPANSION_RESULTS.md) with redacted immutable +artifacts; no generated-answer, official leaderboard, hosted-latency, or paid result is claimed +here. + +The compact payload shape avoids duplicating full memory bodies when the packed context and source +list are enough. The evaluator tokenizes JSON-shaped full and compact payload proxies built from +recall results; it does **not** serialize the MCP envelope or measure a transport response. The +fixture therefore does not measure model-provider charges, end-to-end task time, or customer cost +savings. + +The measures are deliberately separate and **must not be added together**: chunking counts the +content of retrieved memory records before `ContextPacker`, whereas compact recall counts a +serialized JSON-shape payload proxy. “Tokens to evidence” is the size of the smallest +retrieved memory record holding the reference evidence; it is not latency or end-to-end answer +accuracy. Chunking creates more focused stored records, so this is a context-efficiency result, +not a storage-reduction claim. + +Reproduce the registered quality and token/context measurements without a network connection or +API key: + +```bash +python -m eval.grounded +python -m eval.chunking_eval --dataset eval/datasets/longdoc.jsonl --k 5 +python -m eval.performance --dataset eval/datasets/codemem.jsonl --k 5 --iterations 10 --json +``` + +These are small deterministic correctness and efficiency fixtures, not official LoCoMo / +LongMemEval QA scores or a third-party leaderboard result. Compact-response counts use the exact +`engraphis.regex.v1` counter; the chunking evaluation uses its documented deterministic +normalized-character estimator. Chunking measures retrieved memory content, while compact recall +measures a serialized JSON-shape payload proxy, not an MCP transport response. See the registered +artifact and [`BENCHMARKS.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md) +for definitions, limitations, and canonical external-evaluation requirements. + +
+ +--- + +## Full Engraphis install: pip install "engraphis[all]" + +The complete `engraphis[all]` install is the default way to use Engraphis: it includes the local +dashboard, Smart MCP server, documents, Cloud Sync client, and supported optional integrations. +Python 3.10+ is required. + +```bash +pip install "engraphis[all]" +engraphis-dashboard +``` + +The dashboard opens at [http://127.0.0.1:8700](http://127.0.0.1:8700). Local memory needs no +account or API key. + +### Smaller installation options + +Use a smaller package only when you intentionally need a limited surface. The NumPy-only core +continues to support Python 3.9+. + +| Goal | Install | Start | +|---|---|---| +| Local dashboard and REST API | `pip install "engraphis[server]"` | `engraphis-dashboard` | +| Coding-agent memory over Smart MCP | `pip install "engraphis[mcp]"` | `codex mcp add engraphis -- engraphis-mcp` | +| Native SQLite vector acceleration | `pip install "engraphis[vector]"` | Server entrypoints select it automatically | +| Offline Python library | `pip install engraphis` | `MemoryService.create("engraphis.db")` | + +For MCP clients other than Codex, configure a stdio server whose command is `engraphis-mcp`; see +the [agent connection guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/AGENT_CONNECT.md). + +### Updating + +Use `engraphis-update` to upgrade the installation using its detected install method. Package +metadata does not record which extras were selected, so the updater defaults to the safe +superset `engraphis[all]` rather than silently dropping an optional surface. For a deliberate +selection, set `ENGRAPHIS_UPDATE_EXTRAS` to a comma-separated list (for example +`server,mcp`), or set it to `none` for the base package only. + +> **Upgrading to 1.4:** `engraphis-mcp` now exposes the nine-tool Smart gateway. Integrations that +> require the former 35 direct tool names should run `engraphis-mcp-classic`. The SQLite schema +> in the 1.4.0 release was version 9. Existing v7-to-v8 databases already contain `confidence` +> and `pinned_at`/`unpinned_at`; v9 adds the `memory_tombstones` repository-scope column/table +> and performs a one-time entity-canonicalization repair, then migrates automatically on first +> open. A tombstone with a known `repo_id` is terminal only in that repository; legacy repo-less +> tombstones remain global. See the [1.4.0 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#140---2026-08-02). + +> **Upgrading to 1.5:** schema 10 bounds legacy retention state and schema 11 backfills explicit +> approval only for eligible pre-review local memories. Pending and quarantined evidence remains +> gated. Existing 1.4.x databases migrate automatically when Engraphis 1.5 opens them; see the +> [1.5 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#15---2026-08-04). + +> **Upgrading to 1.6:** existing 1.5 databases migrate automatically through schema 12, which +> classifies content-free erasure markers before sync: existing markers become local-only +> `never_export`, while new secure erasures become `remote_erasure` only for non-secret +> `workspace`/`repo` records already eligible for sharing. Schema 13 adds per-memory hybrid +> logical clocks for deterministic descriptive-state sync and durable, content-free proof that a +> memory crossed a sync boundary. Schema 14 adds the Obsidian collection and import manifests; +> schema 15 generalizes them to source-neutral local documents, preserves temporal source lineage +> across re-imports, binds adapters and target scopes, and retains only bounded, content-free +> per-job format/result metadata. The schema 16 migration persists each import job's optional session target +> and requires source lineage and job-item attachments to remain in that exact session. See the +> [1.6 release notes](https://github.com/Coding-Dev-Tools/engraphis/blob/main/CHANGELOG.md#16---2026-08-15). + +--- + +## What Engraphis gives an agent + +An agent should not have to reconstruct a project from scattered chat history on every task. +Engraphis turns local project knowledge into scoped, time-aware memory; retrieves the evidence +that supports the current question; and returns a bounded, attributable context packet. + +The core task is continuity: retrieve the current, supported project decision without dragging the +whole history into the next prompt. See [measured token and context savings](#measured-token-and-context-savings) +for the short version of how much less history an agent has to carry. + +| Agent need | What Engraphis changes | +|---|---| +| Remember a project across sessions | Stores typed memory in a `workspace → repo → session` hierarchy and provides a last-session handoff. | +| Find support for the current task | Fuses vector, lexical, graph, and code-aware retrieval instead of relying on one search signal; `fast` can skip graph traversal for small or latency-sensitive vaults. | +| Know what is true now and what changed | Preserves bi-temporal history and supersession chains instead of silently overwriting a fact. | +| Avoid confident guesses | Returns cited evidence or explicitly abstains when support is too weak. | +| Avoid dragging the whole project into every prompt | Packs context to a configured hard budget and can return a compact MCP response. | +| Keep knowledge in the operator's control | Runs local-first and offline-capable, with scopes, audit records, and optional privacy-safe receipts. | + +## Dashboard and local UI + +The Engraphis dashboard opens `http://127.0.0.1:8700`. Local memory needs no cloud account, +signup, or API key and stays in a SQLite file on your machine. + +**Ledger** is the primary local interface for recall, memories, graph exploration, provenance, +workspaces, and manual consolidation. **Classic** preserves the former full tool suite; both use +the same local data. Switch in **Manage → Settings → Interface** (Ledger) or **Settings → +Appearance & Engine** (Classic). + +### Start it on every platform + +| Platform | How | +|----------|-----| +| **Windows** | Double-click **Engraphis Dashboard** on your Desktop or Start Menu (install: `engraphis-dashboard --install-shortcuts`) | +| **macOS** | Double-click **Engraphis Dashboard.app** on your Desktop (install: same command) | +| **Linux** | Desktop entry in Applications → Development (GNOME/KDE/etc.) | +| **Docker** | `docker compose up`: see `docker-compose.yml` for the one-command deployment | +| **Any** | `engraphis-dashboard` in a terminal | + +In a source checkout, `scripts/launch_dashboard.ps1` is only a Windows convenience wrapper. It +delegates configuration, startup health, browser opening, and process lifecycle to the same +`engraphis-dashboard` entrypoint rather than maintaining a second behavior path. + +### Accessibility-first inspection, built in + +Inspect memories, supersession diffs, recall scores, timelines, links, consolidation, and audit +records in the dashboard. The offline graph renderer is vendored, and the interface is keyboard- +navigable with light and dark themes. Graph exploration offers a focused **High quality** view and +an explicit worker-backed **Every node** view for complete entity projections up to 20,000 +nodes and 200,000 relationships; see the [graph performance profiles](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/GRAPH_PERFORMANCE.md). + +--- + +## How it works + +Engraphis gives agents durable, scoped, *explainable* project knowledge. The local engine combines +Ebbinghaus decay, bi-temporal facts, and hybrid vector/lexical/graph recall; it runs offline with +SQLite, local embeddings, and `numpy` only. + +- **Grounded and governed:** deterministic conflict resolution, cited answers or abstention, + explicit correction/promotion/forgetting, and a complete history. +- **Agent-ready:** MCP tools, hard-budget context packets, handoffs, and code-aware retrieval. +- **Auditable:** content-free receipt chains, provenance, and temporal/entity/code relationships. +- **Practical:** local file and code ingest, optional PDF/OCR/transcription, and SQLCipher at rest. + +### Optional LLM providers + +The memory engine, embeddings, conflict resolution, and recall stay local without an LLM. An +explicitly configured provider adds structured extraction, cited synthesis, consolidation, and +retention supervision. Configure it in **Settings → Connect an LLM**. The activity view records +outcomes, never keys, prompts, or raw provider responses. See the +[LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md) for setup and privacy choices. + +> Privacy boundary: text sent to an explicitly selected provider leaves the local process under +> that provider's terms. Use `ENGRAPHIS_RETENTION_SUPERVISOR=none` (the default) and the offline +> `chunk` extractor when ingestion must remain entirely local. + +Choose and configure an external LLM with the [LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md), +including OpenAI, Anthropic, Google, OpenRouter, Ollama, Cohere Command, Command Code Provider, +and other compatible endpoints. The guide also covers Codex subscription MCP connections. + +--- + +## Install + +```bash +pip install "engraphis[all]" # self-hosted dashboard, MCP, code graph, documents, transcription, PostgreSQL, and Cloud Sync +pip install "engraphis[server]" # dashboard + REST API +pip install "engraphis[mcp]" # MCP server only +pip install "engraphis[documents]" # PDF + image OCR bindings +pip install "engraphis[transcription]" # faster-whisper audio/video +pip install "engraphis[postgres]" # PostgreSQL schema introspection +pip install "engraphis[code]" # tree-sitter code graph indexing +pip install "engraphis[vector]" # native sqlite-vec exact-KNN acceleration +pip install "engraphis[cloud-sync]" # Cloud Sync client crypto/runtime +pip install "engraphis[encryption]" # SQLCipher encryption-at-rest extra +pip install engraphis # core library: numpy only, fully offline +``` + +The official Docker image includes the local Tesseract executable for image OCR. Outside +Docker, the `documents` extra installs its Python bindings; install Tesseract through your +operating system as well if you enable image OCR. + +The NumPy-only core library supports Python 3.9+. Current patched releases of the WebUI +stack, MCP SDK, image parser, and Cloud Sync client require Python 3.10+, so use Python 3.10 +or newer for the `server`, `mcp`, `documents`, `cloud-sync`, or `all` installation paths. + +The default `NumpyVectorIndex` performs an exact full scan. There is no universal memory-count +cutoff because latency depends on vector size, hardware, filters, and the rest of the recall +pipeline. Measure your machine with `python -m eval.vector_scale --backend numpy`, then run +`python -m eval.performance` on a representative corpus. If exact scans miss your latency target, +install `engraphis[vector]`, create the engine with `vector_backend="sqlite-vec"`, and remeasure. +The stable sqlite-vec `vec0` backend executes exact KNN in native code; it is acceleration, not a +claim of sublinear ANN scaling. See [BENCHMARKS.md](https://github.com/Coding-Dev-Tools/engraphis/blob/main/BENCHMARKS.md) for the reproducible commands +and reporting limits. + +Dashboard, REST, and MCP entrypoints default to `ENGRAPHIS_VECTOR_BACKEND=auto`: they use +sqlite-vec when the `vector` extra is installed and compatible, then safely fall back to NumPy. +Programmatic `MemoryEngine.create()` and `MemoryService.create()` retain the deterministic +`numpy` default unless a backend is requested explicitly. +Use `python -m eval.vector_scale --backend sqlite-vec` for an input-identical direct-search +comparison; setup/index-build time is explicitly excluded from the timed search envelope. + +Persistent vectors fail closed unless the embedder can publish a durable, secret-free space +fingerprint. Sentence Transformers use the loaded Hub commit or a manifest of local artifacts; +when a remote model's immutable identity cannot be resolved, persistent vector recall remains +gated instead of mixing spaces. For programmatic OpenAI-compatible embeddings, construct +`ApiEmbedder` with an operator/provider `space_version`; without it the adapter remains usable for +ephemeral embedding only. Its `base_url` may be a provider root or a `/v1` root and is normalized +to exactly one `/v1/embeddings` endpoint. + +`sqlcipher3-binary` publishes CPython manylinux x86-64 wheels. On that target, +`engraphis[encryption]` installs the driver. The cross-platform `all` extra deliberately +omits it so `all` remains resolvable on macOS, Windows, Linux ARM, and musl; on those +targets, provision a compatible SQLCipher driver separately before enabling a database +key. The programmatic core remains plaintext unless a database key is configured. For a +fresh database, `engraphis-init` enables SQLCipher automatically when a compatible driver is +available, creates a private key sidecar, and can be overridden with `--no-encryption`. + +> **Linux / macOS:** if `pip install` fails with `error: externally-managed-environment`, +> your system Python is marked read-only (PEP 668). Install into a virtual environment +> instead. Run `python3 -m venv venv && source venv/bin/activate && pip install "engraphis[server]"` +> Alternatively, use Docker (`docker compose up`). `pipx install "engraphis[server]"` also works. + +> First run downloads `all-MiniLM-L6-v2` (~80 MB). Without it, the engine falls back +> to deterministic feature hashing so it always runs offline. That fallback captures lexical +> overlap, not meaning: recall and grounded MCP responses set `degraded_mode=true` and +> `semantic_support=false`, and disable vector retrieval plus semantic-cosine evidence. Install +> a declared embedding model for semantic retrieval. + +> To require a model that is already local, set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` +> or `local:`. This path never downloads a model. If it is unavailable, Engraphis +> explicitly enters lexical degraded mode instead of presenting hash-vector scores as semantic. + +--- + +## Quickstart: dashboard + +```bash +pip install "engraphis[server]" +engraphis-dashboard # → http://127.0.0.1:8700 +engraphis-dashboard --install-shortcuts # → Desktop + Start Menu icons +``` + +> **Offline first run:** the first launch downloads the `all-MiniLM-L6-v2` embedding model +> (~80 MB), then runs fully offline. To stay offline-only, set +> `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` (never downloads; unknown local models +> enter lexical degraded mode instead of faking semantic scores). Extraction defaults to +> `ENGRAPHIS_EXTRACTOR=none` (verbatim writes), the vector backend defaults to `auto` (native +> acceleration when installed, otherwise NumPy), and recall without a usable semantic space +> reports `degraded_mode=true` with lexical/graph recall. Run `engraphis-init --check` to +> verify the install, extras, and database writability. + +### Docker + +```bash +docker compose up # → http://127.0.0.1:8700 +``` + +For Docker Compose persistence and loopback-port configuration, see the +[Docker deployment guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCKER.md). +`engraphis-server` and `engraphis server` are headless compatibility aliases +for this same v2 service, so every public surface has the same scoped recall and retention model. + +For optional LAN exposure, token configuration, and HTTP MCP setup, see the +[Docker deployment guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCKER.md). + +Set `ENGRAPHIS_API_TOKEN` to require API authentication and `ENGRAPHIS_DB_KEY` to encrypt +the local database at rest. Hosted-plan credentials configure customer clients; they do not +install premium server implementations into this image. See `docker-compose.yml` for options. + +--- + +## Quickstart: MCP server (for coding agents) + +```bash +pip install "engraphis[mcp]" +engraphis-init # writes ~/.engraphis/config.env + prints config snippets +claude mcp add engraphis -- engraphis-mcp +codex mcp add engraphis -- engraphis-mcp # Codex subscription + +``` + +> **Offline first run:** the first tool call lazily loads the `all-MiniLM-L6-v2` embedding +> model (~80 MB, same download as the dashboard), then memory runs fully offline with no API +> key. To stay offline-only, set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path` (never +> downloads); extraction defaults to `ENGRAPHIS_EXTRACTOR=none`, the vector backend `auto` +> falls back to NumPy without the `vector` extra, and recall without a usable semantic space +> reports `degraded_mode=true` with lexical/graph recall. Run `engraphis-init --check` to +> verify the install and database path before registering the server. + +For Codex subscription setup and verification, see the [agent connection guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/AGENT_CONNECT.md) +and the [LLM provider guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LLM_PROVIDERS.md). + +`engraphis-mcp` is zero-configuration Smart MCP: agents begin with nine compact tools for sessions, +prompt-ready recall, durable memory, governed record read/update, conflict review, action discovery, +and safe execution. For code graphs, +governance, audit, or other advanced work, the agent calls `engraphis_discover_actions` and then +the indicated read or action executor; no profile selection is required. The gateway validates +the discovered capability again before it runs it, and clients remain responsible for their +normal destructive-action approval boundary. + +Existing clients that pin the historical 35 named tools can use +`engraphis-mcp-classic` (or `engraphis-mcp-http --classic`). The complete classic inventory, +including `engraphis_check_update`, is in the [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md). + +### Pi extension + +For installation, configuration, lifecycle commands, and the local trust boundary, see the +[Pi extension guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/pi/README.md). + +### Command Code SessionStart hook + +`integrations/commandcode/` ships a SessionStart hook that warms up a new +session with bounded, recalled context from the local Engraphis gateway. Fails +open on timeout and is installed via `python scripts/install_cc_hook.py`. + +### prime-agent fleet + +`integrations/prime_agent/` ships a first-party Python package for +[PrimeIntellect prime-agent](https://github.com/PrimeIntellect-ai/prime-agent) +that exposes the same nine Smart MCP tools, with a `PrimeAgentFleet` of eight +named sub-agents (`researcher`, `planner`, `coder`, `reviewer`, `tester`, +`documenter`, `monitor`, `integrator`) sharing one `engraphis-mcp` stdio +subprocess. Install via `pip install ./integrations/prime_agent` and register +with `python scripts/install_prime_agent.py`. See the +[prime-agent integration guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/prime_agent/README.md). + +**What the integration is.** A `PrimeAgentFleet` is a thin Python layer +around the same `engraphis-mcp` Smart gateway every other host uses. At +runtime the fleet holds one shared `EngraphisMcpClient`, which owns one +`engraphis-mcp` subprocess over JSON-RPC stdio. Each of the eight named +sub-agents gets its own Engraphis session (started lazily on first tool use) +and its own default `repo` scope, so per-role memory is isolated while the +local gateway stays single-process. The eight sub-agent names +(`researcher`, `planner`, `coder`, `reviewer`, `tester`, `documenter`, +`monitor`, `integrator`) are the fixed default; pass `agent_names=[...]` to +`PrimeAgentFleet(...)` for a custom set. Concurrent tool calls serialize at +the JSON-RPC frame layer through an `asyncio.Lock`, so framework-level +parallelism (eight sub-agents reasoning at once) is preserved while the +underlying MCP transport remains one ordered stream. The only integration +surface is `EngraphisPrimeAgent.register()` in +`integrations/prime_agent/src/engraphis_prime_agent/agent.py` -- that is the +single adapter point to override if prime-agent's tool-registration API +differs from the assumed `target.register_tool(name, fn, schema=...)` +contract. + +The design -- eight named sub-agents, one shared stdio subprocess, +per-agent session bootstrap, and `ENGRAPHIS_*`-only environment forwarding +to the gateway -- is recorded in `~/.commandcode/plans/prime-agent-integration.md` +on the host where the integration was developed. When that host plan is not +available (other contributor machines, CI), the same design is summarized in +the PR description that introduced the integration and in the +[prime-agent integration guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/integrations/prime_agent/README.md) +("Architecture" and "Concurrency model" sections). + +## Quickstart: repository graph + +```bash +pip install "engraphis[code]" +engraphis-graph index -w acme -r api --root . +engraphis-graph search -w acme -r api "UserService" +# `query`/`explain` blend code search with your stored memories: query matches symbol +# and file NAMES (a full question sentence won't match anything), and explain's answer +# is drawn from memories recorded against the repo; both are empty on a fresh index. +engraphis-graph query -w acme -r api "UserService" +engraphis-graph explain -w acme -r api "why does deploy depend on approval?" +engraphis-graph path -w acme -r api UserService DatabasePool +engraphis-graph impact -w acme -r api --root . --git-range origin/main...HEAD +engraphis-graph prs -w acme -r api --base main --head HEAD +engraphis-graph export -w acme -r api -o engraphis-graph-out +engraphis-graph install-merge-driver --root . +``` + +The export contains `graph.json`, a self-contained `graph.html`, and `GRAPH_REPORT.md`. +Indexing supports Python, JavaScript, TypeScript, Go, Rust, Java, C#, C, C++, SQL, and +Terraform. Tree-sitter is used when available; the dependency-free regex backend remains a +functional fallback. Definitions, methods, calls, imports, ownership, variables, +inheritance/implementation, and docstrings/comments are indexed. Indexing is incremental by +content hash, honors `.engraphisignore`, and does not follow file symlinks outside the repository +root. Call edges are name-based and best-effort rather than type-resolved. The optional Git merge +driver validates bounded graph JSON and deterministically unions nodes and edges instead of +choosing one export side. + +For a read-only recall and graph API that can be shared without exposing write operations: + +```bash +pip install "engraphis[server]" +engraphis-graph-server # API at http://127.0.0.1:8720; schema at /openapi.json +``` + +A non-loopback bind fails closed unless `ENGRAPHIS_GRAPH_TOKEN` (or +`ENGRAPHIS_API_TOKEN`) is set. See [the v3 architecture/design document](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md). + +--- + +## Quickstart: Python library + +```python +from engraphis.service import MemoryService + +mem = MemoryService.create("engraphis.db") +mem.remember("Auth migrated from JWT to PASETO.", workspace="acme", repo="api") +hit = mem.recall("why did we change auth?", workspace="acme", repo="api") +print(hit["context"]) +``` + +The same `MemoryService` backs the dashboard and the MCP server. The package root also +intentionally exposes the low-level engine facade (`MemoryEngine`, `create_memory_engine`) +for advanced composition, while `MemoryService` remains the high-level service API. + +New writes support `session`, `repo`, and `workspace` visibility. `scope="user"` is reserved and +rejected until records carry an immutable owner identity; it must not be treated as private +per-person memory. Historical user-scope rows remain workspace-bound for compatibility. + +After an upgrade, `stats()` reports prompt-eligibility counts and active embedding-space +coverage. Zero-result recall identifies a review-gated scope instead of silently looking empty, +and `engraphis-cli review list|approve` provides a dry-run-first local bulk workflow. Embedding +model changes trigger a guarded rebuild; vector recall stays disabled until every stored vector +matches the new fingerprint. See [recall recovery](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/RECALL_RECOVERY.md). + +Agent hosts can avoid retrieval when their existing history already fits: + +```python +decision = mem.adaptive_context( + "what should the agent do next?", + current_history, + workspace="acme", + repo="api", + max_context_tokens=8_192, + retrieval_token_budget=1_024, +) +prompt_context = decision["context"] +``` + +The decision is `history_bypass` when the history fits, `retrieval` when compact evidence is +strong, and `history_fallback` when weak retrieval should widen back to recent raw history. + +For an agent prompt, prefer `engraphis_recall_context`: it returns one hard-budget packed +`context` plus compact `sources`, deterministic `usage` accounting (`budget_tokens`, `context_tokens`, +`source_tokens`, `saved_tokens`, `savings_ratio`, `packed_count`, `omitted_count`, and +`token_counter`), and optional diagnostics. Accounting is exact for the named counter; inject the +reader's tokenizer when reader-model token parity is required. `engraphis_recall` remains the compatible full-recall surface; use `response_mode="compact"` when the packed context is enough and full memory bodies would duplicate it. For advanced query-planning configuration, see the [architecture guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md#query-planning). @@ -577,338 +577,338 @@ and title-only revisions preserve valid bindings. History preserves the original MCP response trimming removes binding metadata whenever its supporting context is omitted. For bi-temporal reads, `valid_at` selects what was true at a Unix timestamp and `known_at` selects -what Engraphis had learned then. `as_of` remains a compatibility alias for `valid_at`; supplying -both is allowed only when they match. - -For a mutable claim, pass a stable `subject_key` and optional `claim_kind`, such as -`subject_key="api.rate_limit", claim_kind="configured_value"`. Offline conflict resolution -deterministically adds, reinforces, relates, or supersedes records while preserving temporal -history; it does not need an LLM. Matching claim identities let it supersede substantially -reworded mutable facts. Without them, the dependency-free lexical embedder cannot reliably infer -that a paraphrase is a contradiction, so keep both records or use an explicit `correct` operation. - ---- - -## Govern memories without losing history - -Engraphis separates automatic write resolution from explicit human governance: - -| Operation | Use it when | What happens to history | -|---|---|---| -| `remember` | Adding or restating one fact | Adds, reinforces, safely supersedes, or relates an uncertain neighbor | -| `correct` | Replacing one known-wrong memory | Closes the old validity window and links the replacement | -| `promote` | A narrow learning now applies more broadly | Writes a wider-scope successor and closes/links the source instead of editing scope in place | -| `merge` | Combining two or more overlapping memories | Retires every source and creates one memory that supersedes all of them | -| `retire` | Removing a memory from live recall | Bi-temporally closes it; the audit/history record remains | -| `consolidate` | Distilling recurring episodic memories automatically | Creates linked semantic digests; source episodes remain live | - -Manual N→1 merge is available through `MemoryService.merge()` and `POST /api/merge`: - -```python -a = mem.remember("Deploys happen Friday at 3pm.", workspace="acme") -b = mem.remember("We deploy Fridays around 15:00.", workspace="acme") - -merged = mem.merge( - [a["id"], b["id"]], - "Deploys ship every Friday at approximately 15:00.", - workspace="acme", - reason="deduplicate the deployment schedule", -) -print(merged["compaction"]) -``` - -`retire` is intentionally not deletion: it preserves temporal history, FTS, and vector -evidence for historical reads. If a credential was captured, new writes are blocked before -storage; for a legacy leak use the explicitly destructive `MemoryService.secure_erase()` or -`POST /api/secure-erase`/`engraphis_secure_erase`. That flow removes the one memory and local -FTS/vector-index and derived graph/link rows, runs SQLite secure-delete, WAL checkpoint, and -VACUUM, and scans recognised local SQLite recovery backups. It cannot erase exports, filesystem -snapshots, remote peers, unknown backups, or information a running/compromised agent already -read; rotate the credential. See [secure-erasure limits](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SECURE_ERASURE.md). `forget` -remains a deprecated compatibility alias for `retire`. - -All sources must belong to the named workspace. The result inherits the strictest source -sensitivity, remains untrusted if any source was untrusted, and stays pinned if any source was -pinned. The full multi-predecessor chain remains visible through inspection, Why, and Timeline. - ---- - -## Free forever vs. hosted plans - -The core engine, local dashboard, MCP server, and manual consolidation are Apache-2.0 and free. -**Pro and Team are services** that provide optional access to the official hosted service; its -control-plane, billing, relay, compute, and Team identity modules live in a private repository. -They do not limit the local core. See -[hosted plans](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/HOSTED_PLANS.md), [licensing](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md), and -[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for service boundaries, lifecycle, and pricing. - -[Subscribe to Pro](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_pricing#billing) -to support the project and add hosted services. - -[Compare hosted plans](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro#billing) -when you are ready to evaluate the service boundary and billing options. - -| | Free (available now) | Pro: $10/mo or $100/yr | Team: $20/seat/mo or $200/seat/yr | -|---|---|---|---| -| Dashboard WebUI (with built-in inspector) | ✓ | ✓ | ✓ | -| Memory engine + Smart MCP (Classic 35-tool compatibility) | ✓ | ✓ | ✓ | -| Version-chain diffs, offline knowledge graph | ✓ | ✓ | ✓ | -| Manual local consolidation (dry-run by default) | ✓ | ✓ | ✓ | -| Local workspace export (portable v2 JSON: memories, source manifests, graph/code evidence, sessions, audit, and receipts) | ✓ | ✓ | ✓ | -| Hosted Cloud Sync | | ✓ | ✓ | -| Hosted Analytics | | ✓ | ✓ | -| Hosted Auto Consolidation + retention policy | | ✓ | ✓ | -| Hosted Auto Dreaming + managed proposals | | ✓ | ✓ | -| Priority support | | ✓ | ✓ | -| Hosted multi-user dashboard: invitations, logins, roles, seat management | | | ✓ | -| Hosted Team audit log + CSV export | | | ✓ | -| 72-hour pending invitations (resend/revoke) | | | ✓ | -| Scoped, expiring per-user agent and sync tokens | | | ✓ | - ---- - -## MCP tools - -Engraphis exposes a zero-configuration Smart MCP gateway plus a 35-tool Classic compatibility -server across memory, recall, code graphs, governance, sessions, and privacy-safe audit receipts. -The focused [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md) is the source for -the full inventory and parameters. - ---- - -## Graphs and privacy-safe receipts - -Memory, entity, and code relationships live in one local graph. Engraphis also provides -content-free operation receipts for inspectable audit evidence. See the -[architecture](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md), [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md), and -[security policy](https://github.com/Coding-Dev-Tools/engraphis/blob/main/SECURITY.md) for the data model, tools, and guarantees. - ---- - -## Cloud sync - -Cloud Sync is an optional hosted Pro/Team service. The public package includes the customer client -and deterministic merge implementation; hosted relay and account operations are separate. See -[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for setup, encryption, merge behavior, and the local folder exchange. - -The public package ships the same sync client as a console script and CLI verb: -`engraphis-sync` (installed entry point), `engraphis sync ...`, and -`python -m scripts.sync --status` for local-only state without network activity. See -[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for -flags, encryption, merge behavior, and the local folder exchange. - ---- - -## Security and trust boundaries - -Engraphis is local-first and binds to loopback by default. Read the -[security policy](https://github.com/Coding-Dev-Tools/engraphis/blob/main/SECURITY.md) before remote deployment or integrating external resources; it -covers supported versions, data protections, threat model, and vulnerability reporting. - ---- - -## Encryption at rest - -Set `ENGRAPHIS_DB_KEY` (or `ENGRAPHIS_DB_KEY_FILE`) and install the extra: - -```bash -pip install "engraphis[encryption]" -``` - -The entire main memory database file is transparently encrypted with AES-256 via SQLCipher; -full-text search, the graph, and every query keep working unchanged. Customer authentication -and managed-service state use their respective deployment protections. When a key is set for the -main database, Engraphis **fails closed with an error** rather than silently falling back to -plaintext. Generate a strong key: - -```bash -python -c "import secrets; print(secrets.token_hex(32))" -``` - -When using `ENGRAPHIS_DB_KEY_FILE`, provision a regular secret file readable only by the -service identity. Engraphis rejects links, reparse points, hard links, malformed text, and -oversized key files rather than following an unexpected filesystem object. - -> An existing plaintext database cannot be opened with a key: migrate it (dump → import -> into a fresh keyed DB). See `.env.example` for all encryption options. - ---- - -## Import files and folders - -The dependency-free universal core scans Markdown, plain text, RST, HTML, JSON/JSONL, CSV/TSV, -configuration/XML text, source code, RTF, DOCX/ODT, XLSX/ODS, PPTX/ODP, and EPUB into the normal -v2 memory path. Installed local resource adapters add PDF text, image OCR, and explicitly -local-model audio/video transcription. -Start with a zero-write -preview, then confirm the same source collection explicitly: - -```bash -engraphis import documents /path/to/collection --workspace acme --dry-run -engraphis import documents /path/to/collection --workspace acme --repo product --yes -``` - -The CLI never downloads an embedding model during import. Use a model that is already cached, -set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path`, or explicitly set -`ENGRAPHIS_EMBED_MODEL` to an empty value to use dependency-free deterministic hashing in -lexical degraded mode. - -The dashboard’s **Import local documents** flow offers the same preview, target scope, source -label, conflict policy, cancellation, and resumable progress. Re-imports are idempotent, -preserve temporal history, and report source removals without hard-deleting memories. Obsidian -remains the rich Markdown adapter for frontmatter, aliases, wikilinks, and attachment references: - -```bash -engraphis import obsidian /path/to/vault --workspace acme --dry-run -``` - -See the [document import guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCUMENT_IMPORT.md) -for supported formats, source safety, resume and conflict behavior, optional adapters, and -limitations; see the [Obsidian adapter guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/OBSIDIAN_IMPORT.md) -for Markdown-specific behavior. - ---- - -## Consolidation and automation - -Manual consolidation is free, local, and dry-run by default; use the dashboard, SDK, CLI, or -MCP. Hosted Pro and Team automation is optional managed compute that produces reviewable -proposals rather than silently changing local data. See [hosted plans](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/HOSTED_PLANS.md), -[licensing](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md), and the [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md) for scope and use. - ---- - -## Configuration - -Values come from the process environment. Engraphis also loads the owner-private -`~/.engraphis/config.env`; `ENGRAPHIS_ENV_FILE` can select another absolute owner-private regular -file. It never searches the working directory for `.env`, and explicit process variables win. - -| Env Var | Default | Description | -|---------|---------|-------------| -| `ENGRAPHIS_ENV_FILE` | `~/.engraphis/config.env` | Optional trusted config leaf selected before trusted values load. Its bounded dependency-free parser performs no interpolation. An explicit value must be an absolute path to an owner-private regular file; arbitrary working-directory `.env` files are ignored. | -| `ENGRAPHIS_DB_PATH` | Source: `/engraphis.db`; installed: platform user-data directory | SQLite database file. Installed defaults are `%LOCALAPPDATA%\engraphis\engraphis.db` (Windows), `~/Library/Application Support/engraphis/engraphis.db` (macOS), and `$XDG_DATA_HOME/engraphis/engraphis.db` or `~/.local/share/engraphis/engraphis.db` (Linux). The environment variable overrides every default; a relative value is resolved from the trusted `~/.engraphis/config.env` directory so launch CWD cannot select a different workspace database. | -| `ENGRAPHIS_SQLITE_DURABILITY` | `durable` | Writable file databases use WAL and FULL commit synchronization. Explicit `balanced` selects NORMAL, which can lose recent acknowledged writes after OS/power failure. Effective settings appear in diagnostics; see [SQLite durability](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SQLITE_DURABILITY.md). | -| `ENGRAPHIS_HOST` | `127.0.0.1` | Server bind address | -| `ENGRAPHIS_PORT` | `8700` | Dashboard port. A platform-injected `$PORT` (Railway/Fly/Heroku) takes precedence over this value for the dashboard bind; Compose pins both to `ENGRAPHIS_COMPOSE_PORT` so the mapping stays in sync | -| `ENGRAPHIS_SERVICE_MODE` | `customer` | The public package supports only `customer`; hosted vendor, relay, compute, and worker roles are not distributed here | -| `ENGRAPHIS_API_TOKEN` | Not set | Optional bearer credential for this single-user local customer node; never reuse a hosted credential | -| `ENGRAPHIS_CORS_ORIGINS` | loopback on `ENGRAPHIS_PORT` | Comma-separated REST CORS allow-list; defaults to `127.0.0.1` and `localhost` on the configured port | -| `ENGRAPHIS_INDEX_ROOTS` | Working, home, and temporary directories | Optional path-separator-delimited absolute-path allow-list that replaces the default roots accepted by local code indexing | -| `ENGRAPHIS_HTTP_INDEX_ROOT` | First `ENGRAPHIS_INDEX_ROOTS` entry, or current directory | Single root for dashboard and REST `POST /api/code/index`; submitted paths resolve beneath it. An explicit root (or fallback entry) must be absolute; an explicit HTTP root is included in the engine-approved set. MCP and CLI indexing continue to use `ENGRAPHIS_INDEX_ROOTS`. | -| `ENGRAPHIS_DB_KEY` | Not set | Encrypt the database at rest (SQLCipher). Or use `ENGRAPHIS_DB_KEY_FILE` | -| `ENGRAPHIS_EMBED_MODEL` | `sentence-transformers/all-MiniLM-L6-v2` | sentence-transformers model | -| `ENGRAPHIS_EMBED_REVISION` | Not set | Optional immutable lowercase 40-hex Hugging Face commit for the embedding model. Loaded Hub commits or local artifact manifests identify persistent vector spaces; unresolved mutable identities keep vector recall fail-closed. | -| `ENGRAPHIS_RERANK_MODEL` | Not set | Optional sentence-transformers cross-encoder reranker | -| `ENGRAPHIS_RERANK_REVISION` | Not set | Optional immutable lowercase 40-hex Hugging Face commit for the reranker | -| `ENGRAPHIS_REQUIRE_IMMUTABLE_MODELS` | `false` | When enabled, require a 40-hex commit before loading remote embedding models, rerankers, or chunk tokenizers; `local:` selectors and filesystem paths remain permitted | -| `ENGRAPHIS_REQUIRE_EXACT_BACKENDS` | `false` | When enabled, dashboard and standalone MCP startup fails if a configured optional backend is unavailable instead of silently falling back | -| `ENGRAPHIS_EXTRACTOR` | `none` | `none` = verbatim; `chunk` = offline structure-aware chunks; `llm` = free-form LLM facts; `llm_structured` = schema-validated facts + graph metadata | -| `ENGRAPHIS_CHUNK_TOKENIZER_MODEL` | Not set | Optional Hugging Face tokenizer used to enforce chunk budgets with the downstream reader's real tokenization; requires the optional `transformers` package | -| `ENGRAPHIS_CHUNK_TOKENIZER_REVISION` | Not set | Optional immutable tokenizer/model revision recorded in the chunk-counter identity; pin this for reproducible benchmark artifacts | -| `ENGRAPHIS_GRAPH_EXTRACTOR` | `regex` | `regex` = offline heuristic NER; `none` = disable heuristic text extraction (validated `llm_structured` metadata still feeds the graph) | -| `ENGRAPHIS_RETENTION_SUPERVISOR` | `none` | `none` = deterministic only; `llm` = sends a bounded excerpt to the configured provider for advisory ephemeral/normal/critical classification | -| `ENGRAPHIS_ALLOW_AUTOMATIC_CRITICAL_RETENTION` | `false` | Opt in only when an LLM supervisor may automatically assign the long-lived `critical` class; explicit user-selected critical retention is unaffected | -| `ENGRAPHIS_WHISPER_MODEL` | Not set | Enables local faster-whisper audio/video transcription | -| `ENGRAPHIS_POSTGRES_DSN` | Not set | CLI-only PostgreSQL source; used for the connection and never stored | -| `ENGRAPHIS_POSTGRES_CONNECT_TIMEOUT` | `10` | PostgreSQL introspection connection timeout in seconds (bounded to 1--120) | -| `ENGRAPHIS_POSTGRES_STATEMENT_TIMEOUT_MS` | `30000` | Per-introspection PostgreSQL statement timeout in milliseconds (bounded to 1--300000) | -| `ENGRAPHIS_GRAPH_TOKEN` | Not set | Bearer token for `engraphis-graph-server`; required off-loopback | -| `ENGRAPHIS_GRAPH_HOST` / `ENGRAPHIS_GRAPH_PORT` | `127.0.0.1` / `8720` | Read-only graph/recall server bind address | -| `ENGRAPHIS_LLM_PROVIDER` | `openai` | `openai \| anthropic \| google \| openrouter \| custom` | -| `ENGRAPHIS_LLM_MODEL` | `gpt-4o-mini` | Model name (provider-specific) | -| `ENGRAPHIS_LLM_API_KEY` | Not set | API key for chat/synthesis, `llm` / `llm_structured` extraction, and structured consolidation | -| `ENGRAPHIS_LLM_BASE_URL` | Not set | Base URL for openrouter / custom OpenAI-compatible endpoints | -| `ENGRAPHIS_LLM_AUTO_EXTRACT` | `0` | Opt in to switching the running engine to `llm_structured` after a successful live connection test; the dashboard's extraction Off button persists `0`, and its On button restores `1` | -| `ENGRAPHIS_FORWARDED_ALLOW_IPS` | *(none)* | Proxies trusted for forwarded client/TLS headers (`*` only when the service is reachable exclusively through that proxy) | -| `ENGRAPHIS_LOCAL_TRUSTED_PEERS` | *(none)* | Exact peers/CIDRs treated as local without forwarding headers; use only for trusted Docker/LAN peers, never public deployments | -| `ENGRAPHIS_UPDATE_CACHE` | `86400` | Update-check cache TTL in seconds, bounded to `1..31622400`; this is never a cache-file path | -| `ENGRAPHIS_UPDATE_CHECK` | Off | Opt-in release reminder surfaced in the dashboard, server startup log, and MCP. Update checks run only when this is set to an affirmative value; `0` keeps them off. | -| `ENGRAPHIS_UPDATE_URL` | Not set | Overrides the release-check source URL; the outbound client accepts HTTPS and rejects private/reserved destinations. | -| `ENGRAPHIS_CLOUD_CONTROL_URL` | hosted default | Official entitlement, organization, and credential control API. A saved rotating credential stays bound to the control endpoint recorded for its family; reconnect to change it. | -| `ENGRAPHIS_CLOUD_COMPUTE_URL` | hosted default | Official Analytics and managed-automation API. A saved rotating credential stays bound to its recorded compute endpoint; reconnect to change it. | -| `ENGRAPHIS_CLOUD_ORGANIZATION_ID` | Not set | Hosted organization bound to this customer session | -| `ENGRAPHIS_CLOUD_REFRESH_CREDENTIAL` | Not set | Bootstrap-only rotating hosted credential; after first use the owner-only cloud session replacement takes precedence | -| `ENGRAPHIS_CLOUD_TOKEN_SUBJECT` | `member` | Subject fixed during hosted bootstrap (`device` or `member`); set explicitly with an environment-only refresh credential | -| `ENGRAPHIS_CLOUD_ACCESS_TOKEN` | Not set | Optional short-lived access token for ephemeral jobs | -| `ENGRAPHIS_MANAGED_COMPUTE_CONSENT` | *(unset)* | Deny-only operator override: `0` pauses readable managed processing. A truthy value cannot grant approval. Each workspace requires explicit confirmation in Manage → Settings; encrypted sync is separate | - -The optional cross-encoder reranker is model- and hardware-dependent. Treat its quality and -latency as deployment-specific until a versioned model identity, exact configuration, and -reproducible evaluation artifact are available for the comparison being reported. - -See `.env.example` for the full variable inventory. Supply those values through the process -environment or the trusted config file above; copying it to an arbitrary `./.env` does not make -Engraphis load it. - -> **Ablation fixture:** `python -m eval.ablation` is an offline deterministic check that prints -> `recall@5` comparisons for vector-only and hybrid retrieval, multi-hop graph arms, and -> retrieval policies, plus ordinary-recall age and semantic-confidence checks. It does not -> produce MRR, hit@5, or ms/query results. Use `python -m eval.reinforcement` for retention -> trajectories, and register evidence before quoting any benchmark results. - ---- - -## Project structure - -``` -engraphis/ -├── engraphis/ -│ ├── core/ # v2 engine: interfaces, store, recall, scoring, schema, sync -│ ├── backends/ # pluggable embedder / vector index / reranker / codegraph / sync transports / encryption -│ ├── factory.py # outer v2 composition root; selects and injects concrete backends -│ ├── service.py # validated MemoryService facade -│ ├── mcp_server.py # Smart MCP gateway + 35-tool Classic compatibility server -│ ├── dashboard_app.py # dashboard WebUI (FastAPI) -│ ├── dashboard_assets/ # primary Ledger interface + graph engine -│ ├── classic_assets/ # selectable full operator dashboard backup -│ ├── read_only_api.py # token-protected recall/repository-graph HTTP surface -│ ├── hosted_client.py # hosted URLs, plan labels, and endpoint validation only -│ ├── licensing.py # compatibility facade for hosted presentation metadata -│ ├── cloud_session.py # rotating hosted customer-session client -│ ├── cloud_features.py # consented managed-feature protocol client -│ ├── config.py / app.py # env settings / REST server -│ └── static/ # compatibility dashboard asset paths -├── eval/ # offline retrieval eval harness + datasets -├── tests/ # offline-first pytest suite and release/security contracts -├── scripts/ # dashboard, server, graph, CLI, connect, update, consolidation, sync -├── docs/ # product, API, hosting, sync, and provider guides -├── Dockerfile / docker-compose.yml -└── pyproject.toml -``` - -New capability belongs in the v2 path (`engraphis/core/`, `engraphis/backends/`, and -`MemoryService`) behind the interfaces in `core/interfaces.py`. Algorithm modules in `core/` -remain backend-agnostic; `engraphis/factory.py` is the outer composition root used by -`engraphis.create_memory_engine()` and the compatibility `MemoryEngine.create()` entry point, then -injects the selected collaborators into `core/engine.py`. The flat-namespace v1 server under -`engraphis/app.py`, `routes/`, `stores/`, and `engines/` remains a -compatibility/reference surface; `engraphis-dashboard`, the MCP server, and the Python quickstart -above use v2. - ---- - -## License - -Apache-2.0. See [LICENSE](https://github.com/Coding-Dev-Tools/engraphis/blob/main/LICENSE) and [NOTICE](https://github.com/Coding-Dev-Tools/engraphis/blob/main/NOTICE). "Engraphis" is a trademark of the -Engraphis project; the license does not grant trademark rights. Code already distributed -under Apache-2.0 keeps that grant; later releases cannot retroactively withdraw it. The -official hosted control plane, its production credentials and records, managed operations, -support, and future separately delivered commercial modules are outside the public source -grant. See [`docs/LICENSING.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md) for the complete boundary. - -### Reliability implementation candidate - -The current source uses schema 18 for durable, content-free vector-index repair and -atomic memory-command receipts. Upgrades use the existing verified-backup migration path. -The [rework execution register](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/REWORK_EXECUTION.md) records the current findings, -compatibility decisions, acceptance evidence, remaining work and recovery procedure. -See [the reliability program](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/RELIABILITY_PROGRAM.md) for exact implementation, -validation, migration and release boundaries. Managed processing now requires explicit -workspace approval in Settings. Existing installations start with readable -uploads paused until confirmed; connecting an account does not grant approval. - -For setup diagnostics use `engraphis-init --check --json`. New configurations get an -owner-private local API token. Existing configs are preserved. Record selected install -capabilities with `engraphis-init --extras server,mcp` or `--extras none`; future updates -preserve that choice. `ENGRAPHIS_UPDATE_EXTRAS` remains an explicit override. +what Engraphis had learned then. `as_of` remains a compatibility alias for `valid_at`; supplying +both is allowed only when they match. + +For a mutable claim, pass a stable `subject_key` and optional `claim_kind`, such as +`subject_key="api.rate_limit", claim_kind="configured_value"`. Offline conflict resolution +deterministically adds, reinforces, relates, or supersedes records while preserving temporal +history; it does not need an LLM. Matching claim identities let it supersede substantially +reworded mutable facts. Without them, the dependency-free lexical embedder cannot reliably infer +that a paraphrase is a contradiction, so keep both records or use an explicit `correct` operation. + +--- + +## Govern memories without losing history + +Engraphis separates automatic write resolution from explicit human governance: + +| Operation | Use it when | What happens to history | +|---|---|---| +| `remember` | Adding or restating one fact | Adds, reinforces, safely supersedes, or relates an uncertain neighbor | +| `correct` | Replacing one known-wrong memory | Closes the old validity window and links the replacement | +| `promote` | A narrow learning now applies more broadly | Writes a wider-scope successor and closes/links the source instead of editing scope in place | +| `merge` | Combining two or more overlapping memories | Retires every source and creates one memory that supersedes all of them | +| `retire` | Removing a memory from live recall | Bi-temporally closes it; the audit/history record remains | +| `consolidate` | Distilling recurring episodic memories automatically | Creates linked semantic digests; source episodes remain live | + +Manual N→1 merge is available through `MemoryService.merge()` and `POST /api/merge`: + +```python +a = mem.remember("Deploys happen Friday at 3pm.", workspace="acme") +b = mem.remember("We deploy Fridays around 15:00.", workspace="acme") + +merged = mem.merge( + [a["id"], b["id"]], + "Deploys ship every Friday at approximately 15:00.", + workspace="acme", + reason="deduplicate the deployment schedule", +) +print(merged["compaction"]) +``` + +`retire` is intentionally not deletion: it preserves temporal history, FTS, and vector +evidence for historical reads. If a credential was captured, new writes are blocked before +storage; for a legacy leak use the explicitly destructive `MemoryService.secure_erase()` or +`POST /api/secure-erase`/`engraphis_secure_erase`. That flow removes the one memory and local +FTS/vector-index and derived graph/link rows, runs SQLite secure-delete, WAL checkpoint, and +VACUUM, and scans recognised local SQLite recovery backups. It cannot erase exports, filesystem +snapshots, remote peers, unknown backups, or information a running/compromised agent already +read; rotate the credential. See [secure-erasure limits](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SECURE_ERASURE.md). `forget` +remains a deprecated compatibility alias for `retire`. + +All sources must belong to the named workspace. The result inherits the strictest source +sensitivity, remains untrusted if any source was untrusted, and stays pinned if any source was +pinned. The full multi-predecessor chain remains visible through inspection, Why, and Timeline. + +--- + +## Free forever vs. hosted plans + +The core engine, local dashboard, MCP server, and manual consolidation are Apache-2.0 and free. +**Pro and Team are services** that provide optional access to the official hosted service; its +control-plane, billing, relay, compute, and Team identity modules live in a private repository. +They do not limit the local core. See +[hosted plans](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/HOSTED_PLANS.md), [licensing](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md), and +[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for service boundaries, lifecycle, and pricing. + +[Subscribe to Pro](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_pricing#billing) +to support the project and add hosted services. + +[Compare hosted plans](https://api.engraphis.com/account?plan=pro&interval=monthly&utm_source=engraphis&utm_medium=docs&utm_campaign=pro_conversion&utm_content=readme_intro#billing) +when you are ready to evaluate the service boundary and billing options. + +| | Free (available now) | Pro: $10/mo or $100/yr | Team: $20/seat/mo or $200/seat/yr | +|---|---|---|---| +| Dashboard WebUI (with built-in inspector) | ✓ | ✓ | ✓ | +| Memory engine + Smart MCP (Classic 35-tool compatibility) | ✓ | ✓ | ✓ | +| Version-chain diffs, offline knowledge graph | ✓ | ✓ | ✓ | +| Manual local consolidation (dry-run by default) | ✓ | ✓ | ✓ | +| Local workspace export (portable v2 JSON: memories, source manifests, graph/code evidence, sessions, audit, and receipts) | ✓ | ✓ | ✓ | +| Hosted Cloud Sync | | ✓ | ✓ | +| Hosted Analytics | | ✓ | ✓ | +| Hosted Auto Consolidation + retention policy | | ✓ | ✓ | +| Hosted Auto Dreaming + managed proposals | | ✓ | ✓ | +| Priority support | | ✓ | ✓ | +| Hosted multi-user dashboard: invitations, logins, roles, seat management | | | ✓ | +| Hosted Team audit log + CSV export | | | ✓ | +| 72-hour pending invitations (resend/revoke) | | | ✓ | +| Scoped, expiring per-user agent and sync tokens | | | ✓ | + +--- + +## MCP tools + +Engraphis exposes a zero-configuration Smart MCP gateway plus a 35-tool Classic compatibility +server across memory, recall, code graphs, governance, sessions, and privacy-safe audit receipts. +The focused [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md) is the source for +the full inventory and parameters. + +--- + +## Graphs and privacy-safe receipts + +Memory, entity, and code relationships live in one local graph. Engraphis also provides +content-free operation receipts for inspectable audit evidence. See the +[architecture](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/ARCHITECTURE_V3.md), [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md), and +[security policy](https://github.com/Coding-Dev-Tools/engraphis/blob/main/SECURITY.md) for the data model, tools, and guarantees. + +--- + +## Cloud sync + +Cloud Sync is an optional hosted Pro/Team service. The public package includes the customer client +and deterministic merge implementation; hosted relay and account operations are separate. See +[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for setup, encryption, merge behavior, and the local folder exchange. + +The public package ships the same sync client as a console script and CLI verb: +`engraphis-sync` (installed entry point), `engraphis sync ...`, and +`python -m scripts.sync --status` for local-only state without network activity. See +[Cloud Sync](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SYNC.md) for +flags, encryption, merge behavior, and the local folder exchange. + +--- + +## Security and trust boundaries + +Engraphis is local-first and binds to loopback by default. Read the +[security policy](https://github.com/Coding-Dev-Tools/engraphis/blob/main/SECURITY.md) before remote deployment or integrating external resources; it +covers supported versions, data protections, threat model, and vulnerability reporting. + +--- + +## Encryption at rest + +Set `ENGRAPHIS_DB_KEY` (or `ENGRAPHIS_DB_KEY_FILE`) and install the extra: + +```bash +pip install "engraphis[encryption]" +``` + +The entire main memory database file is transparently encrypted with AES-256 via SQLCipher; +full-text search, the graph, and every query keep working unchanged. Customer authentication +and managed-service state use their respective deployment protections. When a key is set for the +main database, Engraphis **fails closed with an error** rather than silently falling back to +plaintext. Generate a strong key: + +```bash +python -c "import secrets; print(secrets.token_hex(32))" +``` + +When using `ENGRAPHIS_DB_KEY_FILE`, provision a regular secret file readable only by the +service identity. Engraphis rejects links, reparse points, hard links, malformed text, and +oversized key files rather than following an unexpected filesystem object. + +> An existing plaintext database cannot be opened with a key: migrate it (dump → import +> into a fresh keyed DB). See `.env.example` for all encryption options. + +--- + +## Import files and folders + +The dependency-free universal core scans Markdown, plain text, RST, HTML, JSON/JSONL, CSV/TSV, +configuration/XML text, source code, RTF, DOCX/ODT, XLSX/ODS, PPTX/ODP, and EPUB into the normal +v2 memory path. Installed local resource adapters add PDF text, image OCR, and explicitly +local-model audio/video transcription. +Start with a zero-write +preview, then confirm the same source collection explicitly: + +```bash +engraphis import documents /path/to/collection --workspace acme --dry-run +engraphis import documents /path/to/collection --workspace acme --repo product --yes +``` + +The CLI never downloads an embedding model during import. Use a model that is already cached, +set `ENGRAPHIS_EMBED_MODEL=local:/absolute/model/path`, or explicitly set +`ENGRAPHIS_EMBED_MODEL` to an empty value to use dependency-free deterministic hashing in +lexical degraded mode. + +The dashboard’s **Import local documents** flow offers the same preview, target scope, source +label, conflict policy, cancellation, and resumable progress. Re-imports are idempotent, +preserve temporal history, and report source removals without hard-deleting memories. Obsidian +remains the rich Markdown adapter for frontmatter, aliases, wikilinks, and attachment references: + +```bash +engraphis import obsidian /path/to/vault --workspace acme --dry-run +``` + +See the [document import guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/DOCUMENT_IMPORT.md) +for supported formats, source safety, resume and conflict behavior, optional adapters, and +limitations; see the [Obsidian adapter guide](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/OBSIDIAN_IMPORT.md) +for Markdown-specific behavior. + +--- + +## Consolidation and automation + +Manual consolidation is free, local, and dry-run by default; use the dashboard, SDK, CLI, or +MCP. Hosted Pro and Team automation is optional managed compute that produces reviewable +proposals rather than silently changing local data. See [hosted plans](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/HOSTED_PLANS.md), +[licensing](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md), and the [MCP tool reference](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/MCP_TOOLS.md) for scope and use. + +--- + +## Configuration + +Values come from the process environment. Engraphis also loads the owner-private +`~/.engraphis/config.env`; `ENGRAPHIS_ENV_FILE` can select another absolute owner-private regular +file. It never searches the working directory for `.env`, and explicit process variables win. + +| Env Var | Default | Description | +|---------|---------|-------------| +| `ENGRAPHIS_ENV_FILE` | `~/.engraphis/config.env` | Optional trusted config leaf selected before trusted values load. Its bounded dependency-free parser performs no interpolation. An explicit value must be an absolute path to an owner-private regular file; arbitrary working-directory `.env` files are ignored. | +| `ENGRAPHIS_DB_PATH` | Source: `/engraphis.db`; installed: platform user-data directory | SQLite database file. Installed defaults are `%LOCALAPPDATA%\engraphis\engraphis.db` (Windows), `~/Library/Application Support/engraphis/engraphis.db` (macOS), and `$XDG_DATA_HOME/engraphis/engraphis.db` or `~/.local/share/engraphis/engraphis.db` (Linux). The environment variable overrides every default; a relative value is resolved from the trusted `~/.engraphis/config.env` directory so launch CWD cannot select a different workspace database. | +| `ENGRAPHIS_SQLITE_DURABILITY` | `durable` | Writable file databases use WAL and FULL commit synchronization. Explicit `balanced` selects NORMAL, which can lose recent acknowledged writes after OS/power failure. Effective settings appear in diagnostics; see [SQLite durability](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/SQLITE_DURABILITY.md). | +| `ENGRAPHIS_HOST` | `127.0.0.1` | Server bind address | +| `ENGRAPHIS_PORT` | `8700` | Dashboard port. A platform-injected `$PORT` (Railway/Fly/Heroku) takes precedence over this value for the dashboard bind; Compose pins both to `ENGRAPHIS_COMPOSE_PORT` so the mapping stays in sync | +| `ENGRAPHIS_SERVICE_MODE` | `customer` | The public package supports only `customer`; hosted vendor, relay, compute, and worker roles are not distributed here | +| `ENGRAPHIS_API_TOKEN` | Not set | Optional bearer credential for this single-user local customer node; never reuse a hosted credential | +| `ENGRAPHIS_CORS_ORIGINS` | loopback on `ENGRAPHIS_PORT` | Comma-separated REST CORS allow-list; defaults to `127.0.0.1` and `localhost` on the configured port | +| `ENGRAPHIS_INDEX_ROOTS` | Working, home, and temporary directories | Optional path-separator-delimited absolute-path allow-list that replaces the default roots accepted by local code indexing | +| `ENGRAPHIS_HTTP_INDEX_ROOT` | First `ENGRAPHIS_INDEX_ROOTS` entry, or current directory | Single root for dashboard and REST `POST /api/code/index`; submitted paths resolve beneath it. An explicit root (or fallback entry) must be absolute; an explicit HTTP root is included in the engine-approved set. MCP and CLI indexing continue to use `ENGRAPHIS_INDEX_ROOTS`. | +| `ENGRAPHIS_DB_KEY` | Not set | Encrypt the database at rest (SQLCipher). Or use `ENGRAPHIS_DB_KEY_FILE` | +| `ENGRAPHIS_EMBED_MODEL` | `sentence-transformers/all-MiniLM-L6-v2` | sentence-transformers model | +| `ENGRAPHIS_EMBED_REVISION` | Not set | Optional immutable lowercase 40-hex Hugging Face commit for the embedding model. Loaded Hub commits or local artifact manifests identify persistent vector spaces; unresolved mutable identities keep vector recall fail-closed. | +| `ENGRAPHIS_RERANK_MODEL` | Not set | Optional sentence-transformers cross-encoder reranker | +| `ENGRAPHIS_RERANK_REVISION` | Not set | Optional immutable lowercase 40-hex Hugging Face commit for the reranker | +| `ENGRAPHIS_REQUIRE_IMMUTABLE_MODELS` | `false` | When enabled, require a 40-hex commit before loading remote embedding models, rerankers, or chunk tokenizers; `local:` selectors and filesystem paths remain permitted | +| `ENGRAPHIS_REQUIRE_EXACT_BACKENDS` | `false` | When enabled, dashboard and standalone MCP startup fails if a configured optional backend is unavailable instead of silently falling back | +| `ENGRAPHIS_EXTRACTOR` | `none` | `none` = verbatim; `chunk` = offline structure-aware chunks; `llm` = free-form LLM facts; `llm_structured` = schema-validated facts + graph metadata | +| `ENGRAPHIS_CHUNK_TOKENIZER_MODEL` | Not set | Optional Hugging Face tokenizer used to enforce chunk budgets with the downstream reader's real tokenization; requires the optional `transformers` package | +| `ENGRAPHIS_CHUNK_TOKENIZER_REVISION` | Not set | Optional immutable tokenizer/model revision recorded in the chunk-counter identity; pin this for reproducible benchmark artifacts | +| `ENGRAPHIS_GRAPH_EXTRACTOR` | `regex` | `regex` = offline heuristic NER; `none` = disable heuristic text extraction (validated `llm_structured` metadata still feeds the graph) | +| `ENGRAPHIS_RETENTION_SUPERVISOR` | `none` | `none` = deterministic only; `llm` = sends a bounded excerpt to the configured provider for advisory ephemeral/normal/critical classification | +| `ENGRAPHIS_ALLOW_AUTOMATIC_CRITICAL_RETENTION` | `false` | Opt in only when an LLM supervisor may automatically assign the long-lived `critical` class; explicit user-selected critical retention is unaffected | +| `ENGRAPHIS_WHISPER_MODEL` | Not set | Enables local faster-whisper audio/video transcription | +| `ENGRAPHIS_POSTGRES_DSN` | Not set | CLI-only PostgreSQL source; used for the connection and never stored | +| `ENGRAPHIS_POSTGRES_CONNECT_TIMEOUT` | `10` | PostgreSQL introspection connection timeout in seconds (bounded to 1--120) | +| `ENGRAPHIS_POSTGRES_STATEMENT_TIMEOUT_MS` | `30000` | Per-introspection PostgreSQL statement timeout in milliseconds (bounded to 1--300000) | +| `ENGRAPHIS_GRAPH_TOKEN` | Not set | Bearer token for `engraphis-graph-server`; required off-loopback | +| `ENGRAPHIS_GRAPH_HOST` / `ENGRAPHIS_GRAPH_PORT` | `127.0.0.1` / `8720` | Read-only graph/recall server bind address | +| `ENGRAPHIS_LLM_PROVIDER` | `openai` | `openai \| anthropic \| google \| openrouter \| custom` | +| `ENGRAPHIS_LLM_MODEL` | `gpt-4o-mini` | Model name (provider-specific) | +| `ENGRAPHIS_LLM_API_KEY` | Not set | API key for chat/synthesis, `llm` / `llm_structured` extraction, and structured consolidation | +| `ENGRAPHIS_LLM_BASE_URL` | Not set | Base URL for openrouter / custom OpenAI-compatible endpoints | +| `ENGRAPHIS_LLM_AUTO_EXTRACT` | `0` | Opt in to switching the running engine to `llm_structured` after a successful live connection test; the dashboard's extraction Off button persists `0`, and its On button restores `1` | +| `ENGRAPHIS_FORWARDED_ALLOW_IPS` | *(none)* | Proxies trusted for forwarded client/TLS headers (`*` only when the service is reachable exclusively through that proxy) | +| `ENGRAPHIS_LOCAL_TRUSTED_PEERS` | *(none)* | Exact peers/CIDRs treated as local without forwarding headers; use only for trusted Docker/LAN peers, never public deployments | +| `ENGRAPHIS_UPDATE_CACHE` | `86400` | Update-check cache TTL in seconds, bounded to `1..31622400`; this is never a cache-file path | +| `ENGRAPHIS_UPDATE_CHECK` | Off | Opt-in release reminder surfaced in the dashboard, server startup log, and MCP. Update checks run only when this is set to an affirmative value; `0` keeps them off. | +| `ENGRAPHIS_UPDATE_URL` | Not set | Overrides the release-check source URL; the outbound client accepts HTTPS and rejects private/reserved destinations. | +| `ENGRAPHIS_CLOUD_CONTROL_URL` | hosted default | Official entitlement, organization, and credential control API. A saved rotating credential stays bound to the control endpoint recorded for its family; reconnect to change it. | +| `ENGRAPHIS_CLOUD_COMPUTE_URL` | hosted default | Official Analytics and managed-automation API. A saved rotating credential stays bound to its recorded compute endpoint; reconnect to change it. | +| `ENGRAPHIS_CLOUD_ORGANIZATION_ID` | Not set | Hosted organization bound to this customer session | +| `ENGRAPHIS_CLOUD_REFRESH_CREDENTIAL` | Not set | Bootstrap-only rotating hosted credential; after first use the owner-only cloud session replacement takes precedence | +| `ENGRAPHIS_CLOUD_TOKEN_SUBJECT` | `member` | Subject fixed during hosted bootstrap (`device` or `member`); set explicitly with an environment-only refresh credential | +| `ENGRAPHIS_CLOUD_ACCESS_TOKEN` | Not set | Optional short-lived access token for ephemeral jobs | +| `ENGRAPHIS_MANAGED_COMPUTE_CONSENT` | *(unset)* | Deny-only operator override: `0` pauses readable managed processing. A truthy value cannot grant approval. Each workspace requires explicit confirmation in Manage → Settings; encrypted sync is separate | + +The optional cross-encoder reranker is model- and hardware-dependent. Treat its quality and +latency as deployment-specific until a versioned model identity, exact configuration, and +reproducible evaluation artifact are available for the comparison being reported. + +See `.env.example` for the full variable inventory. Supply those values through the process +environment or the trusted config file above; copying it to an arbitrary `./.env` does not make +Engraphis load it. + +> **Ablation fixture:** `python -m eval.ablation` is an offline deterministic check that prints +> `recall@5` comparisons for vector-only and hybrid retrieval, multi-hop graph arms, and +> retrieval policies, plus ordinary-recall age and semantic-confidence checks. It does not +> produce MRR, hit@5, or ms/query results. Use `python -m eval.reinforcement` for retention +> trajectories, and register evidence before quoting any benchmark results. + +--- + +## Project structure + +``` +engraphis/ +├── engraphis/ +│ ├── core/ # v2 engine: interfaces, store, recall, scoring, schema, sync +│ ├── backends/ # pluggable embedder / vector index / reranker / codegraph / sync transports / encryption +│ ├── factory.py # outer v2 composition root; selects and injects concrete backends +│ ├── service.py # validated MemoryService facade +│ ├── mcp_server.py # Smart MCP gateway + 35-tool Classic compatibility server +│ ├── dashboard_app.py # dashboard WebUI (FastAPI) +│ ├── dashboard_assets/ # primary Ledger interface + graph engine +│ ├── classic_assets/ # selectable full operator dashboard backup +│ ├── read_only_api.py # token-protected recall/repository-graph HTTP surface +│ ├── hosted_client.py # hosted URLs, plan labels, and endpoint validation only +│ ├── licensing.py # compatibility facade for hosted presentation metadata +│ ├── cloud_session.py # rotating hosted customer-session client +│ ├── cloud_features.py # consented managed-feature protocol client +│ ├── config.py / app.py # env settings / REST server +│ └── static/ # compatibility dashboard asset paths +├── eval/ # offline retrieval eval harness + datasets +├── tests/ # offline-first pytest suite and release/security contracts +├── scripts/ # dashboard, server, graph, CLI, connect, update, consolidation, sync +├── docs/ # product, API, hosting, sync, and provider guides +├── Dockerfile / docker-compose.yml +└── pyproject.toml +``` + +New capability belongs in the v2 path (`engraphis/core/`, `engraphis/backends/`, and +`MemoryService`) behind the interfaces in `core/interfaces.py`. Algorithm modules in `core/` +remain backend-agnostic; `engraphis/factory.py` is the outer composition root used by +`engraphis.create_memory_engine()` and the compatibility `MemoryEngine.create()` entry point, then +injects the selected collaborators into `core/engine.py`. The flat-namespace v1 server under +`engraphis/app.py`, `routes/`, `stores/`, and `engines/` remains a +compatibility/reference surface; `engraphis-dashboard`, the MCP server, and the Python quickstart +above use v2. + +--- + +## License + +Apache-2.0. See [LICENSE](https://github.com/Coding-Dev-Tools/engraphis/blob/main/LICENSE) and [NOTICE](https://github.com/Coding-Dev-Tools/engraphis/blob/main/NOTICE). "Engraphis" is a trademark of the +Engraphis project; the license does not grant trademark rights. Code already distributed +under Apache-2.0 keeps that grant; later releases cannot retroactively withdraw it. The +official hosted control plane, its production credentials and records, managed operations, +support, and future separately delivered commercial modules are outside the public source +grant. See [`docs/LICENSING.md`](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/LICENSING.md) for the complete boundary. + +### Reliability implementation candidate + +The current source uses schema 18 for durable, content-free vector-index repair and +atomic memory-command receipts. Upgrades use the existing verified-backup migration path. +The [rework execution register](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/REWORK_EXECUTION.md) records the current findings, +compatibility decisions, acceptance evidence, remaining work and recovery procedure. +See [the reliability program](https://github.com/Coding-Dev-Tools/engraphis/blob/main/docs/RELIABILITY_PROGRAM.md) for exact implementation, +validation, migration and release boundaries. Managed processing now requires explicit +workspace approval in Settings. Existing installations start with readable +uploads paused until confirmed; connecting an account does not grant approval. + +For setup diagnostics use `engraphis-init --check --json`. New configurations get an +owner-private local API token. Existing configs are preserved. Record selected install +capabilities with `engraphis-init --extras server,mcp` or `--extras none`; future updates +preserve that choice. `ENGRAPHIS_UPDATE_EXTRAS` remains an explicit override. diff --git a/tests/test_benchmark_evidence.py b/tests/test_benchmark_evidence.py index 9650db24..4f3f3d0c 100644 --- a/tests/test_benchmark_evidence.py +++ b/tests/test_benchmark_evidence.py @@ -1,1282 +1,1282 @@ -import hashlib -import json -import re -import struct -from copy import deepcopy -from pathlib import Path -from xml.etree import ElementTree - -import pytest - -from eval import metrics -from eval import grounded as grounded_eval -from eval.benchmark import ( - SCHEMA, - CANONICAL_TOKEN_BUDGETS, - LONGMEMEVAL_V2_CANONICAL_PROFILE_TEMPLATE, - canonical_benchmark_config, - count_tokens, - fixed_budget_curve, - paired_bootstrap_ci, - redact_command, - redact_public_record, - main, - question_record, - report_envelope, - stratified_bootstrap_ci, - validate_report, - write_canonical_artifact, -) -from eval.chunking_eval import compare as compare_chunking, load as load_chunking -from eval.harness import load_dataset as load_performance_dataset -from eval.performance import run as run_performance - - +import hashlib +import json +import re +import struct +from copy import deepcopy +from pathlib import Path +from xml.etree import ElementTree + +import pytest + +from eval import metrics +from eval import grounded as grounded_eval +from eval.benchmark import ( + SCHEMA, + CANONICAL_TOKEN_BUDGETS, + LONGMEMEVAL_V2_CANONICAL_PROFILE_TEMPLATE, + canonical_benchmark_config, + count_tokens, + fixed_budget_curve, + paired_bootstrap_ci, + redact_command, + redact_public_record, + main, + question_record, + report_envelope, + stratified_bootstrap_ci, + validate_report, + write_canonical_artifact, +) +from eval.chunking_eval import compare as compare_chunking, load as load_chunking +from eval.harness import load_dataset as load_performance_dataset +from eval.performance import run as run_performance + + ROOT = Path(__file__).resolve().parents[1] PUBLIC_OFFLINE_ARTIFACT = "offline-fixtures-v76.json" PUBLIC_OFFLINE_SHA = "2fb5ce5b2cbc21541f7ae9cad5d2ef615014c0e86f9881f00621d00986422ba8" - - -@pytest.fixture(scope="module") -def offline_release_evidence(): - """Run the exact small offline commands that back the public documentation.""" - longdoc = ROOT / "eval" / "datasets" / "longdoc.jsonl" - codemem = ROOT / "eval" / "datasets" / "codemem.jsonl" - return { - "chunking": compare_chunking( - load_chunking(str(longdoc)), k=5, embed_model=None - ), - "performance": run_performance( - load_performance_dataset(str(codemem)), k=5, iterations=10 - ), - "grounded": grounded_eval.run(), - } - - -def test_public_facing_docs_do_not_use_em_dashes(): - """Published prose uses straightforward punctuation that renders consistently.""" - public_files = [ - *(ROOT / name for name in ("README.md", "BENCHMARKS.md", "CHANGELOG.md", "SECURITY.md")), - *(ROOT / "docs").rglob("*.md"), - *(ROOT / "docs" / "images").glob("*.svg"), - *(ROOT / "skills" / "engraphis-memory").rglob("*.md"), - ] - offenders = [ - path.relative_to(ROOT).as_posix() - for path in public_files - if "—" in path.read_text(encoding="utf-8") - ] - - assert not offenders, f"Public-facing files still contain em dashes: {offenders}" - - -class CharacterTokenizer: - def encode(self, text): - return list(text) - - -def test_public_record_redaction_omits_raw_payloads_and_content_fingerprints(): - record = redact_public_record({ - "question_id": "q1", - "query": "private query", - "answer_variants": ["private answer"], - "model_output": "private completion", - "context": "private context", - "retrieved_context": "private retrieved context", - "prompt": "private prompt", - "input": "private input", - "conversation": ["private conversation"], - "history": ["private history"], - "tool_calls": [{"arguments": "private tool input"}], - }) - - assert record == {"question_id": "q1"} - - - -def _committed_evidence() -> dict: - """Load the COMMITTED registry artifact — the publication source of truth that the - README/BENCHMARKS/SVG prose was written from. - - Prose tests must interpolate values from this artifact, not from a fresh evaluator - run. Timed latency aggregates are machine-dependent, while the context, payload, - question-count, and quality aggregates used by the publication contract are - deterministic and compared exactly in ``test_public_numeric_evidence_registry_is_complete_and_live``. - """ - artifact = json.loads( + + +@pytest.fixture(scope="module") +def offline_release_evidence(): + """Run the exact small offline commands that back the public documentation.""" + longdoc = ROOT / "eval" / "datasets" / "longdoc.jsonl" + codemem = ROOT / "eval" / "datasets" / "codemem.jsonl" + return { + "chunking": compare_chunking( + load_chunking(str(longdoc)), k=5, embed_model=None + ), + "performance": run_performance( + load_performance_dataset(str(codemem)), k=5, iterations=10 + ), + "grounded": grounded_eval.run(), + } + + +def test_public_facing_docs_do_not_use_em_dashes(): + """Published prose uses straightforward punctuation that renders consistently.""" + public_files = [ + *(ROOT / name for name in ("README.md", "BENCHMARKS.md", "CHANGELOG.md", "SECURITY.md")), + *(ROOT / "docs").rglob("*.md"), + *(ROOT / "docs" / "images").glob("*.svg"), + *(ROOT / "skills" / "engraphis-memory").rglob("*.md"), + ] + offenders = [ + path.relative_to(ROOT).as_posix() + for path in public_files + if "—" in path.read_text(encoding="utf-8") + ] + + assert not offenders, f"Public-facing files still contain em dashes: {offenders}" + + +class CharacterTokenizer: + def encode(self, text): + return list(text) + + +def test_public_record_redaction_omits_raw_payloads_and_content_fingerprints(): + record = redact_public_record({ + "question_id": "q1", + "query": "private query", + "answer_variants": ["private answer"], + "model_output": "private completion", + "context": "private context", + "retrieved_context": "private retrieved context", + "prompt": "private prompt", + "input": "private input", + "conversation": ["private conversation"], + "history": ["private history"], + "tool_calls": [{"arguments": "private tool input"}], + }) + + assert record == {"question_id": "q1"} + + + +def _committed_evidence() -> dict: + """Load the COMMITTED registry artifact — the publication source of truth that the + README/BENCHMARKS/SVG prose was written from. + + Prose tests must interpolate values from this artifact, not from a fresh evaluator + run. Timed latency aggregates are machine-dependent, while the context, payload, + question-count, and quality aggregates used by the publication contract are + deterministic and compared exactly in ``test_public_numeric_evidence_registry_is_complete_and_live``. + """ + artifact = json.loads( (ROOT / "docs" / "benchmark-evidence" / PUBLIC_OFFLINE_ARTIFACT).read_text( - encoding="utf-8" - ) - ) - return { - "chunking": artifact["runs"][0]["result"], - "performance": artifact["runs"][1]["result"], - "grounded": artifact["runs"][2]["result"], - } - -def test_readme_distinguishes_every_registered_token_context_measurement(): - """Public token-efficiency copy preserves each registered metric boundary. - - Values are interpolated from the COMMITTED registry artifact — the publication - source of truth — so prose cannot drift from the evidence it cites. - """ - committed = _committed_evidence() - readme = (ROOT / "README.md").read_text(encoding="utf-8") - chunking = committed["chunking"] - whole = chunking["whole"] - chunked = chunking["chunked"] - performance = committed["performance"] - context_full = performance["full_serialized_payload_tokens"] - context_compact = performance["compact_serialized_payload_tokens"] - payload_samples = performance["questions"] - timed_recalls = performance["timed_recalls"] - - for evidence in ( - "## Measured token and context savings", - "See benchmark details and reproduce the results", - "### Measurement details and reproducibility", - f"{whole['mean_context_tokens']:.1f}** tokens → structure-aware chunks: " - f"**{chunked['mean_context_tokens']:.1f}** tokens", - f"{chunking['context_reduction_pct']:.1f}% lower", - f"{whole['mean_evidence_tokens']:.1f}** tokens → chunks: " - f"**{chunked['mean_evidence_tokens']:.1f}** tokens", - "73.9% lower", - f"{context_full:,}** `engraphis.regex.v1` tokens → " - f"compact proxy: **{context_compact:,}** tokens", - f"{performance['saved_serialized_payload_tokens']:,} proxy tokens avoided", - f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower", - f"{payload_samples} payload samples; {timed_recalls} timed recalls", - f"1,500** tokens; observed mean: **{performance['mean_context_tokens']:.2f}**; " - f"observed maximum: **{performance['max_context_tokens']}**", - "does **not** serialize the MCP envelope", - "not an MCP transport response", - "must not be added together", - "not a storage-reduction claim", + encoding="utf-8" + ) + ) + return { + "chunking": artifact["runs"][0]["result"], + "performance": artifact["runs"][1]["result"], + "grounded": artifact["runs"][2]["result"], + } + +def test_readme_distinguishes_every_registered_token_context_measurement(): + """Public token-efficiency copy preserves each registered metric boundary. + + Values are interpolated from the COMMITTED registry artifact — the publication + source of truth — so prose cannot drift from the evidence it cites. + """ + committed = _committed_evidence() + readme = (ROOT / "README.md").read_text(encoding="utf-8") + chunking = committed["chunking"] + whole = chunking["whole"] + chunked = chunking["chunked"] + performance = committed["performance"] + context_full = performance["full_serialized_payload_tokens"] + context_compact = performance["compact_serialized_payload_tokens"] + payload_samples = performance["questions"] + timed_recalls = performance["timed_recalls"] + + for evidence in ( + "## Measured token and context savings", + "See benchmark details and reproduce the results", + "### Measurement details and reproducibility", + f"{whole['mean_context_tokens']:.1f}** tokens → structure-aware chunks: " + f"**{chunked['mean_context_tokens']:.1f}** tokens", + f"{chunking['context_reduction_pct']:.1f}% lower", + f"{whole['mean_evidence_tokens']:.1f}** tokens → chunks: " + f"**{chunked['mean_evidence_tokens']:.1f}** tokens", + "73.9% lower", + f"{context_full:,}** `engraphis.regex.v1` tokens → " + f"compact proxy: **{context_compact:,}** tokens", + f"{performance['saved_serialized_payload_tokens']:,} proxy tokens avoided", + f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower", + f"{payload_samples} payload samples; {timed_recalls} timed recalls", + f"1,500** tokens; observed mean: **{performance['mean_context_tokens']:.2f}**; " + f"observed maximum: **{performance['max_context_tokens']}**", + "does **not** serialize the MCP envelope", + "not an MCP transport response", + "must not be added together", + "not a storage-reduction claim", PUBLIC_OFFLINE_ARTIFACT, - "offline-chunking", - "offline-performance", + "offline-chunking", + "offline-performance", PUBLIC_OFFLINE_SHA, - "There is no universal memory-count", - "python -m eval.vector_scale", - 'vector_backend="sqlite-vec"', - ): - assert evidence in readme - - for unsupported in ( - "49,915,394", - "891,857", - "98.2133%", - "Repeated-memory consolidation fixture", - "1,883** total agent-facing tokens", - "3.1% higher", - ): - assert unsupported not in readme - - - -def test_public_docs_scope_external_numbers_and_withhold_historical_claims(): - """The offline registry is scoped while separate diagnostics remain artifact-bound.""" - readme = (ROOT / "README.md").read_text(encoding="utf-8") - benchmarks = (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") - expansion = (ROOT / "docs" / "BENCHMARK_EXPANSION_RESULTS.md").read_text( - encoding="utf-8" - ) - additional = (ROOT / "docs" / "ADDITIONAL_BENCHMARK_DIAGNOSTICS.md").read_text( - encoding="utf-8" - ) - security = (ROOT / "SECURITY.md").read_text(encoding="utf-8") - readme_normalized = " ".join(readme.split()) - benchmarks_normalized = " ".join(benchmarks.split()) - additional_normalized = " ".join(additional.split()) - - assert "See benchmark details and reproduce the results" in readme - assert "offline fixture registry intentionally excludes external" in readme_normalized - assert "Completed retrieval-only diagnostics are published separately" in readme_normalized - assert "absence from this registry" in benchmarks_normalized - assert "LoCoMo and LongMemEval retrieval diagnostics are retained as separate public-safe artifacts" in benchmarks_normalized - assert "Mem2ActBench's declared small retrieval diagnostic is complete" in benchmarks_normalized - assert "LongMemEval 4,096-token context experiment | COMPLETE" in expansion - assert "+24.03 percentage points" in expansion - assert "source-preparation metadata" in expansion - assert "public source lock records 20 preparation exclusions" in additional_normalized - assert "Exact vector scale envelope" in benchmarks - assert "python -m eval.redteam_poisoning" in security - - for stale in ( - "model-dependent, consolidation, productivity, and latency results remain unpublished", - "private diagnostic; it is not an official benchmark-harness or public evidence artifact", - "withholds their case counts, retrieval scores", - ): - assert stale not in readme - assert stale not in benchmarks - - for unsupported in ( - "49,915,394", - "891,857", - "98.2133%", - "0.6045", - "0.6625", - "0.1259", - "0.5100", - "20.666 ms", - ): - assert unsupported not in readme - assert unsupported not in benchmarks - - for supporting_detail in ( - "### Choose a vector backend for your corpus", - "python -m eval.redteam_poisoning", - "[local and hosted plans]", - ): - assert supporting_detail not in readme - - -def test_readme_makes_agent_benefits_and_visual_evidence_scannable(): - """The public overview and its visual evidence must stay wired to real assets.""" - readme = (ROOT / "README.md").read_text(encoding="utf-8") - - for evidence in ( - "## What Engraphis gives an agent", - "Remember a project across sessions", - "Avoid confident guesses", - "Avoid dragging the whole project into every prompt", - "docs/images/knowledge-graph.png", - "docs/images/context-efficiency.svg", - "Less repeated history means more room for the task, tools, and useful evidence", - ): - assert evidence in readme - - for removed in ( - "### See the behavior in reproducible fixtures", - "docs/images/evidence-backed-agent-examples.svg", - "Run `python -m eval.chunking_eval` and `python -m eval.grounded`", - ): - assert removed not in readme - - for filename in ( - "engraphis-benefit-flow.svg", - "engraphis-benefit-flow.png", - "context-efficiency.svg", - "context-efficiency.png", - "evidence-backed-agent-examples.svg", - "evidence-backed-agent-examples.png", - ): - assert (ROOT / "docs" / "images" / filename).is_file() - - -def test_readme_visual_pngs_match_their_svg_canvas(): - """README image exports must not carry hidden screenshot padding.""" - image_dir = ROOT / "docs" / "images" - - for stem in ( - "engraphis-benefit-flow", - "evidence-backed-agent-examples", - "context-efficiency", - ): - svg = ElementTree.parse(image_dir / f"{stem}.svg").getroot() - expected = (int(svg.attrib["width"]), int(svg.attrib["height"])) - png_header = (image_dir / f"{stem}.png").read_bytes()[:24] - - assert png_header[:8] == b"\x89PNG\r\n\x1a\n" - assert struct.unpack(">II", png_header[16:24]) == expected - - -def test_example_visual_uses_the_checked_in_offline_fixture_results( - offline_release_evidence, -): - """The examples stay tied to executable fixtures and their public artifact.""" - chunking = offline_release_evidence["chunking"] - whole = chunking["reports"]["whole"] - chunked = chunking["reports"]["chunked"] - grounded = offline_release_evidence["grounded"] - visual = ( - ROOT / "docs" / "images" / "evidence-backed-agent-examples.svg" - ).read_text(encoding="utf-8") - - assert chunking["context_reduction_pct"] == 71.1 - result = ( - f"{whole['mean_context_tokens']:.1f} → " - f"{chunked['mean_context_tokens']:.1f} tokens" - ) - assert result in visual - assert grounded == { - "answer_rate": 1.0, - "abstain_rate": 1.0, - "accuracy": 1.0, - "grounded_hits": 5, - "abstain_hits": 6, - "quarantine_hits": 1, - "n_quarantine": 1, - "n_answerable": 5, - "n_unanswerable": 6, - } - assert "5/5 answerable questions" in visual - assert "6/6 off-topic questions" in visual + "There is no universal memory-count", + "python -m eval.vector_scale", + 'vector_backend="sqlite-vec"', + ): + assert evidence in readme + + for unsupported in ( + "49,915,394", + "891,857", + "98.2133%", + "Repeated-memory consolidation fixture", + "1,883** total agent-facing tokens", + "3.1% higher", + ): + assert unsupported not in readme + + + +def test_public_docs_scope_external_numbers_and_withhold_historical_claims(): + """The offline registry is scoped while separate diagnostics remain artifact-bound.""" + readme = (ROOT / "README.md").read_text(encoding="utf-8") + benchmarks = (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") + expansion = (ROOT / "docs" / "BENCHMARK_EXPANSION_RESULTS.md").read_text( + encoding="utf-8" + ) + additional = (ROOT / "docs" / "ADDITIONAL_BENCHMARK_DIAGNOSTICS.md").read_text( + encoding="utf-8" + ) + security = (ROOT / "SECURITY.md").read_text(encoding="utf-8") + readme_normalized = " ".join(readme.split()) + benchmarks_normalized = " ".join(benchmarks.split()) + additional_normalized = " ".join(additional.split()) + + assert "See benchmark details and reproduce the results" in readme + assert "offline fixture registry intentionally excludes external" in readme_normalized + assert "Completed retrieval-only diagnostics are published separately" in readme_normalized + assert "absence from this registry" in benchmarks_normalized + assert "LoCoMo and LongMemEval retrieval diagnostics are retained as separate public-safe artifacts" in benchmarks_normalized + assert "Mem2ActBench's declared small retrieval diagnostic is complete" in benchmarks_normalized + assert "LongMemEval 4,096-token context experiment | COMPLETE" in expansion + assert "+24.03 percentage points" in expansion + assert "source-preparation metadata" in expansion + assert "public source lock records 20 preparation exclusions" in additional_normalized + assert "Exact vector scale envelope" in benchmarks + assert "python -m eval.redteam_poisoning" in security + + for stale in ( + "model-dependent, consolidation, productivity, and latency results remain unpublished", + "private diagnostic; it is not an official benchmark-harness or public evidence artifact", + "withholds their case counts, retrieval scores", + ): + assert stale not in readme + assert stale not in benchmarks + + for unsupported in ( + "49,915,394", + "891,857", + "98.2133%", + "0.6045", + "0.6625", + "0.1259", + "0.5100", + "20.666 ms", + ): + assert unsupported not in readme + assert unsupported not in benchmarks + + for supporting_detail in ( + "### Choose a vector backend for your corpus", + "python -m eval.redteam_poisoning", + "[local and hosted plans]", + ): + assert supporting_detail not in readme + + +def test_readme_makes_agent_benefits_and_visual_evidence_scannable(): + """The public overview and its visual evidence must stay wired to real assets.""" + readme = (ROOT / "README.md").read_text(encoding="utf-8") + + for evidence in ( + "## What Engraphis gives an agent", + "Remember a project across sessions", + "Avoid confident guesses", + "Avoid dragging the whole project into every prompt", + "docs/images/knowledge-graph.png", + "docs/images/context-efficiency.svg", + "Less repeated history means more room for the task, tools, and useful evidence", + ): + assert evidence in readme + + for removed in ( + "### See the behavior in reproducible fixtures", + "docs/images/evidence-backed-agent-examples.svg", + "Run `python -m eval.chunking_eval` and `python -m eval.grounded`", + ): + assert removed not in readme + + for filename in ( + "engraphis-benefit-flow.svg", + "engraphis-benefit-flow.png", + "context-efficiency.svg", + "context-efficiency.png", + "evidence-backed-agent-examples.svg", + "evidence-backed-agent-examples.png", + ): + assert (ROOT / "docs" / "images" / filename).is_file() + + +def test_readme_visual_pngs_match_their_svg_canvas(): + """README image exports must not carry hidden screenshot padding.""" + image_dir = ROOT / "docs" / "images" + + for stem in ( + "engraphis-benefit-flow", + "evidence-backed-agent-examples", + "context-efficiency", + ): + svg = ElementTree.parse(image_dir / f"{stem}.svg").getroot() + expected = (int(svg.attrib["width"]), int(svg.attrib["height"])) + png_header = (image_dir / f"{stem}.png").read_bytes()[:24] + + assert png_header[:8] == b"\x89PNG\r\n\x1a\n" + assert struct.unpack(">II", png_header[16:24]) == expected + + +def test_example_visual_uses_the_checked_in_offline_fixture_results( + offline_release_evidence, +): + """The examples stay tied to executable fixtures and their public artifact.""" + chunking = offline_release_evidence["chunking"] + whole = chunking["reports"]["whole"] + chunked = chunking["reports"]["chunked"] + grounded = offline_release_evidence["grounded"] + visual = ( + ROOT / "docs" / "images" / "evidence-backed-agent-examples.svg" + ).read_text(encoding="utf-8") + + assert chunking["context_reduction_pct"] == 71.1 + result = ( + f"{whole['mean_context_tokens']:.1f} → " + f"{chunked['mean_context_tokens']:.1f} tokens" + ) + assert result in visual + assert grounded == { + "answer_rate": 1.0, + "abstain_rate": 1.0, + "accuracy": 1.0, + "grounded_hits": 5, + "abstain_hits": 6, + "quarantine_hits": 1, + "n_quarantine": 1, + "n_answerable": 5, + "n_unanswerable": 6, + } + assert "5/5 answerable questions" in visual + assert "6/6 off-topic questions" in visual assert PUBLIC_OFFLINE_SHA in visual - - -def test_context_savings_visual_uses_only_registered_measurements(): - """The headline chart contains only registered values and explicit scope labels. - - Values are interpolated from the COMMITTED registry artifact — the publication - source of truth — so chart text cannot drift from the evidence it cites. - """ - visual = (ROOT / "docs" / "images" / "context-efficiency.svg").read_text( - encoding="utf-8" - ) - committed = _committed_evidence() - chunking = committed["chunking"] - whole = chunking["whole"] - chunked = chunking["chunked"] - performance = committed["performance"] - context_full = performance["full_serialized_payload_tokens"] - context_compact = performance["compact_serialized_payload_tokens"] - payload_samples = performance["questions"] - timed_recalls = performance["timed_recalls"] - - for evidence in ( - "Measured context and retrieval boundaries", - "CONTEXT BOUNDARIES", - "QUALITY SCOPES", - "PENDING EVALUATION TRACKS", - "Whole documents", - f"{whole['mean_context_tokens']:.1f} tokens", - "Structure-aware chunks", - f"{chunked['mean_context_tokens']:.1f} tokens", - f"{chunking['context_reduction_pct']:.1f}% lower", - "Smallest evidence:", - "Serialized JSON-shape payload proxy", - f"{payload_samples:,} payload samples / {timed_recalls:,} timed recalls", - "Full JSON-shape proxy", - f"{context_full:,} tokens", - "Compact JSON-shape proxy", - f"{context_compact:,} tokens", - f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower", - "Retrieved candidate quality", - "Packed context", - "Recall@5 1.000 / hit@5 1.000 / answer tokens 1.000", - "MCP transport not measured", - "JSON proxy only", - "Pinned LoCoMo and LongMemEval artifacts with answer evaluators", - ): - assert evidence in visual - - svg = ElementTree.fromstring(visual) - namespace = "{http://www.w3.org/2000/svg}" - # Numeric source text must be rendered by SVG, not hidden beside a stale bitmap. - assert not svg.findall(f".//{namespace}image") - visible_text = {node.text for node in svg.iter(f"{namespace}text")} - assert f"{context_compact:,} tokens" in visible_text - assert f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower" in visible_text - assert f"Mean {performance['mean_context_tokens']:.2f} / max {performance['max_context_tokens']:,} tokens" in visual - assert "Recall@5 1.000 / hit@5 1.000 / answer tokens 1.000" in visual - assert "MCP transport not measured" in visible_text - - for unsupported in ( - "Public evidence is checksum-bound", + + +def test_context_savings_visual_uses_only_registered_measurements(): + """The headline chart contains only registered values and explicit scope labels. + + Values are interpolated from the COMMITTED registry artifact — the publication + source of truth — so chart text cannot drift from the evidence it cites. + """ + visual = (ROOT / "docs" / "images" / "context-efficiency.svg").read_text( + encoding="utf-8" + ) + committed = _committed_evidence() + chunking = committed["chunking"] + whole = chunking["whole"] + chunked = chunking["chunked"] + performance = committed["performance"] + context_full = performance["full_serialized_payload_tokens"] + context_compact = performance["compact_serialized_payload_tokens"] + payload_samples = performance["questions"] + timed_recalls = performance["timed_recalls"] + + for evidence in ( + "Measured context and retrieval boundaries", + "CONTEXT BOUNDARIES", + "QUALITY SCOPES", + "PENDING EVALUATION TRACKS", + "Whole documents", + f"{whole['mean_context_tokens']:.1f} tokens", + "Structure-aware chunks", + f"{chunked['mean_context_tokens']:.1f} tokens", + f"{chunking['context_reduction_pct']:.1f}% lower", + "Smallest evidence:", + "Serialized JSON-shape payload proxy", + f"{payload_samples:,} payload samples / {timed_recalls:,} timed recalls", + "Full JSON-shape proxy", + f"{context_full:,} tokens", + "Compact JSON-shape proxy", + f"{context_compact:,} tokens", + f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower", + "Retrieved candidate quality", + "Packed context", + "Recall@5 1.000 / hit@5 1.000 / answer tokens 1.000", + "MCP transport not measured", + "JSON proxy only", + "Pinned LoCoMo and LongMemEval artifacts with answer evaluators", + ): + assert evidence in visual + + svg = ElementTree.fromstring(visual) + namespace = "{http://www.w3.org/2000/svg}" + # Numeric source text must be rendered by SVG, not hidden beside a stale bitmap. + assert not svg.findall(f".//{namespace}image") + visible_text = {node.text for node in svg.iter(f"{namespace}text")} + assert f"{context_compact:,} tokens" in visible_text + assert f"{100 * performance['serialized_payload_savings_ratio']:.2f}% lower" in visible_text + assert f"Mean {performance['mean_context_tokens']:.2f} / max {performance['max_context_tokens']:,} tokens" in visual + assert "Recall@5 1.000 / hit@5 1.000 / answer tokens 1.000" in visual + assert "MCP transport not measured" in visible_text + + for unsupported in ( + "Public evidence is checksum-bound", PUBLIC_OFFLINE_ARTIFACT, - "No external or model-dependent number is published without the same evidence", - "Evidence pending", - "No external or model-dependent number is published", - "808.8", - "218.4", - "17,172", - "7,663", - "Repeated memories · 230 tokens", - "47.8% less", - "53× more evidence", - "97.72% less total", - "87.7 average · 106 max", - ): - assert unsupported not in visual - - text_sizes = { - float(value) - for value in re.findall(r'font-size="([^"]+)"', visual) - } - assert {12.5, 13.2, 14.3, 17.4, 18.7, 20.0, 24.0, 33.0} <= text_sizes - - -def test_public_numeric_evidence_registry_is_complete_and_live( - offline_release_evidence, -): - """Every retained public aggregate resolves to one checksum-bound live run.""" - artifact_path = ( + "No external or model-dependent number is published without the same evidence", + "Evidence pending", + "No external or model-dependent number is published", + "808.8", + "218.4", + "17,172", + "7,663", + "Repeated memories · 230 tokens", + "47.8% less", + "53× more evidence", + "97.72% less total", + "87.7 average · 106 max", + ): + assert unsupported not in visual + + text_sizes = { + float(value) + for value in re.findall(r'font-size="([^"]+)"', visual) + } + assert {12.5, 13.2, 14.3, 17.4, 18.7, 20.0, 24.0, 33.0} <= text_sizes + + +def test_public_numeric_evidence_registry_is_complete_and_live( + offline_release_evidence, +): + """Every retained public aggregate resolves to one checksum-bound live run.""" + artifact_path = ( ROOT / "docs" / "benchmark-evidence" / PUBLIC_OFFLINE_ARTIFACT - ) - sidecar_path = artifact_path.with_suffix(".json.sha256") - artifact_bytes = artifact_path.read_bytes() - artifact_sha = hashlib.sha256(artifact_bytes).hexdigest() + ) + sidecar_path = artifact_path.with_suffix(".json.sha256") + artifact_bytes = artifact_path.read_bytes() + artifact_sha = hashlib.sha256(artifact_bytes).hexdigest() expected_sha = PUBLIC_OFFLINE_SHA - - assert artifact_sha == expected_sha - assert sidecar_path.read_text(encoding="ascii") == ( - f"{expected_sha} {artifact_path.name}\n" - ) - artifact = json.loads(artifact_bytes) - assert artifact["schema"] == "engraphis-public-offline-fixtures/v1" - assert not any(artifact["privacy"].values()) - - file_hashes = artifact["suite"]["files"] - assert file_hashes == { - path: hashlib.sha256((ROOT / path).read_bytes()).hexdigest() - for path in sorted(file_hashes) - } - suite_manifest = json.dumps( - file_hashes, sort_keys=True, separators=(",", ":") - ).encode() + + assert artifact_sha == expected_sha + assert sidecar_path.read_text(encoding="ascii") == ( + f"{expected_sha} {artifact_path.name}\n" + ) + artifact = json.loads(artifact_bytes) + assert artifact["schema"] == "engraphis-public-offline-fixtures/v1" + assert not any(artifact["privacy"].values()) + + file_hashes = artifact["suite"]["files"] + assert file_hashes == { + path: hashlib.sha256((ROOT / path).read_bytes()).hexdigest() + for path in sorted(file_hashes) + } + suite_manifest = json.dumps( + file_hashes, sort_keys=True, separators=(",", ":") + ).encode() assert hashlib.sha256(suite_manifest).hexdigest() == artifact["suite"]["digest"] assert artifact["suite"]["digest"] in (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") - - runs = {run["id"]: run for run in artifact["runs"]} - assert set(runs) == { - "offline-chunking", - "offline-performance", - "offline-grounded", - } - for run in runs.values(): - assert hashlib.sha256(run["command"].encode()).hexdigest() == run["config_digest"] - - chunking = offline_release_evidence["chunking"] - chunking_result = runs["offline-chunking"]["result"] - for mode in ("whole", "chunked"): - live = chunking["reports"][mode] - recorded = chunking_result[mode] - assert recorded["memories"] == live["memories_stored"] - assert recorded["recall_at_k"] == live["recall_at_k"] - assert recorded["mean_context_tokens"] == live["mean_context_tokens"] - assert recorded["mean_evidence_tokens"] == live["mean_evidence_tokens"] - assert recorded["max_stored_tokens"] == live["max_stored_tokens"] - assert chunking_result["context_reduction_pct"] == chunking["context_reduction_pct"] - - performance = offline_release_evidence["performance"] - performance_result = runs["offline-performance"]["result"] - assert performance_result["questions"] == performance["corpus"]["questions"] - assert performance_result["timed_recalls"] == performance["run"]["timed_recalls"] - assert performance_result["recall_at_k"] == performance["quality"]["recall_at_k"] - assert performance_result["hit_at_k"] == performance["quality"]["hit_at_k"] - assert ( - performance_result["answer_token_recall"] - == performance["quality"]["answer_token_recall"] - ) - - # These values are deterministic fixture aggregates, not wall-clock timing - # observations. Approximate comparisons would let serializer or count drift - # pass the publication contract unnoticed. - assert performance_result["mean_context_tokens"] == performance["context"]["mean_tokens"] - assert performance_result["max_context_tokens"] == performance["context"]["max_tokens"] - assert ( - performance_result["full_serialized_payload_tokens"] - == performance["context"]["full_serialized_payload_tokens"] - ) - assert ( - performance_result["compact_serialized_payload_tokens"] - == performance["context"]["compact_serialized_payload_tokens"] - ) - assert ( - performance_result["saved_serialized_payload_tokens"] - == performance["context"]["saved_serialized_payload_tokens"] - ) - assert ( - performance_result["serialized_payload_savings_ratio"] - == performance["context"]["serialized_payload_savings_ratio"] - ) - - grounded = offline_release_evidence["grounded"] - grounded_result = runs["offline-grounded"]["result"] - assert grounded_result == { - "answerable": grounded["n_answerable"], - "grounded": grounded["grounded_hits"], - "off_topic": grounded["n_unanswerable"], - "quarantined": grounded["n_quarantine"], - "abstained": grounded["abstain_hits"], - "quarantine_hits": grounded["quarantine_hits"], - "decision_accuracy": grounded["accuracy"], - } - - surfaces = ( - ROOT / "README.md", - ROOT / "BENCHMARKS.md", - ROOT / "docs" / "images" / "context-efficiency.svg", - ROOT / "docs" / "images" / "evidence-backed-agent-examples.svg", - ) - for surface in surfaces: - assert expected_sha in surface.read_text(encoding="utf-8") - - claimed_ids = set( - re.findall( - r"offline-(?:chunking|performance|grounded)", - "\n".join(path.read_text(encoding="utf-8") for path in surfaces), - ) - ) - assert claimed_ids == set(runs) - - -def test_benchmark_guide_tracks_the_live_offline_evaluators(): - """Method prose must change whenever its executable offline evidence changes. - - Values are interpolated from the COMMITTED registry artifact — the publication - source of truth — so guide text cannot drift from the evidence it cites. - """ - benchmarks = (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") - normalized = " ".join(benchmarks.split()) - committed = _committed_evidence() - chunking = committed["chunking"] - whole = chunking["whole"] - chunked = chunking["chunked"] - performance = committed["performance"] - payload_samples = performance["questions"] - - for evidence in ( - f"falls from {whole['mean_context_tokens']:.1f} to " - f"{chunked['mean_context_tokens']:.1f} tokens", - f"{whole['mean_context_tokens'] - chunked['mean_context_tokens']:.1f} fewer, " - f"{chunking['context_reduction_pct']:.1f}% lower", - f"falls from {whole['mean_evidence_tokens']:.1f} to " - f"{chunked['mean_evidence_tokens']:.1f} tokens", - "Payload proxies are sampled once per question", - "not serialized MCP envelopes or transport responses", - f"{payload_samples} payload samples total **" - f"{performance['full_serialized_payload_tokens']:,}** full-proxy", - f"versus **{performance['compact_serialized_payload_tokens']:,}** compact-proxy tokens", - f"avoiding **{performance['saved_serialized_payload_tokens']:,}** proxy tokens", - f"**{100 * performance['serialized_payload_savings_ratio']:.2f}% lower**", - f"averages **{performance['mean_context_tokens']:.2f}** tokens and reaches " - f"**{performance['max_context_tokens']}**", - ): - assert evidence in normalized - - - -def _complete_canonical_report(dataset, config): - """Minimal but fully auditable canonical envelope for validator coverage.""" - profile = config["canonical_profile"] - tokenizer_identity = ( - f"{profile['reader']['model']}@{profile['reader']['revision']}" - ) - record = question_record( - "q1", category="state", context_tokens=3, latency_ms=1.25, - retrieved_ids=["support"], supporting_ids=["support"], - recall_at_1=1.0, recall_at_5=1.0, recall_at_10=1.0, - mrr_at_1=1.0, mrr_at_5=1.0, mrr_at_10=1.0, - ndcg_at_1=1.0, ndcg_at_5=1.0, ndcg_at_10=1.0, - usage={ - "budget_tokens": config.get("token_budget") or 3, - "context_tokens": 3, - "token_counter": tokenizer_identity, - }, - ) - record["context_token_method"] = "pinned_reader_content_tokenizer" - record["context_tokenizer_identity"] = tokenizer_identity - rank_metrics = { - f"{metric}_at_{depth}": 1.0 - for metric in ("recall", "mrr", "ndcg") - for depth in (1, 5, 10) - } - curve_record = { - "question_id": "q1", - "excluded": False, - "context_tokens": 3, - "context_token_method": "pinned_reader_content_tokenizer", - "context_tokenizer_identity": tokenizer_identity, - "retrieved_ids": ["support"], - "supporting_ids": ["support"], - **rank_metrics, - } - report = report_envelope( - suite="fixture", dataset_path=dataset, config=config, records=[record], - metrics={ - **rank_metrics, - "confidence_intervals": { - field: { - "point": 1.0, - "low": 1.0, - "high": 1.0, - "n": 1, - "seed": 20260729, - "iterations": 1, - "strata_key": "category", - } - for field in rank_metrics - }, - "paired_bootstrap": { - "available": False, - "reason": "baseline_records_not_supplied", - "n": 0, - "delta": None, - "low": None, - "high": None, - "iterations": 1, - }, - "grounded_f1": {"available": False, "reason": "not_measured"}, - "abstention_f1": {"available": False, "reason": "not_measured"}, - "fixed_budget_curve": { - "available": True, - "rows": [{ - "token_budget": budget, - "status": "measured", - "n_total": 1, - "n_scored": 1, - "records": [dict(curve_record)], - **rank_metrics, - } for budget in CANONICAL_TOKEN_BUDGETS], - }, - }, - git_commit="a" * 40, - ) - report["system"]["git_dirty"] = False - report["models"] = {"embedder": { - "name": "FixtureEmbedder", - "model_id": profile["embedding"]["model"], - "revision": profile["embedding"]["revision"], - "sha256": "b" * 64, - }} - report["protocol"]["complete_dataset"] = True - report["protocol"]["source_questions"] = len(report["records"]) - return report - - -def test_metrics_cover_rank_sensitive_retrieval_quality(): - retrieved = ["noise", "evidence-a", "evidence-b"] - supporting = ["evidence-a", "evidence-b"] - assert metrics.mrr_at_k(retrieved, supporting, 3) == 0.5 - assert metrics.ndcg_at_k(retrieved, supporting, 3) > 0.6 - assert metrics.recall_at_k(retrieved[:1], supporting) == 0.0 - assert metrics.hit_at_k(retrieved[:1], supporting) == 0.0 - bundle = metrics.retrieval_metrics_at_depths(retrieved, supporting) - assert bundle["recall_at_1"] == 0.0 - assert bundle["recall_at_5"] == 1.0 - assert bundle["mrr_at_5"] == 0.5 - - -def test_envelope_hashes_dataset_config_and_retains_exclusions(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - excluded = {"question_id": "q2", "reason": "no_gold_evidence", "detail": ""} - records = [ - question_record("q1", category="state", supporting_ids=["m1"]), - question_record("q2", category="abstention", excluded=excluded), - ] - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"k": 5}, records=records, - metrics={"recall": 1.0}, git_commit="abc123", - ) - assert report["schema"] == SCHEMA - assert report["suite"]["sha256"] - assert report["system"]["config_sha256"] - assert report["protocol"] == { - "command": ["in_process"], - "config": {"k": 5}, - "token_accounting": { - "identity": "unspecified", - "revision": None, - "scope": "unspecified", - "method": "unspecified", - }, - "n_total": 2, - "n_scored": 1, - } - assert report["exclusions"] == [{ - "question_id": "q2", - "reason": "no_gold_evidence", - }] - assert json.loads(json.dumps(report))["schema"] == SCHEMA - - -def test_envelope_redacts_top_level_exclusion_detail(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"k": 5}, records=[], - exclusions=[{ - "question_id": "q1", "reason": "invalid", "detail": "private prompt text", - }], - ) - - assert report["exclusions"] == [{ - "question_id": "q1", - "reason": "invalid", - }] - - -def test_command_provenance_redacts_explicit_credential_arguments(): - assert redact_command([ - "python", "-m", "runner", "--api-key", "do-not-publish", "--token=value", - ]) == [ - "python", "-m", "runner", "--api-key", "", "--token", "", - ] - - -def test_command_provenance_redacts_assignment_header_and_url_credentials(): - assert redact_command([ - "API_KEY=super-secret", "--api_key", "also-secret", - "-H", "Authorization: Bearer another-secret", - "https://alice:password@example.test/run?access_token=last-secret&format=json", - ]) == [ - "API_KEY=", "--api_key", "", - "-H", "", - "https://@example.test/run?access_token=%3Credacted%3E&format=json", - ] - assert redact_command([ - "-ualice:password", "-psecret", "--user=alice:password", - "--header=Authorization: Bearer secret", - ]) == [ - "-u", "", "-p", "", "--user", "", - "--header", "", - ] - - -def test_command_provenance_redacts_compound_credential_assignments(): - assert redact_command([ - "AWS_SECRET_ACCESS_KEY=do-not-publish", - "AWS_ACCESS_KEY_ID=also-private", - "HTTP_AUTHORIZATION=Bearer another-secret", - "--token-budget", "512", - ]) == [ - "AWS_SECRET_ACCESS_KEY=", - "AWS_ACCESS_KEY_ID=", - "HTTP_AUTHORIZATION=", - "--token-budget", "512", - ] - - -def test_command_provenance_redacts_fragment_credentials_without_hiding_normal_options(): - assert redact_command([ - "--token-budget", "512", "--tokenizer-model", "reader-v1", - "https://example.test/callback#access_token=do-not-publish&state=visible", - ]) == [ - "--token-budget", "512", "--tokenizer-model", "reader-v1", - "https://example.test/callback#access_token=%3Credacted%3E&state=visible", - ] - - -def test_command_provenance_redacts_embedded_and_signed_url_credentials(): - assert redact_command([ - "DATASET_URL=https://example.test/data?access_token=do-not-publish", - "--dataset-url=https://example.test/data?X-Amz-Signature=signed&sig=azure", - "https://example.test/data?signature=generic", - ]) == [ - "DATASET_URL=https://example.test/data?access_token=%3Credacted%3E", - "--dataset-url=https://example.test/data?X-Amz-Signature=%3Credacted%3E&sig=%3Credacted%3E", - "https://example.test/data?signature=%3Credacted%3E", - ] - - -def test_command_provenance_redacts_userinfo_when_a_url_port_is_malformed(): - assert redact_command([ - "https://alice:password@example.test:notaport/path?access_token=do-not-publish", - ]) == [ - "https://@example.test:notaport/path?access_token=%3Credacted%3E", - ] - - -def test_command_provenance_fails_closed_when_url_splitting_rejects_userinfo(): - assert redact_command(["https://user:password@[invalid/path"]) == [""] - - -def test_canonical_profile_validator_and_immutable_artifact_writer(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - profile = json.loads(json.dumps(LONGMEMEVAL_V2_CANONICAL_PROFILE_TEMPLATE)) - profile["benchmark"]["repository_revision"] = "a" * 40 - profile["benchmark"]["dataset_revision"] = "b" * 40 - profile["reader"]["revision"] = "c" * 40 - profile["embedding"]["revision"] = "d" * 40 - profile["baseline_label"] = "full_hybrid" - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid", profile=profile - ) - report = _complete_canonical_report(dataset, config) - assert validate_report(report, canonical=True) == [] - dirty = deepcopy(report) - dirty["system"]["git_dirty"] = True - assert "canonical reports require a clean git worktree" in validate_report( - dirty, canonical=True - ) - artifact = tmp_path / "artifacts" / "run.json" - written = write_canonical_artifact(report, artifact, canonical=True) - assert written["sha256"] in artifact.with_name("run.json.sha256").read_text("ascii") - assert json.loads(artifact.read_text("utf-8"))["schema"] == SCHEMA - assert write_canonical_artifact(report, artifact, canonical=True) == written - changed = dict(report) - changed["records"] = [dict(report["records"][0])] - changed["records"][0]["latency_ms"] = 2.0 - with pytest.raises(FileExistsError): - write_canonical_artifact(changed, artifact, canonical=True) - - -def test_report_validator_recomputes_embedded_config_digest(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"baseline_label": "full_hybrid"}, - records=[question_record("q1")], git_commit="abc123", - ) - report["protocol"]["config"]["baseline_label"] = "dense_only" - - errors = validate_report(report) - - assert "system.config_sha256 must match the canonical protocol.config digest" in errors - - -def test_report_validator_rejects_inconsistent_or_duplicate_exclusions(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - excluded = {"question_id": "q2", "reason": "no_gold_evidence", "detail": ""} - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"k": 5}, - records=[ - question_record("q1"), - question_record("q2", excluded=excluded), - ], - git_commit="abc123", - ) - assert validate_report(report) == [] - - report["exclusions"] = [excluded, excluded] - errors = validate_report(report) - assert "exclusion question_id values must be unique" in errors - - report["exclusions"] = [] - errors = validate_report(report) - assert "top-level exclusions must exactly match per-record exclusions" in errors - - -def test_default_canonical_profile_is_pinned_and_rejects_mutable_revisions(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - report = _complete_canonical_report(dataset, config) - assert validate_report(report, canonical=True) == [] - assert all( - len(value) == 40 - for value in ( - config["canonical_profile"]["benchmark"]["repository_revision"], - config["canonical_profile"]["benchmark"]["dataset_revision"], - config["canonical_profile"]["reader"]["revision"], - config["canonical_profile"]["embedding"]["revision"], - ) - ) - assert config["token_budgets"] == list(CANONICAL_TOKEN_BUDGETS) - - config["canonical_profile"]["reader"]["revision"] = "main" - errors = validate_report(report, canonical=True) - assert any("reader.revision" in error and "immutable" in error for error in errors) - - -def test_canonical_validator_rejects_unpinned_commit_private_prompts_and_unlabeled_measurements(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - report = _complete_canonical_report(dataset, config) - report["system"]["git_commit"] = "not-a-commit" - report["records"][0]["q"] = "private source question" - report["records"][0]["question_sha256"] = "a" * 64 - report["records"][0].pop("context_token_method") - report["metrics"].pop("recall_at_10") - - errors = validate_report(report, canonical=True) - - assert any("git_commit" in error for error in errors) - assert "canonical records must not contain raw query text" in errors - assert "canonical records must not contain question-derived hashes" in errors - assert any("context_token_method" in error for error in errors) - assert any("metrics.recall_at_10" in error for error in errors) - - config["canonical_profile"]["reader"]["revision"] = "C" * 40 - errors = validate_report(report, canonical=True) - assert any("reader.revision" in error and "immutable" in error for error in errors) - - -def test_canonical_validator_requires_grounded_metrics_or_explicit_unavailability(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - report = _complete_canonical_report(dataset, config) - report["metrics"].pop("grounded_f1") - report["metrics"]["abstention_f1"] = {"available": False} - - errors = validate_report(report, canonical=True) - - assert any("grounded_f1" in error and "unavailable reason" in error for error in errors) - assert any("abstention_f1" in error and "unavailable reason" in error for error in errors) - - -def test_canonical_validator_requires_measured_rows_for_every_fixed_budget(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - report = _complete_canonical_report(dataset, config) - report["metrics"]["fixed_budget_curve"]["rows"].pop() - - errors = validate_report(report, canonical=True) - - assert "canonical fixed-budget curve must contain every canonical token budget" in errors - report["metrics"]["fixed_budget_curve"] = {"available": False, "reason": "not_run"} - errors = validate_report(report, canonical=True) - assert "canonical fixed-budget curve is unavailable and cannot qualify as evidence" in errors - - report = _complete_canonical_report(dataset, config) - report["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0]["excluded"] = True - errors = validate_report(report, canonical=True) - assert "canonical fixed-budget curve 256 records must preserve exclusion state" in errors - - -def test_canonical_validator_requires_complete_dataset_cardinality(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - assert validate_report(valid, canonical=True) == [] - - missing_complete = deepcopy(valid) - missing_complete["protocol"].pop("complete_dataset") - assert "canonical protocol.complete_dataset must be true" in validate_report( - missing_complete, canonical=True - ) - - for invalid_count in (True, 0, 2): - mismatched = deepcopy(valid) - mismatched["protocol"]["source_questions"] = invalid_count - errors = validate_report(mismatched, canonical=True) - assert any("protocol.source_questions" in error for error in errors) - - -def test_canonical_validator_rejects_invalid_numeric_and_token_accounting(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - config["token_budget"] = 4 - valid = _complete_canonical_report(dataset, config) - valid["records"][0]["usage"] = { - "budget_tokens": 4, - "context_tokens": 3, - "token_counter": valid["records"][0]["context_tokenizer_identity"], - } - assert validate_report(valid, canonical=True) == [] - - mutations = ( - (("metrics", "recall_at_1"), True, "metrics.recall_at_1"), - (("records", 0, "recall_at_1"), True, "records require recall_at_1"), - (("records", 0, "latency_ms"), float("inf"), "latency_ms"), - (("records", 0, "context_tokens"), float("nan"), "context_tokens"), - (("records", 0, "context_tokens"), -1, "context_tokens"), - (("records", 0, "context_tokens"), 5, "must not exceed protocol token_budget"), - ( - ("records", 0, "usage", "context_tokens"), - 5, - "usage.context_tokens must not exceed usage.budget_tokens", - ), - ( - ("records", 0, "usage", "budget_tokens"), - 5, - "usage.budget_tokens must equal protocol token_budget", - ), - ( - ("records", 0, "usage", "source_tokens"), - True, - "usage.source_tokens must be non-negative and finite", - ), - ( - ("records", 0, "usage", "savings_ratio"), - float("inf"), - "usage.savings_ratio must be a number in [0, 1]", - ), - ( - ("metrics", "fixed_budget_curve", "rows", 0, "recall_at_1"), - True, - "fixed-budget curve 256 requires recall_at_1", - ), - ( - ("metrics", "fixed_budget_curve", "rows", 0, "records", 0, "context_tokens"), - 257, - "context_tokens within budget", - ), - ) - for path, value, expected in mutations: - report = deepcopy(valid) - target = report - for key in path[:-1]: - target = target[key] - target[path[-1]] = value - errors = validate_report(report, canonical=True) - assert any(expected in error for error in errors), (path, errors) - - -def test_canonical_validator_rejects_tampered_confidence_intervals(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - assert validate_report(valid, canonical=True) == [] - - mutations = ( - ("point", float("nan"), "point/low/high must be finite"), - ("low", -0.1, "point/low/high must be finite"), - ("high", 1.1, "point/low/high must be finite"), - ("high", 0.5, "low <= point <= high"), - ("point", 0.5, ".point must match metrics.recall_at_1"), - ("n", 2, ".n must equal the non-excluded record count"), - ("seed", -1, ".seed must be a non-negative integer"), - ("iterations", 0, ".iterations must be a positive integer"), - ("iterations", -1, ".iterations must be a positive integer"), - ("iterations", True, ".iterations must be a positive integer"), - ("strata_key", "topic", ".strata_key must equal category"), - ("low", 0.75, "must exactly match deterministic recomputation"), - ) - for key, value, expected in mutations: - report = deepcopy(valid) - report["metrics"]["confidence_intervals"]["recall_at_1"][key] = value - errors = validate_report(report, canonical=True) - assert any(expected in error for error in errors), (key, value, errors) - for metric_name in ( - "recall_at_1", "recall_at_5", "recall_at_10", - "mrr_at_1", "mrr_at_5", "mrr_at_10", - "ndcg_at_1", "ndcg_at_5", "ndcg_at_10", - ): - report = deepcopy(valid) - interval = report["metrics"]["confidence_intervals"][metric_name] - if interval["low"] > 0: - interval["low"] = round(interval["low"] - 0.000001, 6) - else: - interval["high"] = round(interval["high"] + 0.000001, 6) - errors = validate_report(report, canonical=True) - assert any( - "must exactly match deterministic recomputation" in error - for error in errors - ), (metric_name, errors) - - extra = deepcopy(valid) - extra["metrics"]["confidence_intervals"]["recall_at_1"]["mean"] = 1.0 - errors = validate_report(extra, canonical=True) - assert any("must match the canonical confidence interval schema" in error for error in errors) - - missing = deepcopy(valid) - missing["metrics"]["confidence_intervals"].pop("recall_at_1") - errors = validate_report(missing, canonical=True) - assert ( - "canonical metrics.confidence_intervals must exactly cover every rank metric" - in errors - ) - - -def test_canonical_validator_rejects_tampered_paired_bootstrap_payloads(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - - unavailable_mutations = ( - ("reason", "", ".reason must be a non-empty string"), - ("n", 1, ".n must be zero when unavailable"), - ("delta", 0.0, "delta/low/high must be null when unavailable"), - ("iterations", 0, ".iterations must be a positive integer"), - ("iterations", True, ".iterations must be a positive integer"), - ) - for key, value, expected in unavailable_mutations: - report = deepcopy(valid) - report["metrics"]["paired_bootstrap"][key] = value - errors = validate_report(report, canonical=True) - assert any(expected in error for error in errors), (key, value, errors) - - available = deepcopy(valid) - available["metrics"]["paired_bootstrap"] = { - "available": True, - "metric": "recall_at_5", - "delta": 0.25, - "low": 0.0, - "high": 0.5, - "n": 1, - "seed": 20260729, - "iterations": 20, - } - errors = validate_report(available, canonical=True) - assert any( - "must be unavailable until an immutable baseline artifact" in error - for error in errors - ) - - -def test_canonical_validator_recomputes_all_rank_aggregates_from_record_ids(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - - top_level = deepcopy(valid) - top_level["metrics"]["recall_at_5"] = 0.5 - errors = validate_report(top_level, canonical=True) - assert ( - "canonical metrics.recall_at_5 must equal the non-excluded record mean" - in errors - ) - - curve_aggregate = deepcopy(valid) - curve_aggregate["metrics"]["fixed_budget_curve"]["rows"][0]["ndcg_at_10"] = 0.5 - errors = validate_report(curve_aggregate, canonical=True) - assert any( - "fixed-budget curve 256 ndcg_at_10" in error - and "non-excluded record mean" in error - for error in errors - ) - - curve_measurement = deepcopy(valid) - measurement = curve_measurement["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0] - measurement["retrieved_ids"] = [] - errors = validate_report(curve_measurement, canonical=True) - assert any( - "fixed-budget curve 256 record recall_at_1" in error - and "retrieved_ids and supporting_ids" in error - for error in errors - ) - - -def test_canonical_validator_derives_numeric_grounded_metrics_from_labels(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - - unlabeled = _complete_canonical_report(dataset, config) - unlabeled["metrics"]["grounded_f1"] = 0.75 - unlabeled["metrics"]["abstention_f1"] = 0.75 - errors = validate_report(unlabeled, canonical=True) - assert any( - "metrics.grounded_f1 requires labeled per-question grounded values" in error - and "unavailable reason" in error - for error in errors - ) - assert any( - "metrics.abstention_f1 requires labeled per-question abstained values" in error - and "unavailable reason" in error - for error in errors - ) - - measured = _complete_canonical_report(dataset, config) - measured["records"][0].update({ - "answerable": True, - "grounded": True, - "abstained": False, - }) - measured["metrics"]["grounded"] = { - "available": True, - **metrics.grounded_precision_recall_f1([True], [True]), - } - measured["metrics"]["abstention"] = { - "available": True, - **metrics.abstention_precision_recall_f1([False], [True]), - } - measured["metrics"]["grounded_f1"] = 1.0 - measured["metrics"]["abstention_f1"] = 1.0 - assert validate_report(measured, canonical=True) == [] - - bad_count = deepcopy(measured) - bad_count["metrics"]["grounded"]["n"] = 2 - errors = validate_report(bad_count, canonical=True) - assert ( - "canonical metrics.grounded.n must be recomputed from per-question labels" - in errors - ) - - measured["metrics"]["grounded_f1"] = 0.0 - errors = validate_report(measured, canonical=True) - assert ( - "canonical metrics.grounded_f1 must be recomputed from per-question labels" - in errors - ) - - -def test_canonical_validator_requires_pinned_reader_tokenizer_identity(tmp_path): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - config = canonical_benchmark_config( - run_label="release-candidate", baseline_label="full_hybrid" - ) - valid = _complete_canonical_report(dataset, config) - - estimated = deepcopy(valid) - estimated["records"][0]["context_token_method"] = "deterministic_estimate" - estimated["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0][ - "context_token_method" - ] = "deterministic_estimate" - errors = validate_report(estimated, canonical=True) - assert any( - "context_token_method=pinned_reader_content_tokenizer" in error - for error in errors - ) - assert any( - "fixed-budget curve 256 records require" in error - and "context_token_method=pinned_reader_content_tokenizer" in error - for error in errors - ) - - mismatched = deepcopy(valid) - mismatched["records"][0]["context_tokenizer_identity"] = "other/model@" + "e" * 40 - mismatched["records"][0]["usage"]["token_counter"] = "other/model@" + "e" * 40 - errors = validate_report(mismatched, canonical=True) - assert any("context_tokenizer_identity must match" in error for error in errors) - assert any("usage.token_counter must match" in error for error in errors) - - -def test_benchmark_cli_writes_canonical_json_and_checksum(tmp_path, capsys): - dataset = tmp_path / "fixture.jsonl" - dataset.write_text('{"id":"one"}\n', encoding="utf-8") - report = report_envelope( - suite="fixture", dataset_path=dataset, config={"k": 5}, - records=[question_record("q1")], git_commit="abc123", - ) - source = tmp_path / "source.json" - source.write_text(json.dumps(report), encoding="utf-8") - artifact = tmp_path / "artifact.json" - assert main(["--input", str(source), "--output", str(artifact)]) == 0 - assert artifact.exists() and artifact.with_name("artifact.json.sha256").exists() - assert "sha256" in capsys.readouterr().out - - -def test_exact_tokenizer_fallback_budget_curves_and_deterministic_cis(): - assert count_tokens("abc", CharacterTokenizer()) == {"tokens": 3, "method": "injected"} - assert count_tokens("one two")["method"] == "deterministic_estimate" - records = [ - {"category": "a", "supporting_ids": ["m1"], "chunks": [ - {"id": "m1", "tokens": 3}, {"id": "m2", "tokens": 3} - ]}, - {"category": "b", "supporting_ids": ["m2"], "chunks": [ - {"id": "m1", "tokens": 3}, {"id": "m2", "tokens": 3} - ]}, - ] - curve = fixed_budget_curve(records, [3, 6]) - assert curve[0]["recall"] == 0.5 - assert curve[1]["recall"] == 1.0 - def metric(rows): - return sum(row["value"] for row in rows) / len(rows) - ci_one = stratified_bootstrap_ci( - [{"category": "a", "value": 1.0}, {"category": "b", "value": 0.0}], - metric, iterations=40, seed=4, - ) - ci_two = stratified_bootstrap_ci( - [{"category": "a", "value": 1.0}, {"category": "b", "value": 0.0}], - metric, iterations=40, seed=4, - ) - assert ci_one == ci_two - paired = paired_bootstrap_ci([(1.0, 0.0), (0.0, 0.0)], iterations=40, seed=4) - assert paired["delta"] == 0.5 and paired["n"] == 2 + + runs = {run["id"]: run for run in artifact["runs"]} + assert set(runs) == { + "offline-chunking", + "offline-performance", + "offline-grounded", + } + for run in runs.values(): + assert hashlib.sha256(run["command"].encode()).hexdigest() == run["config_digest"] + + chunking = offline_release_evidence["chunking"] + chunking_result = runs["offline-chunking"]["result"] + for mode in ("whole", "chunked"): + live = chunking["reports"][mode] + recorded = chunking_result[mode] + assert recorded["memories"] == live["memories_stored"] + assert recorded["recall_at_k"] == live["recall_at_k"] + assert recorded["mean_context_tokens"] == live["mean_context_tokens"] + assert recorded["mean_evidence_tokens"] == live["mean_evidence_tokens"] + assert recorded["max_stored_tokens"] == live["max_stored_tokens"] + assert chunking_result["context_reduction_pct"] == chunking["context_reduction_pct"] + + performance = offline_release_evidence["performance"] + performance_result = runs["offline-performance"]["result"] + assert performance_result["questions"] == performance["corpus"]["questions"] + assert performance_result["timed_recalls"] == performance["run"]["timed_recalls"] + assert performance_result["recall_at_k"] == performance["quality"]["recall_at_k"] + assert performance_result["hit_at_k"] == performance["quality"]["hit_at_k"] + assert ( + performance_result["answer_token_recall"] + == performance["quality"]["answer_token_recall"] + ) + + # These values are deterministic fixture aggregates, not wall-clock timing + # observations. Approximate comparisons would let serializer or count drift + # pass the publication contract unnoticed. + assert performance_result["mean_context_tokens"] == performance["context"]["mean_tokens"] + assert performance_result["max_context_tokens"] == performance["context"]["max_tokens"] + assert ( + performance_result["full_serialized_payload_tokens"] + == performance["context"]["full_serialized_payload_tokens"] + ) + assert ( + performance_result["compact_serialized_payload_tokens"] + == performance["context"]["compact_serialized_payload_tokens"] + ) + assert ( + performance_result["saved_serialized_payload_tokens"] + == performance["context"]["saved_serialized_payload_tokens"] + ) + assert ( + performance_result["serialized_payload_savings_ratio"] + == performance["context"]["serialized_payload_savings_ratio"] + ) + + grounded = offline_release_evidence["grounded"] + grounded_result = runs["offline-grounded"]["result"] + assert grounded_result == { + "answerable": grounded["n_answerable"], + "grounded": grounded["grounded_hits"], + "off_topic": grounded["n_unanswerable"], + "quarantined": grounded["n_quarantine"], + "abstained": grounded["abstain_hits"], + "quarantine_hits": grounded["quarantine_hits"], + "decision_accuracy": grounded["accuracy"], + } + + surfaces = ( + ROOT / "README.md", + ROOT / "BENCHMARKS.md", + ROOT / "docs" / "images" / "context-efficiency.svg", + ROOT / "docs" / "images" / "evidence-backed-agent-examples.svg", + ) + for surface in surfaces: + assert expected_sha in surface.read_text(encoding="utf-8") + + claimed_ids = set( + re.findall( + r"offline-(?:chunking|performance|grounded)", + "\n".join(path.read_text(encoding="utf-8") for path in surfaces), + ) + ) + assert claimed_ids == set(runs) + + +def test_benchmark_guide_tracks_the_live_offline_evaluators(): + """Method prose must change whenever its executable offline evidence changes. + + Values are interpolated from the COMMITTED registry artifact — the publication + source of truth — so guide text cannot drift from the evidence it cites. + """ + benchmarks = (ROOT / "BENCHMARKS.md").read_text(encoding="utf-8") + normalized = " ".join(benchmarks.split()) + committed = _committed_evidence() + chunking = committed["chunking"] + whole = chunking["whole"] + chunked = chunking["chunked"] + performance = committed["performance"] + payload_samples = performance["questions"] + + for evidence in ( + f"falls from {whole['mean_context_tokens']:.1f} to " + f"{chunked['mean_context_tokens']:.1f} tokens", + f"{whole['mean_context_tokens'] - chunked['mean_context_tokens']:.1f} fewer, " + f"{chunking['context_reduction_pct']:.1f}% lower", + f"falls from {whole['mean_evidence_tokens']:.1f} to " + f"{chunked['mean_evidence_tokens']:.1f} tokens", + "Payload proxies are sampled once per question", + "not serialized MCP envelopes or transport responses", + f"{payload_samples} payload samples total **" + f"{performance['full_serialized_payload_tokens']:,}** full-proxy", + f"versus **{performance['compact_serialized_payload_tokens']:,}** compact-proxy tokens", + f"avoiding **{performance['saved_serialized_payload_tokens']:,}** proxy tokens", + f"**{100 * performance['serialized_payload_savings_ratio']:.2f}% lower**", + f"averages **{performance['mean_context_tokens']:.2f}** tokens and reaches " + f"**{performance['max_context_tokens']}**", + ): + assert evidence in normalized + + + +def _complete_canonical_report(dataset, config): + """Minimal but fully auditable canonical envelope for validator coverage.""" + profile = config["canonical_profile"] + tokenizer_identity = ( + f"{profile['reader']['model']}@{profile['reader']['revision']}" + ) + record = question_record( + "q1", category="state", context_tokens=3, latency_ms=1.25, + retrieved_ids=["support"], supporting_ids=["support"], + recall_at_1=1.0, recall_at_5=1.0, recall_at_10=1.0, + mrr_at_1=1.0, mrr_at_5=1.0, mrr_at_10=1.0, + ndcg_at_1=1.0, ndcg_at_5=1.0, ndcg_at_10=1.0, + usage={ + "budget_tokens": config.get("token_budget") or 3, + "context_tokens": 3, + "token_counter": tokenizer_identity, + }, + ) + record["context_token_method"] = "pinned_reader_content_tokenizer" + record["context_tokenizer_identity"] = tokenizer_identity + rank_metrics = { + f"{metric}_at_{depth}": 1.0 + for metric in ("recall", "mrr", "ndcg") + for depth in (1, 5, 10) + } + curve_record = { + "question_id": "q1", + "excluded": False, + "context_tokens": 3, + "context_token_method": "pinned_reader_content_tokenizer", + "context_tokenizer_identity": tokenizer_identity, + "retrieved_ids": ["support"], + "supporting_ids": ["support"], + **rank_metrics, + } + report = report_envelope( + suite="fixture", dataset_path=dataset, config=config, records=[record], + metrics={ + **rank_metrics, + "confidence_intervals": { + field: { + "point": 1.0, + "low": 1.0, + "high": 1.0, + "n": 1, + "seed": 20260729, + "iterations": 1, + "strata_key": "category", + } + for field in rank_metrics + }, + "paired_bootstrap": { + "available": False, + "reason": "baseline_records_not_supplied", + "n": 0, + "delta": None, + "low": None, + "high": None, + "iterations": 1, + }, + "grounded_f1": {"available": False, "reason": "not_measured"}, + "abstention_f1": {"available": False, "reason": "not_measured"}, + "fixed_budget_curve": { + "available": True, + "rows": [{ + "token_budget": budget, + "status": "measured", + "n_total": 1, + "n_scored": 1, + "records": [dict(curve_record)], + **rank_metrics, + } for budget in CANONICAL_TOKEN_BUDGETS], + }, + }, + git_commit="a" * 40, + ) + report["system"]["git_dirty"] = False + report["models"] = {"embedder": { + "name": "FixtureEmbedder", + "model_id": profile["embedding"]["model"], + "revision": profile["embedding"]["revision"], + "sha256": "b" * 64, + }} + report["protocol"]["complete_dataset"] = True + report["protocol"]["source_questions"] = len(report["records"]) + return report + + +def test_metrics_cover_rank_sensitive_retrieval_quality(): + retrieved = ["noise", "evidence-a", "evidence-b"] + supporting = ["evidence-a", "evidence-b"] + assert metrics.mrr_at_k(retrieved, supporting, 3) == 0.5 + assert metrics.ndcg_at_k(retrieved, supporting, 3) > 0.6 + assert metrics.recall_at_k(retrieved[:1], supporting) == 0.0 + assert metrics.hit_at_k(retrieved[:1], supporting) == 0.0 + bundle = metrics.retrieval_metrics_at_depths(retrieved, supporting) + assert bundle["recall_at_1"] == 0.0 + assert bundle["recall_at_5"] == 1.0 + assert bundle["mrr_at_5"] == 0.5 + + +def test_envelope_hashes_dataset_config_and_retains_exclusions(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + excluded = {"question_id": "q2", "reason": "no_gold_evidence", "detail": ""} + records = [ + question_record("q1", category="state", supporting_ids=["m1"]), + question_record("q2", category="abstention", excluded=excluded), + ] + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"k": 5}, records=records, + metrics={"recall": 1.0}, git_commit="abc123", + ) + assert report["schema"] == SCHEMA + assert report["suite"]["sha256"] + assert report["system"]["config_sha256"] + assert report["protocol"] == { + "command": ["in_process"], + "config": {"k": 5}, + "token_accounting": { + "identity": "unspecified", + "revision": None, + "scope": "unspecified", + "method": "unspecified", + }, + "n_total": 2, + "n_scored": 1, + } + assert report["exclusions"] == [{ + "question_id": "q2", + "reason": "no_gold_evidence", + }] + assert json.loads(json.dumps(report))["schema"] == SCHEMA + + +def test_envelope_redacts_top_level_exclusion_detail(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"k": 5}, records=[], + exclusions=[{ + "question_id": "q1", "reason": "invalid", "detail": "private prompt text", + }], + ) + + assert report["exclusions"] == [{ + "question_id": "q1", + "reason": "invalid", + }] + + +def test_command_provenance_redacts_explicit_credential_arguments(): + assert redact_command([ + "python", "-m", "runner", "--api-key", "do-not-publish", "--token=value", + ]) == [ + "python", "-m", "runner", "--api-key", "", "--token", "", + ] + + +def test_command_provenance_redacts_assignment_header_and_url_credentials(): + assert redact_command([ + "API_KEY=super-secret", "--api_key", "also-secret", + "-H", "Authorization: Bearer another-secret", + "https://alice:password@example.test/run?access_token=last-secret&format=json", + ]) == [ + "API_KEY=", "--api_key", "", + "-H", "", + "https://@example.test/run?access_token=%3Credacted%3E&format=json", + ] + assert redact_command([ + "-ualice:password", "-psecret", "--user=alice:password", + "--header=Authorization: Bearer secret", + ]) == [ + "-u", "", "-p", "", "--user", "", + "--header", "", + ] + + +def test_command_provenance_redacts_compound_credential_assignments(): + assert redact_command([ + "AWS_SECRET_ACCESS_KEY=do-not-publish", + "AWS_ACCESS_KEY_ID=also-private", + "HTTP_AUTHORIZATION=Bearer another-secret", + "--token-budget", "512", + ]) == [ + "AWS_SECRET_ACCESS_KEY=", + "AWS_ACCESS_KEY_ID=", + "HTTP_AUTHORIZATION=", + "--token-budget", "512", + ] + + +def test_command_provenance_redacts_fragment_credentials_without_hiding_normal_options(): + assert redact_command([ + "--token-budget", "512", "--tokenizer-model", "reader-v1", + "https://example.test/callback#access_token=do-not-publish&state=visible", + ]) == [ + "--token-budget", "512", "--tokenizer-model", "reader-v1", + "https://example.test/callback#access_token=%3Credacted%3E&state=visible", + ] + + +def test_command_provenance_redacts_embedded_and_signed_url_credentials(): + assert redact_command([ + "DATASET_URL=https://example.test/data?access_token=do-not-publish", + "--dataset-url=https://example.test/data?X-Amz-Signature=signed&sig=azure", + "https://example.test/data?signature=generic", + ]) == [ + "DATASET_URL=https://example.test/data?access_token=%3Credacted%3E", + "--dataset-url=https://example.test/data?X-Amz-Signature=%3Credacted%3E&sig=%3Credacted%3E", + "https://example.test/data?signature=%3Credacted%3E", + ] + + +def test_command_provenance_redacts_userinfo_when_a_url_port_is_malformed(): + assert redact_command([ + "https://alice:password@example.test:notaport/path?access_token=do-not-publish", + ]) == [ + "https://@example.test:notaport/path?access_token=%3Credacted%3E", + ] + + +def test_command_provenance_fails_closed_when_url_splitting_rejects_userinfo(): + assert redact_command(["https://user:password@[invalid/path"]) == [""] + + +def test_canonical_profile_validator_and_immutable_artifact_writer(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + profile = json.loads(json.dumps(LONGMEMEVAL_V2_CANONICAL_PROFILE_TEMPLATE)) + profile["benchmark"]["repository_revision"] = "a" * 40 + profile["benchmark"]["dataset_revision"] = "b" * 40 + profile["reader"]["revision"] = "c" * 40 + profile["embedding"]["revision"] = "d" * 40 + profile["baseline_label"] = "full_hybrid" + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid", profile=profile + ) + report = _complete_canonical_report(dataset, config) + assert validate_report(report, canonical=True) == [] + dirty = deepcopy(report) + dirty["system"]["git_dirty"] = True + assert "canonical reports require a clean git worktree" in validate_report( + dirty, canonical=True + ) + artifact = tmp_path / "artifacts" / "run.json" + written = write_canonical_artifact(report, artifact, canonical=True) + assert written["sha256"] in artifact.with_name("run.json.sha256").read_text("ascii") + assert json.loads(artifact.read_text("utf-8"))["schema"] == SCHEMA + assert write_canonical_artifact(report, artifact, canonical=True) == written + changed = dict(report) + changed["records"] = [dict(report["records"][0])] + changed["records"][0]["latency_ms"] = 2.0 + with pytest.raises(FileExistsError): + write_canonical_artifact(changed, artifact, canonical=True) + + +def test_report_validator_recomputes_embedded_config_digest(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"baseline_label": "full_hybrid"}, + records=[question_record("q1")], git_commit="abc123", + ) + report["protocol"]["config"]["baseline_label"] = "dense_only" + + errors = validate_report(report) + + assert "system.config_sha256 must match the canonical protocol.config digest" in errors + + +def test_report_validator_rejects_inconsistent_or_duplicate_exclusions(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + excluded = {"question_id": "q2", "reason": "no_gold_evidence", "detail": ""} + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"k": 5}, + records=[ + question_record("q1"), + question_record("q2", excluded=excluded), + ], + git_commit="abc123", + ) + assert validate_report(report) == [] + + report["exclusions"] = [excluded, excluded] + errors = validate_report(report) + assert "exclusion question_id values must be unique" in errors + + report["exclusions"] = [] + errors = validate_report(report) + assert "top-level exclusions must exactly match per-record exclusions" in errors + + +def test_default_canonical_profile_is_pinned_and_rejects_mutable_revisions(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + report = _complete_canonical_report(dataset, config) + assert validate_report(report, canonical=True) == [] + assert all( + len(value) == 40 + for value in ( + config["canonical_profile"]["benchmark"]["repository_revision"], + config["canonical_profile"]["benchmark"]["dataset_revision"], + config["canonical_profile"]["reader"]["revision"], + config["canonical_profile"]["embedding"]["revision"], + ) + ) + assert config["token_budgets"] == list(CANONICAL_TOKEN_BUDGETS) + + config["canonical_profile"]["reader"]["revision"] = "main" + errors = validate_report(report, canonical=True) + assert any("reader.revision" in error and "immutable" in error for error in errors) + + +def test_canonical_validator_rejects_unpinned_commit_private_prompts_and_unlabeled_measurements(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + report = _complete_canonical_report(dataset, config) + report["system"]["git_commit"] = "not-a-commit" + report["records"][0]["q"] = "private source question" + report["records"][0]["question_sha256"] = "a" * 64 + report["records"][0].pop("context_token_method") + report["metrics"].pop("recall_at_10") + + errors = validate_report(report, canonical=True) + + assert any("git_commit" in error for error in errors) + assert "canonical records must not contain raw query text" in errors + assert "canonical records must not contain question-derived hashes" in errors + assert any("context_token_method" in error for error in errors) + assert any("metrics.recall_at_10" in error for error in errors) + + config["canonical_profile"]["reader"]["revision"] = "C" * 40 + errors = validate_report(report, canonical=True) + assert any("reader.revision" in error and "immutable" in error for error in errors) + + +def test_canonical_validator_requires_grounded_metrics_or_explicit_unavailability(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + report = _complete_canonical_report(dataset, config) + report["metrics"].pop("grounded_f1") + report["metrics"]["abstention_f1"] = {"available": False} + + errors = validate_report(report, canonical=True) + + assert any("grounded_f1" in error and "unavailable reason" in error for error in errors) + assert any("abstention_f1" in error and "unavailable reason" in error for error in errors) + + +def test_canonical_validator_requires_measured_rows_for_every_fixed_budget(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + report = _complete_canonical_report(dataset, config) + report["metrics"]["fixed_budget_curve"]["rows"].pop() + + errors = validate_report(report, canonical=True) + + assert "canonical fixed-budget curve must contain every canonical token budget" in errors + report["metrics"]["fixed_budget_curve"] = {"available": False, "reason": "not_run"} + errors = validate_report(report, canonical=True) + assert "canonical fixed-budget curve is unavailable and cannot qualify as evidence" in errors + + report = _complete_canonical_report(dataset, config) + report["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0]["excluded"] = True + errors = validate_report(report, canonical=True) + assert "canonical fixed-budget curve 256 records must preserve exclusion state" in errors + + +def test_canonical_validator_requires_complete_dataset_cardinality(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + assert validate_report(valid, canonical=True) == [] + + missing_complete = deepcopy(valid) + missing_complete["protocol"].pop("complete_dataset") + assert "canonical protocol.complete_dataset must be true" in validate_report( + missing_complete, canonical=True + ) + + for invalid_count in (True, 0, 2): + mismatched = deepcopy(valid) + mismatched["protocol"]["source_questions"] = invalid_count + errors = validate_report(mismatched, canonical=True) + assert any("protocol.source_questions" in error for error in errors) + + +def test_canonical_validator_rejects_invalid_numeric_and_token_accounting(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + config["token_budget"] = 4 + valid = _complete_canonical_report(dataset, config) + valid["records"][0]["usage"] = { + "budget_tokens": 4, + "context_tokens": 3, + "token_counter": valid["records"][0]["context_tokenizer_identity"], + } + assert validate_report(valid, canonical=True) == [] + + mutations = ( + (("metrics", "recall_at_1"), True, "metrics.recall_at_1"), + (("records", 0, "recall_at_1"), True, "records require recall_at_1"), + (("records", 0, "latency_ms"), float("inf"), "latency_ms"), + (("records", 0, "context_tokens"), float("nan"), "context_tokens"), + (("records", 0, "context_tokens"), -1, "context_tokens"), + (("records", 0, "context_tokens"), 5, "must not exceed protocol token_budget"), + ( + ("records", 0, "usage", "context_tokens"), + 5, + "usage.context_tokens must not exceed usage.budget_tokens", + ), + ( + ("records", 0, "usage", "budget_tokens"), + 5, + "usage.budget_tokens must equal protocol token_budget", + ), + ( + ("records", 0, "usage", "source_tokens"), + True, + "usage.source_tokens must be non-negative and finite", + ), + ( + ("records", 0, "usage", "savings_ratio"), + float("inf"), + "usage.savings_ratio must be a number in [0, 1]", + ), + ( + ("metrics", "fixed_budget_curve", "rows", 0, "recall_at_1"), + True, + "fixed-budget curve 256 requires recall_at_1", + ), + ( + ("metrics", "fixed_budget_curve", "rows", 0, "records", 0, "context_tokens"), + 257, + "context_tokens within budget", + ), + ) + for path, value, expected in mutations: + report = deepcopy(valid) + target = report + for key in path[:-1]: + target = target[key] + target[path[-1]] = value + errors = validate_report(report, canonical=True) + assert any(expected in error for error in errors), (path, errors) + + +def test_canonical_validator_rejects_tampered_confidence_intervals(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + assert validate_report(valid, canonical=True) == [] + + mutations = ( + ("point", float("nan"), "point/low/high must be finite"), + ("low", -0.1, "point/low/high must be finite"), + ("high", 1.1, "point/low/high must be finite"), + ("high", 0.5, "low <= point <= high"), + ("point", 0.5, ".point must match metrics.recall_at_1"), + ("n", 2, ".n must equal the non-excluded record count"), + ("seed", -1, ".seed must be a non-negative integer"), + ("iterations", 0, ".iterations must be a positive integer"), + ("iterations", -1, ".iterations must be a positive integer"), + ("iterations", True, ".iterations must be a positive integer"), + ("strata_key", "topic", ".strata_key must equal category"), + ("low", 0.75, "must exactly match deterministic recomputation"), + ) + for key, value, expected in mutations: + report = deepcopy(valid) + report["metrics"]["confidence_intervals"]["recall_at_1"][key] = value + errors = validate_report(report, canonical=True) + assert any(expected in error for error in errors), (key, value, errors) + for metric_name in ( + "recall_at_1", "recall_at_5", "recall_at_10", + "mrr_at_1", "mrr_at_5", "mrr_at_10", + "ndcg_at_1", "ndcg_at_5", "ndcg_at_10", + ): + report = deepcopy(valid) + interval = report["metrics"]["confidence_intervals"][metric_name] + if interval["low"] > 0: + interval["low"] = round(interval["low"] - 0.000001, 6) + else: + interval["high"] = round(interval["high"] + 0.000001, 6) + errors = validate_report(report, canonical=True) + assert any( + "must exactly match deterministic recomputation" in error + for error in errors + ), (metric_name, errors) + + extra = deepcopy(valid) + extra["metrics"]["confidence_intervals"]["recall_at_1"]["mean"] = 1.0 + errors = validate_report(extra, canonical=True) + assert any("must match the canonical confidence interval schema" in error for error in errors) + + missing = deepcopy(valid) + missing["metrics"]["confidence_intervals"].pop("recall_at_1") + errors = validate_report(missing, canonical=True) + assert ( + "canonical metrics.confidence_intervals must exactly cover every rank metric" + in errors + ) + + +def test_canonical_validator_rejects_tampered_paired_bootstrap_payloads(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + + unavailable_mutations = ( + ("reason", "", ".reason must be a non-empty string"), + ("n", 1, ".n must be zero when unavailable"), + ("delta", 0.0, "delta/low/high must be null when unavailable"), + ("iterations", 0, ".iterations must be a positive integer"), + ("iterations", True, ".iterations must be a positive integer"), + ) + for key, value, expected in unavailable_mutations: + report = deepcopy(valid) + report["metrics"]["paired_bootstrap"][key] = value + errors = validate_report(report, canonical=True) + assert any(expected in error for error in errors), (key, value, errors) + + available = deepcopy(valid) + available["metrics"]["paired_bootstrap"] = { + "available": True, + "metric": "recall_at_5", + "delta": 0.25, + "low": 0.0, + "high": 0.5, + "n": 1, + "seed": 20260729, + "iterations": 20, + } + errors = validate_report(available, canonical=True) + assert any( + "must be unavailable until an immutable baseline artifact" in error + for error in errors + ) + + +def test_canonical_validator_recomputes_all_rank_aggregates_from_record_ids(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + + top_level = deepcopy(valid) + top_level["metrics"]["recall_at_5"] = 0.5 + errors = validate_report(top_level, canonical=True) + assert ( + "canonical metrics.recall_at_5 must equal the non-excluded record mean" + in errors + ) + + curve_aggregate = deepcopy(valid) + curve_aggregate["metrics"]["fixed_budget_curve"]["rows"][0]["ndcg_at_10"] = 0.5 + errors = validate_report(curve_aggregate, canonical=True) + assert any( + "fixed-budget curve 256 ndcg_at_10" in error + and "non-excluded record mean" in error + for error in errors + ) + + curve_measurement = deepcopy(valid) + measurement = curve_measurement["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0] + measurement["retrieved_ids"] = [] + errors = validate_report(curve_measurement, canonical=True) + assert any( + "fixed-budget curve 256 record recall_at_1" in error + and "retrieved_ids and supporting_ids" in error + for error in errors + ) + + +def test_canonical_validator_derives_numeric_grounded_metrics_from_labels(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + + unlabeled = _complete_canonical_report(dataset, config) + unlabeled["metrics"]["grounded_f1"] = 0.75 + unlabeled["metrics"]["abstention_f1"] = 0.75 + errors = validate_report(unlabeled, canonical=True) + assert any( + "metrics.grounded_f1 requires labeled per-question grounded values" in error + and "unavailable reason" in error + for error in errors + ) + assert any( + "metrics.abstention_f1 requires labeled per-question abstained values" in error + and "unavailable reason" in error + for error in errors + ) + + measured = _complete_canonical_report(dataset, config) + measured["records"][0].update({ + "answerable": True, + "grounded": True, + "abstained": False, + }) + measured["metrics"]["grounded"] = { + "available": True, + **metrics.grounded_precision_recall_f1([True], [True]), + } + measured["metrics"]["abstention"] = { + "available": True, + **metrics.abstention_precision_recall_f1([False], [True]), + } + measured["metrics"]["grounded_f1"] = 1.0 + measured["metrics"]["abstention_f1"] = 1.0 + assert validate_report(measured, canonical=True) == [] + + bad_count = deepcopy(measured) + bad_count["metrics"]["grounded"]["n"] = 2 + errors = validate_report(bad_count, canonical=True) + assert ( + "canonical metrics.grounded.n must be recomputed from per-question labels" + in errors + ) + + measured["metrics"]["grounded_f1"] = 0.0 + errors = validate_report(measured, canonical=True) + assert ( + "canonical metrics.grounded_f1 must be recomputed from per-question labels" + in errors + ) + + +def test_canonical_validator_requires_pinned_reader_tokenizer_identity(tmp_path): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + config = canonical_benchmark_config( + run_label="release-candidate", baseline_label="full_hybrid" + ) + valid = _complete_canonical_report(dataset, config) + + estimated = deepcopy(valid) + estimated["records"][0]["context_token_method"] = "deterministic_estimate" + estimated["metrics"]["fixed_budget_curve"]["rows"][0]["records"][0][ + "context_token_method" + ] = "deterministic_estimate" + errors = validate_report(estimated, canonical=True) + assert any( + "context_token_method=pinned_reader_content_tokenizer" in error + for error in errors + ) + assert any( + "fixed-budget curve 256 records require" in error + and "context_token_method=pinned_reader_content_tokenizer" in error + for error in errors + ) + + mismatched = deepcopy(valid) + mismatched["records"][0]["context_tokenizer_identity"] = "other/model@" + "e" * 40 + mismatched["records"][0]["usage"]["token_counter"] = "other/model@" + "e" * 40 + errors = validate_report(mismatched, canonical=True) + assert any("context_tokenizer_identity must match" in error for error in errors) + assert any("usage.token_counter must match" in error for error in errors) + + +def test_benchmark_cli_writes_canonical_json_and_checksum(tmp_path, capsys): + dataset = tmp_path / "fixture.jsonl" + dataset.write_text('{"id":"one"}\n', encoding="utf-8") + report = report_envelope( + suite="fixture", dataset_path=dataset, config={"k": 5}, + records=[question_record("q1")], git_commit="abc123", + ) + source = tmp_path / "source.json" + source.write_text(json.dumps(report), encoding="utf-8") + artifact = tmp_path / "artifact.json" + assert main(["--input", str(source), "--output", str(artifact)]) == 0 + assert artifact.exists() and artifact.with_name("artifact.json.sha256").exists() + assert "sha256" in capsys.readouterr().out + + +def test_exact_tokenizer_fallback_budget_curves_and_deterministic_cis(): + assert count_tokens("abc", CharacterTokenizer()) == {"tokens": 3, "method": "injected"} + assert count_tokens("one two")["method"] == "deterministic_estimate" + records = [ + {"category": "a", "supporting_ids": ["m1"], "chunks": [ + {"id": "m1", "tokens": 3}, {"id": "m2", "tokens": 3} + ]}, + {"category": "b", "supporting_ids": ["m2"], "chunks": [ + {"id": "m1", "tokens": 3}, {"id": "m2", "tokens": 3} + ]}, + ] + curve = fixed_budget_curve(records, [3, 6]) + assert curve[0]["recall"] == 0.5 + assert curve[1]["recall"] == 1.0 + def metric(rows): + return sum(row["value"] for row in rows) / len(rows) + ci_one = stratified_bootstrap_ci( + [{"category": "a", "value": 1.0}, {"category": "b", "value": 0.0}], + metric, iterations=40, seed=4, + ) + ci_two = stratified_bootstrap_ci( + [{"category": "a", "value": 1.0}, {"category": "b", "value": 0.0}], + metric, iterations=40, seed=4, + ) + assert ci_one == ci_two + paired = paired_bootstrap_ci([(1.0, 0.0), (0.0, 0.0)], iterations=40, seed=4) + assert paired["delta"] == 0.5 and paired["n"] == 2 diff --git a/tests/test_documentation_contracts.py b/tests/test_documentation_contracts.py index b7f3e17a..5ef087b3 100644 --- a/tests/test_documentation_contracts.py +++ b/tests/test_documentation_contracts.py @@ -1,312 +1,312 @@ -from __future__ import annotations - -import ast -import json -import re -import xml.etree.ElementTree as ET -from pathlib import Path - - -from engraphis.core.schema import SCHEMA_VERSION - - -ROOT = Path(__file__).resolve().parents[1] - - -def _read(path: str) -> str: - return (ROOT / path).read_text(encoding="utf-8") - - - -def test_readme_long_description_uses_no_repository_relative_targets() -> None: - readme = _read("README.md") - destinations = re.findall( - r"!?\[[^\]]*\]\(([^) ]+)|(?:href|src)=\"([^\"]+)\"", - readme, - ) - flattened = [markdown or html for markdown, html in destinations] - relative = [ - destination - for destination in flattened - if not destination.startswith(("#", "https://", "http://")) - ] - assert not relative - - image_targets = [ - destination - for destination in flattened - if destination.endswith((".png", ".svg")) - ] - assert image_targets - assert all( - target.startswith( - "https://raw.githubusercontent.com/Coding-Dev-Tools/engraphis/main/" - ) - or target.startswith("https://img.shields.io/") - for target in image_targets - ) - -def test_canonical_offline_gate_tracks_ci() -> None: - agents = _read("AGENTS.md") - claude = _read("CLAUDE.md") - workflow = _read(".github/workflows/ci.yml") - required = ( - "ruff check .", - "python scripts/check_commercial_manifest.py", - "python scripts/externalize_dashboard_assets.py", - "python -m pytest", - "python -m eval.harness --dataset eval/datasets/sample.jsonl --k 5", - "python -m eval.harness --dataset eval/datasets/codemem.jsonl --k 5", - "python -m eval.ablation", - "python -m eval.reinforcement", - "python -m eval.adversarial_memory_security", - "python -m eval.grounded", - "python -m eval.code_arm", - "pyright", - ) - - for command in required: - assert command in agents, f"AGENTS.md omits the canonical gate command: {command}" - assert command in workflow, f"CI omits the documented gate command: {command}" - - assert "Use the exact primary offline gate in `AGENTS.md` §1" in claude - assert "do not maintain a smaller duplicate here" in claude - - -def test_core_backend_imports_stay_behind_outer_composition_root() -> None: - violations: list[str] = [] - for path in sorted((ROOT / "engraphis" / "core").glob("*.py")): - tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path)) - for node in ast.walk(tree): - if not isinstance(node, ast.ImportFrom): - continue - module = node.module or "" - if module.startswith("engraphis.backends"): - violations.append(f"{path.relative_to(ROOT)} imports {module}") - assert not violations, violations - - factory = ast.parse(_read("engraphis/factory.py"), filename="engraphis/factory.py") - backend_modules = { - node.module - for node in ast.walk(factory) - if isinstance(node, ast.ImportFrom) - and (node.module or "").startswith("engraphis.backends") - } - assert backend_modules, "outer composition root no longer imports concrete backends" - package = _read("engraphis/__init__.py") - assert "configure_engine_factory(_default_memory_engine_factory)" in package - assert "create_memory_engine" in package - - for document in (_read("AGENTS.md"), _read("CLAUDE.md"), _read("README.md")): - normalized = " ".join(document.split()) - assert "engraphis/factory.py" in normalized - assert "outer composition root" in normalized - assert "core/engine.py" in normalized - - -def test_benchmark_text_alternatives_match_registered_fixture_boundary() -> None: - """The current image and its alt text expose only current registered boundaries.""" +from __future__ import annotations + +import ast +import json +import re +import xml.etree.ElementTree as ET +from pathlib import Path + + +from engraphis.core.schema import SCHEMA_VERSION + + +ROOT = Path(__file__).resolve().parents[1] + + +def _read(path: str) -> str: + return (ROOT / path).read_text(encoding="utf-8") + + + +def test_readme_long_description_uses_no_repository_relative_targets() -> None: + readme = _read("README.md") + destinations = re.findall( + r"!?\[[^\]]*\]\(([^) ]+)|(?:href|src)=\"([^\"]+)\"", + readme, + ) + flattened = [markdown or html for markdown, html in destinations] + relative = [ + destination + for destination in flattened + if not destination.startswith(("#", "https://", "http://")) + ] + assert not relative + + image_targets = [ + destination + for destination in flattened + if destination.endswith((".png", ".svg")) + ] + assert image_targets + assert all( + target.startswith( + "https://raw.githubusercontent.com/Coding-Dev-Tools/engraphis/main/" + ) + or target.startswith("https://img.shields.io/") + for target in image_targets + ) + +def test_canonical_offline_gate_tracks_ci() -> None: + agents = _read("AGENTS.md") + claude = _read("CLAUDE.md") + workflow = _read(".github/workflows/ci.yml") + required = ( + "ruff check .", + "python scripts/check_commercial_manifest.py", + "python scripts/externalize_dashboard_assets.py", + "python -m pytest", + "python -m eval.harness --dataset eval/datasets/sample.jsonl --k 5", + "python -m eval.harness --dataset eval/datasets/codemem.jsonl --k 5", + "python -m eval.ablation", + "python -m eval.reinforcement", + "python -m eval.adversarial_memory_security", + "python -m eval.grounded", + "python -m eval.code_arm", + "pyright", + ) + + for command in required: + assert command in agents, f"AGENTS.md omits the canonical gate command: {command}" + assert command in workflow, f"CI omits the documented gate command: {command}" + + assert "Use the exact primary offline gate in `AGENTS.md` §1" in claude + assert "do not maintain a smaller duplicate here" in claude + + +def test_core_backend_imports_stay_behind_outer_composition_root() -> None: + violations: list[str] = [] + for path in sorted((ROOT / "engraphis" / "core").glob("*.py")): + tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path)) + for node in ast.walk(tree): + if not isinstance(node, ast.ImportFrom): + continue + module = node.module or "" + if module.startswith("engraphis.backends"): + violations.append(f"{path.relative_to(ROOT)} imports {module}") + assert not violations, violations + + factory = ast.parse(_read("engraphis/factory.py"), filename="engraphis/factory.py") + backend_modules = { + node.module + for node in ast.walk(factory) + if isinstance(node, ast.ImportFrom) + and (node.module or "").startswith("engraphis.backends") + } + assert backend_modules, "outer composition root no longer imports concrete backends" + package = _read("engraphis/__init__.py") + assert "configure_engine_factory(_default_memory_engine_factory)" in package + assert "create_memory_engine" in package + + for document in (_read("AGENTS.md"), _read("CLAUDE.md"), _read("README.md")): + normalized = " ".join(document.split()) + assert "engraphis/factory.py" in normalized + assert "outer composition root" in normalized + assert "core/engine.py" in normalized + + +def test_benchmark_text_alternatives_match_registered_fixture_boundary() -> None: + """The current image and its alt text expose only current registered boundaries.""" registry = json.loads(_read("docs/benchmark-evidence/offline-fixtures-v76.json")) - measurements = {run["id"]: run["result"] for run in registry["runs"]} - payload = measurements["offline-performance"] - readme = _read("README.md") - svg_text = _read("docs/images/context-efficiency.svg") - svg_root = ET.fromstring(svg_text) - namespace = {"svg": "http://www.w3.org/2000/svg"} - description_node = svg_root.find("svg:desc", namespace) - assert description_node is not None - description = " ".join("".join(description_node.itertext()).lower().split()) - image = re.search( - r']+context-efficiency\.svg[^>]+alt="([^"]+)"', - readme, - flags=re.IGNORECASE, - ) - assert image is not None - alternative = " ".join(image.group(1).lower().split()) - - assert "registered deterministic fixtures" in alternative - assert "structure-aware chunks reduce retrieved context" in alternative - assert "retrieved-candidate quality is labeled separately" in alternative - assert "packed-context quality" in alternative - assert "both measured in the selected report" in alternative - assert "actual mcp transport and provider billing are not measured" in alternative - assert "740.3 to 214.3 tokens" in alternative - assert "162.2 to 42.4 tokens" in alternative - assert ( - f"{payload['compact_serialized_payload_tokens']:,} rather than " - f"{payload['full_serialized_payload_tokens']:,} tokens" - ) in alternative - - for evidence in ( - "artifact-driven local deterministic benchmark report", - "structure-aware chunks report 740.3 to 214.3 retrieved tokens per question", - "retrieved-candidate quality and packed-context quality are separate views", + measurements = {run["id"]: run["result"] for run in registry["runs"]} + payload = measurements["offline-performance"] + readme = _read("README.md") + svg_text = _read("docs/images/context-efficiency.svg") + svg_root = ET.fromstring(svg_text) + namespace = {"svg": "http://www.w3.org/2000/svg"} + description_node = svg_root.find("svg:desc", namespace) + assert description_node is not None + description = " ".join("".join(description_node.itertext()).lower().split()) + image = re.search( + r']+context-efficiency\.svg[^>]+alt="([^"]+)"', + readme, + flags=re.IGNORECASE, + ) + assert image is not None + alternative = " ".join(image.group(1).lower().split()) + + assert "registered deterministic fixtures" in alternative + assert "structure-aware chunks reduce retrieved context" in alternative + assert "retrieved-candidate quality is labeled separately" in alternative + assert "packed-context quality" in alternative + assert "both measured in the selected report" in alternative + assert "actual mcp transport and provider billing are not measured" in alternative + assert "740.3 to 214.3 tokens" in alternative + assert "162.2 to 42.4 tokens" in alternative + assert ( + f"{payload['compact_serialized_payload_tokens']:,} rather than " + f"{payload['full_serialized_payload_tokens']:,} tokens" + ) in alternative + + for evidence in ( + "artifact-driven local deterministic benchmark report", + "structure-aware chunks report 740.3 to 214.3 retrieved tokens per question", + "retrieved-candidate quality and packed-context quality are separate views", f"{payload['full_serialized_payload_tokens']:,} full-proxy versus " f"{payload['compact_serialized_payload_tokens']:,} compact-proxy tokens", - "not an mcp transport measurement", - "does not measure provider billing", - "1,500-token cap", - ): - assert evidence in description - - for unsupported in ("unpinned", "noncanonical", "leaderboard"): - assert unsupported not in alternative - assert unsupported not in description - - for retired in ("local locomo diagnostic", "3 of 15 queries", "0 of 3 to 3 of 3"): - assert retired not in alternative - assert retired not in description - - -def test_official_longmemeval_runbook_tracks_attested_evidence_contract() -> None: - benchmarks = _read("BENCHMARKS.md") - runbook = _read("docs/PUBLIC_BENCHMARK_RUNBOOK.md") - normalized_benchmarks = " ".join(benchmarks.split()) - normalized_runbook = " ".join(runbook.split()) - - for value in ( - "balanced", - "planner", - "episodic_cap_2", - "planner_episodic_cap_2", - "context_k_2", - "planner_context_k_2", - ): - assert value in runbook - assert "30 official runs" in runbook - assert "six declared variants at all five token budgets" in normalized_benchmarks - assert "context_k=2" in runbook - - for option in ( - "--engraphis-execution-manifest", - "--engraphis-per-question", - "--engraphis-questions", - "--engraphis-haystack", - "--engraphis-trajectories", - "--engraphis-memory-config", - "--engraphis-matrix-manifest", - "--engraphis-seed", - "--execution-manifest", - "--claims-input", - ): - assert option in runbook - assert "set equality between every source question ID and output question ID" in runbook - assert "only after a successful return" in normalized_benchmarks.lower() - assert "inserted and retrieved counts by memory type" in normalized_runbook - assert "at least two inserted memory types" in normalized_runbook - - assert "does not publish per-record content fingerprints" in normalized_benchmarks - assert "whole-input/source-file digests" in normalized_runbook - assert "no raw questions, answers, prompts, context" in normalized_runbook - assert "no per-record content hashes or fingerprints" in normalized_runbook - - -def test_scope_and_event_guidance_match_fail_closed_runtime_contract() -> None: - readme = _read("README.md") - skill = _read("skills/engraphis-memory/SKILL.md") - scoping = _read("skills/engraphis-memory/references/SCOPING.md") - conventions = _read("skills/engraphis-memory/references/CONVENTIONS.md") - tools = _read("skills/engraphis-memory/references/TOOLS.md") - kilo = _read("docs/KILO_CODE_INTEGRATION.md") - - for document in (readme, skill, scoping, tools, kilo): - normalized = " ".join(document.split()) - assert "reserved and rejected" in normalized - assert "owner identity" in normalized - - for document in (conventions, tools): - normalized = " ".join(document.lower().split()) - assert "event rows are not memories" in normalized - assert "not recalled" in normalized - assert "not" in normalized and "consolidated" in normalized - - assert 'mtype="episodic"' in conventions - assert "≤0.2" in conventions - - -def test_configuration_and_recovery_guidance_matches_public_contracts() -> None: - readme = _read("README.md") - security = _read("SECURITY.md") - connect = _read("docs/AGENT_CONNECT.md") - providers = _read("docs/LLM_PROVIDERS.md") - recovery = _read("docs/RECALL_RECOVERY.md") - sync = _read("docs/SYNC.md") - - for document in (readme, security, connect, providers, sync): - normalized = " ".join(document.split()) - assert "~/.engraphis/config.env" in normalized - assert "ENGRAPHIS_ENV_FILE" in normalized - assert re.search(r"(?:never|does not) search(?:es)? the working directory", normalized) - - assert "repaired_fields" in recovery - assert "v1_memory_id" in recovery - assert "v1_thought_id" in recovery - assert "v1_document_id" in recovery - assert "first contact" in sync - assert "incomplete" in sync - assert "unanchored" in sync - assert "--relay-token" in sync and "--relay-e2ee-key" in sync - assert "intentionally has no secret-valued" in sync - - - -def test_schema_and_erasure_docs_match_live_export_policy() -> None: - agents = _read("AGENTS.md") - readme = _read("README.md") - changelog = _read("CHANGELOG.md") - sync = _read("docs/SYNC.md") - erasure = _read("docs/SECURE_ERASURE.md") - schema = _read("engraphis/core/schema.py") - - assert f"SCHEMA_VERSION = {SCHEMA_VERSION}" in schema - assert agents.count(f"`SCHEMA_VERSION = {SCHEMA_VERSION}`") == 2 - assert f"schema {SCHEMA_VERSION}" in readme - assert f"schema {SCHEMA_VERSION}" in changelog - - for document in (agents, readme, changelog, sync, erasure): - normalized = " ".join(document.split()) - assert "never_export" in normalized - assert "remote_erasure" in normalized - - normalized_sync = " ".join(sync.split()) - assert "only `remote_erasure`" in normalized_sync - assert "never leave the device" in normalized_sync - assert "cannot later be upgraded" in normalized_sync - assert "only a non-secret workspace/repo record" in erasure - - -def test_document_import_docs_describe_the_source_neutral_contract() -> None: - readme = _read("README.md") - agents = _read("AGENTS.md") - guide = _read("docs/DOCUMENT_IMPORT.md") - obsidian = _read("docs/OBSIDIAN_IMPORT.md") - - for document in (readme, guide): - assert "engraphis import documents" in document - assert "--dry-run" in document - assert "--yes" in document - for format_name in ( - "Markdown", "reStructuredText", "HTML", "JSON", "CSV", "DOCX", "ODT", - "RTF", "XLSX", "ODS", "PPTX", "ODP", "EPUB", "Source code", - ): - assert format_name in guide - for safety_term in ("symlink", "secret", "unsupported", "resumable", "temporal", "conflict"): - assert safety_term in guide - assert f"SCHEMA_VERSION = {SCHEMA_VERSION}" in agents - assert "source-neutral" in agents - assert "rich Markdown adapter" in obsidian - assert "DOCUMENT_IMPORT.md" in obsidian - - -def test_consolidation_docs_expose_only_live_public_options() -> None: - readme = _read("README.md") - tools = _read("skills/engraphis-memory/references/TOOLS.md") - changelog = _read("CHANGELOG.md") - - for document in (readme, tools, changelog): - assert "supersede_sources" not in document - assert "supersede-sources" not in document - - assert "source episodes remain live" in readme - normalized_tools = " ".join(tools.split()) - assert "`profiles (bool, false)`; `structured (bool, false)`." in normalized_tools + "not an mcp transport measurement", + "does not measure provider billing", + "1,500-token cap", + ): + assert evidence in description + + for unsupported in ("unpinned", "noncanonical", "leaderboard"): + assert unsupported not in alternative + assert unsupported not in description + + for retired in ("local locomo diagnostic", "3 of 15 queries", "0 of 3 to 3 of 3"): + assert retired not in alternative + assert retired not in description + + +def test_official_longmemeval_runbook_tracks_attested_evidence_contract() -> None: + benchmarks = _read("BENCHMARKS.md") + runbook = _read("docs/PUBLIC_BENCHMARK_RUNBOOK.md") + normalized_benchmarks = " ".join(benchmarks.split()) + normalized_runbook = " ".join(runbook.split()) + + for value in ( + "balanced", + "planner", + "episodic_cap_2", + "planner_episodic_cap_2", + "context_k_2", + "planner_context_k_2", + ): + assert value in runbook + assert "30 official runs" in runbook + assert "six declared variants at all five token budgets" in normalized_benchmarks + assert "context_k=2" in runbook + + for option in ( + "--engraphis-execution-manifest", + "--engraphis-per-question", + "--engraphis-questions", + "--engraphis-haystack", + "--engraphis-trajectories", + "--engraphis-memory-config", + "--engraphis-matrix-manifest", + "--engraphis-seed", + "--execution-manifest", + "--claims-input", + ): + assert option in runbook + assert "set equality between every source question ID and output question ID" in runbook + assert "only after a successful return" in normalized_benchmarks.lower() + assert "inserted and retrieved counts by memory type" in normalized_runbook + assert "at least two inserted memory types" in normalized_runbook + + assert "does not publish per-record content fingerprints" in normalized_benchmarks + assert "whole-input/source-file digests" in normalized_runbook + assert "no raw questions, answers, prompts, context" in normalized_runbook + assert "no per-record content hashes or fingerprints" in normalized_runbook + + +def test_scope_and_event_guidance_match_fail_closed_runtime_contract() -> None: + readme = _read("README.md") + skill = _read("skills/engraphis-memory/SKILL.md") + scoping = _read("skills/engraphis-memory/references/SCOPING.md") + conventions = _read("skills/engraphis-memory/references/CONVENTIONS.md") + tools = _read("skills/engraphis-memory/references/TOOLS.md") + kilo = _read("docs/KILO_CODE_INTEGRATION.md") + + for document in (readme, skill, scoping, tools, kilo): + normalized = " ".join(document.split()) + assert "reserved and rejected" in normalized + assert "owner identity" in normalized + + for document in (conventions, tools): + normalized = " ".join(document.lower().split()) + assert "event rows are not memories" in normalized + assert "not recalled" in normalized + assert "not" in normalized and "consolidated" in normalized + + assert 'mtype="episodic"' in conventions + assert "≤0.2" in conventions + + +def test_configuration_and_recovery_guidance_matches_public_contracts() -> None: + readme = _read("README.md") + security = _read("SECURITY.md") + connect = _read("docs/AGENT_CONNECT.md") + providers = _read("docs/LLM_PROVIDERS.md") + recovery = _read("docs/RECALL_RECOVERY.md") + sync = _read("docs/SYNC.md") + + for document in (readme, security, connect, providers, sync): + normalized = " ".join(document.split()) + assert "~/.engraphis/config.env" in normalized + assert "ENGRAPHIS_ENV_FILE" in normalized + assert re.search(r"(?:never|does not) search(?:es)? the working directory", normalized) + + assert "repaired_fields" in recovery + assert "v1_memory_id" in recovery + assert "v1_thought_id" in recovery + assert "v1_document_id" in recovery + assert "first contact" in sync + assert "incomplete" in sync + assert "unanchored" in sync + assert "--relay-token" in sync and "--relay-e2ee-key" in sync + assert "intentionally has no secret-valued" in sync + + + +def test_schema_and_erasure_docs_match_live_export_policy() -> None: + agents = _read("AGENTS.md") + readme = _read("README.md") + changelog = _read("CHANGELOG.md") + sync = _read("docs/SYNC.md") + erasure = _read("docs/SECURE_ERASURE.md") + schema = _read("engraphis/core/schema.py") + + assert f"SCHEMA_VERSION = {SCHEMA_VERSION}" in schema + assert agents.count(f"`SCHEMA_VERSION = {SCHEMA_VERSION}`") == 2 + assert f"schema {SCHEMA_VERSION}" in readme + assert f"schema {SCHEMA_VERSION}" in changelog + + for document in (agents, readme, changelog, sync, erasure): + normalized = " ".join(document.split()) + assert "never_export" in normalized + assert "remote_erasure" in normalized + + normalized_sync = " ".join(sync.split()) + assert "only `remote_erasure`" in normalized_sync + assert "never leave the device" in normalized_sync + assert "cannot later be upgraded" in normalized_sync + assert "only a non-secret workspace/repo record" in erasure + + +def test_document_import_docs_describe_the_source_neutral_contract() -> None: + readme = _read("README.md") + agents = _read("AGENTS.md") + guide = _read("docs/DOCUMENT_IMPORT.md") + obsidian = _read("docs/OBSIDIAN_IMPORT.md") + + for document in (readme, guide): + assert "engraphis import documents" in document + assert "--dry-run" in document + assert "--yes" in document + for format_name in ( + "Markdown", "reStructuredText", "HTML", "JSON", "CSV", "DOCX", "ODT", + "RTF", "XLSX", "ODS", "PPTX", "ODP", "EPUB", "Source code", + ): + assert format_name in guide + for safety_term in ("symlink", "secret", "unsupported", "resumable", "temporal", "conflict"): + assert safety_term in guide + assert f"SCHEMA_VERSION = {SCHEMA_VERSION}" in agents + assert "source-neutral" in agents + assert "rich Markdown adapter" in obsidian + assert "DOCUMENT_IMPORT.md" in obsidian + + +def test_consolidation_docs_expose_only_live_public_options() -> None: + readme = _read("README.md") + tools = _read("skills/engraphis-memory/references/TOOLS.md") + changelog = _read("CHANGELOG.md") + + for document in (readme, tools, changelog): + assert "supersede_sources" not in document + assert "supersede-sources" not in document + + assert "source episodes remain live" in readme + normalized_tools = " ".join(tools.split()) + assert "`profiles (bool, false)`; `structured (bool, false)`." in normalized_tools From 6799ace738d1a003a449c3ace0a94b00edfafb4b Mon Sep 17 00:00:00 2001 From: Coding-Dev-Tools Date: Sun, 27 Sep 2026 02:37:58 -0400 Subject: [PATCH 6/6] fix(release): distinguish repair dispatcher from authorizer --- .github/workflows/release.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 0ffe56d6..a5e08f08 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -977,7 +977,7 @@ jobs: { printf '# Release qualification waiver\n\n' printf 'Tag: `%s`\n' "$RELEASE_TAG" - printf 'Authorized by: `%s`\n' "$GH_ACTOR" + printf 'Triggered by: `%s`\n' "$GH_ACTOR" printf 'Workflow run: %s\n\n' "$GH_RUN_URL" printf 'The owner explicitly waived full-product qualification for this repair. No unverified release gate is represented as passed.\n' } >> "$GITHUB_STEP_SUMMARY"