diff --git a/hf_space/README.md b/hf_space/README.md index 9967a20..7dcef70 100644 --- a/hf_space/README.md +++ b/hf_space/README.md @@ -1,38 +1,54 @@ --- -title: CompText Prompt Compression Lab -emoji: πŸ—œοΈ -colorFrom: blue -colorTo: purple +title: CompText Universe +emoji: 🧭 +colorFrom: indigo +colorTo: blue sdk: gradio +python_version: 3.10.13 sdk_version: 5.44.1 app_file: app.py +fullWidth: true +header: mini pinned: false license: apache-2.0 +short_description: Explore CompText architecture, contracts and safe context compression. +models: + - microsoft/llmlingua-2-bert-base-multilingual-cased-meetingbank +tags: + - context-engineering + - prompt-compression + - software-engineering + - gradio +preload_from_hub: + - microsoft/llmlingua-2-bert-base-multilingual-cased-meetingbank --- -# CompText Prompt Compression Lab +# CompText Universe -A CPU-friendly Hugging Face Space for evaluating prompt and context compression with Microsoft LLMLingua-2. +> Models are providers. Context is the product. Evidence is the trust layer. CompText is the kernel. -## Features +A public, experimental showcase for the CompText local engineering-orchestration architecture. The real CompText runtime remains local; this Space visualizes architecture and contracts, builds non-executable AIR and simulated Evidence previews, and tests fail-closed context compression. -- Compress arbitrary prompts at configurable retention rates -- Compare original and compressed text -- Measure token reduction and runtime -- Check preservation of negations, CLI flags, file paths, JSON keys, version numbers, and code-like symbols -- Run a built-in benchmark suite -- Export results as JSON +## Included surfaces -## Default model +- Static seven-layer architecture explorer +- Explicit capability maturity matrix +- Workspace skill and safety-boundary explorer +- Fail-closed hybrid LLMLingua-2 compression +- Deterministic, non-executable AIR previews +- Simulated, non-persistent Evidence previews +- CompText-specific benchmark and JSON exports -`microsoft/llmlingua-2-bert-base-multilingual-cased-meetingbank` +## Hard boundaries -The first startup can take several minutes because the model must be downloaded. +- No provider calls +- No repository writes +- No runtime GitHub access +- No API keys or runtime secrets +- No AIR execution +- No persistent prompt storage +- No claim that planned or scaffolded components are production-ready -## Hardware +## Runtime -Designed for Hugging Face Spaces `CPU Basic` (2 vCPU, 16 GB RAM, 50 GB ephemeral storage). - -## Safety - -This Space does not call external LLM APIs, require runtime secrets, access private repositories, or modify repositories. \ No newline at end of file +Only compression callbacks request ZeroGPU. Universe navigation, preview construction, contract views and secret-pattern checks are CPU-only and deterministic. The model is preloaded from the Hub during build to reduce first-request latency. diff --git a/hf_space/app.py b/hf_space/app.py index 679c539..9d66caa 100644 --- a/hf_space/app.py +++ b/hf_space/app.py @@ -10,18 +10,21 @@ from compression import DEFAULT_MODEL, compress_text from metrics import summarize_checks +from previews import build_air_preview, build_evidence_preview, scan_secrets from safety_checks import run_safety_checks +from universe import capabilities_frame, contracts_frame, layers_frame, load_universe, overview_markdown, skills_frame ROOT = Path(__file__).parent BENCHMARK_CASES = json.loads((ROOT / "benchmark_cases.json").read_text(encoding="utf-8")) +UNIVERSE = load_universe() EXAMPLE_TEXT = ( - "Γ„ndere AGENTS.md nicht und arbeite nur mit --dry-run. " - "CompText contains a detailed orchestration layer that coordinates providers, context preparation, policy checks, and output validation. " - "The architecture explanation is intentionally verbose so that the safe hybrid compressor has natural-language prose to reduce while preserving every technical constraint unchanged." + "FΓΌhre keine Live-Provider-Aufrufe aus. Γ„ndere AGENTS.md nicht. " + "Analysiere ausfΓΌhrlich, wie CompText Context Packs fΓΌr lokale Softwareentwicklung vorbereitet. " + "Nutze modules/cli/cli_entrypoint.py nur lesend mit --dry-run und gib einen JSON-Bericht zurΓΌck." ) -def _export(payload: object) -> str: +def _json_export(payload: object) -> str: with NamedTemporaryFile("w", encoding="utf-8", suffix=".json", delete=False) as handle: json.dump(payload, handle, ensure_ascii=False, indent=2) return handle.name @@ -29,32 +32,55 @@ def _export(payload: object) -> str: @spaces.GPU def compress_ui(text: str, retention_percent: int): + secret_matches = scan_secrets(text) + if secret_matches: + payload = { + "decision": "blocked", + "reason": "Potential secret material detected. Input was not processed.", + "secret_matches": secret_matches, + } + return text, payload, pd.DataFrame(), json.dumps(payload, ensure_ascii=False, indent=2), None + result = compress_text(text, retention_rate=retention_percent / 100) checks = run_safety_checks(result.original_text, result.candidate_text) summary = summarize_checks(checks) + decision = "accepted" if result.accepted else "fallback" metrics = { - "Decision": "ACCEPTED" if result.accepted else "FALLBACK TO ORIGINAL", - "Fallback reason": result.fallback_reason or "", + "Decision": decision, "Original tokens": result.origin_tokens, "Output tokens": result.compressed_tokens, - "Net token reduction": f"{result.token_reduction_percent}%", + "Net reduction": f"{result.token_reduction_percent}%", "Protected segments": result.protected_segments, "Compressed segments": result.compressed_segments, + "Candidate safety": f"{summary['score_percent']}%", "Runtime": f"{result.runtime_seconds}s", - "Candidate safety": f"{summary['passed']}/{summary['total']}", + "Reason": result.fallback_reason or "", } rows = [ { - "Check": c.name, - "Relevant": "Yes" if c.relevant else "No", - "Passed": "Yes" if c.passed else "No", - "Expected": ", ".join(c.expected), - "Missing": ", ".join(c.missing), + "Check": check.name, + "Relevant": "Yes" if check.relevant else "No", + "Passed": "Yes" if check.passed else "No", + "Expected": ", ".join(check.expected), + "Missing": ", ".join(check.missing), } - for c in checks + for check in checks ] - payload = {"compression": result.to_dict(), "candidate_safety": summary} - return result.compressed_text, metrics, pd.DataFrame(rows), json.dumps(payload, ensure_ascii=False, indent=2), _export(payload) + payload = { + "decision": decision, + "compression": result.to_dict(), + "safety": summary, + "air_preview": build_air_preview(text, {"decision": decision, "metrics": metrics}), + "evidence_preview": build_evidence_preview(text, result.compressed_text, decision, metrics), + } + return result.compressed_text, metrics, pd.DataFrame(rows), json.dumps(payload, ensure_ascii=False, indent=2), _json_export(payload) + + +def preview_contracts(text: str): + matches = scan_secrets(text) + air = build_air_preview(text) + evidence = build_evidence_preview(text, text, "preview_only", {"secret_matches": matches}) + return air, evidence @spaces.GPU @@ -80,39 +106,92 @@ def run_benchmark(retention_percent: int): }) except Exception as exc: rows.append({"Case": case["name"], "Category": case["category"], "Decision": "error", "Reason": str(exc)}) - return pd.DataFrame(rows), _export(rows) + return pd.DataFrame(rows), _json_export(rows) -with gr.Blocks(title="CompText Prompt Compression Lab") as demo: +with gr.Blocks(title="CompText Universe") as demo: + gr.Markdown(overview_markdown(UNIVERSE)) gr.Markdown( - "# πŸ—œοΈ CompText Safe Hybrid Compression Lab\n" - "Critical instructions, flags, paths, JSON, versions, and code symbols are protected. " - "Only natural-language segments are compressed. Unsafe or ineffective candidates automatically fall back to the original text." + "**Experimental public showcase:** no provider calls, no repository writes, no AIR execution, " + "no persistent prompt storage. The local CompText kernel remains the product runtime." ) - with gr.Tab("Single prompt"): - input_text = gr.Textbox(label="Original prompt", value=EXAMPLE_TEXT, lines=14) + + with gr.Tab("Overview"): + with gr.Row(): + gr.Markdown( + "## Context is the product\n" + "CompText treats models as interchangeable providers while context, contracts and Evidence " + "form the durable engineering system. This Space visualizes that model without executing it." + ) + gr.JSON(value={ + "product_status": UNIVERSE["product"]["status"], + "snapshot_version": UNIVERSE["snapshot_version"], + "runtime_github_access": UNIVERSE["source"]["runtime_github_access"], + "compression_model": DEFAULT_MODEL, + }, label="Snapshot identity") + gr.Dataframe(value=contracts_frame(UNIVERSE), label="Core contracts", interactive=False) + + with gr.Tab("Architecture"): + gr.Markdown( + "## Seven-layer model\n" + "`User β†’ Terminal OS / UI β†’ Runtime / Gateway / Agent Bus β†’ AIR / Evidence / Memory β†’ Provider Router`" + ) + gr.Dataframe(value=layers_frame(UNIVERSE), label="Architecture layers", interactive=False) + + with gr.Tab("Capabilities"): + gr.Markdown("## What exists, what is experimental, and what remains future") + gr.Dataframe(value=capabilities_frame(UNIVERSE), label="Capability matrix", interactive=False) + + with gr.Tab("Compression Lab"): + gr.Markdown( + "## Fail-closed hybrid compression\n" + "Critical instructions, flags, paths, structured data and code-like symbols are protected. " + "Unsafe or unhelpful candidates return the exact original input." + ) + input_text = gr.Textbox(label="Engineering context", value=EXAMPLE_TEXT, lines=14) retention = gr.Slider(10, 100, value=60, step=5, label="Retention rate (%)") - button = gr.Button("Compress safely", variant="primary") - compressed = gr.Textbox(label="Safe output", lines=14) - metrics = gr.JSON(label="Decision and metrics") - checks = gr.Dataframe(label="Candidate safety checks", interactive=False) - raw = gr.Code(label="Raw result", language="json") - download = gr.File(label="Download JSON result") - button.click(compress_ui, [input_text, retention], [compressed, metrics, checks, raw, download]) - with gr.Tab("Benchmark suite"): + compress_button = gr.Button("Analyze and compress", variant="primary") + compressed = gr.Textbox(label="Final output", lines=14) + compression_metrics = gr.JSON(label="Decision and metrics") + checks = gr.Dataframe(label="Relevant preservation checks", interactive=False) + raw = gr.Code(label="Compression + AIR + Evidence payload", language="json") + download = gr.File(label="Download JSON") + compress_button.click(compress_ui, [input_text, retention], [compressed, compression_metrics, checks, raw, download]) + + with gr.Tab("AIR & Evidence"): + gr.Markdown( + "## Non-executable contract previews\n" + "AIR describes intended work. Evidence describes observed work. Here, AIR is always disabled and " + "Evidence is always marked as simulated." + ) + preview_text = gr.Textbox(label="Engineering task", value=EXAMPLE_TEXT, lines=10) + preview_button = gr.Button("Build previews") + air_output = gr.JSON(label="AIR preview") + evidence_output = gr.JSON(label="Simulated Evidence preview") + preview_button.click(preview_contracts, preview_text, [air_output, evidence_output]) + + with gr.Tab("Skills"): + gr.Markdown("## Skill-grounded local workflow and safety boundaries") + gr.Dataframe(value=skills_frame(UNIVERSE), label="Workspace skills", interactive=False) + + with gr.Tab("Benchmarks"): + gr.Markdown("## CompText-specific protected-context benchmark") bench_rate = gr.Slider(10, 100, value=60, step=5, label="Retention rate (%)") - bench_button = gr.Button("Run safe hybrid benchmark", variant="primary") + bench_button = gr.Button("Run benchmark", variant="primary") bench_table = gr.Dataframe(label="Benchmark results", interactive=False) bench_download = gr.File(label="Download benchmark JSON") bench_button.click(run_benchmark, bench_rate, [bench_table, bench_download]) - with gr.Accordion("Decision policy", open=False): + + with gr.Accordion("Model, provenance and limitations", open=False): + gr.JSON(value=UNIVERSE["source"], label="Static snapshot provenance") gr.Markdown( - f"**Model:** `{DEFAULT_MODEL}`\n\n" - "A result is accepted only when all relevant protected elements survive and net token reduction is at least 10%. " - "Otherwise the exact original input is returned." + f"**Compression model:** `{DEFAULT_MODEL}`\n\n" + "The Universe snapshot is committed data, not a live GitHub view. Safety checks are deterministic " + "preservation heuristics, not a semantic equivalence proof." ) + demo.queue(default_concurrency_limit=1, max_size=8) if __name__ == "__main__": - demo.launch() + demo.launch(ssr_mode=False) diff --git a/hf_space/data/universe_snapshot.json b/hf_space/data/universe_snapshot.json new file mode 100644 index 0000000..565d260 --- /dev/null +++ b/hf_space/data/universe_snapshot.json @@ -0,0 +1,65 @@ +{ + "snapshot_version": "1.0", + "product": { + "name": "CompText", + "repository_name": "comptext", + "status": "local-dry-run-mvp", + "claim": "Models are providers. Context is the product. Evidence is the trust layer. CompText is the kernel.", + "description": "A local AI orchestration platform for software engineering. This public Space is an experimental, non-executing showcase." + }, + "source": { + "repository": "ProfRandom92/Comptext", + "mode": "committed-static-snapshot", + "runtime_github_access": false, + "allowed_sources": [ + "AGENTS.md", + "docs/COMPTEXT_ARCHITECTURE_v1.md", + ".agents/skills/comptext-local-autonomy/SKILL.md", + ".agents/skills/comptext-local-verify/SKILL.md", + ".agents/skills/comptext-workspace-validation/SKILL.md", + ".agents/skills/comptext-status/SKILL.md", + ".agents/skills/workspace-state/SKILL.md" + ] + }, + "boundaries": [ + "No provider calls", + "No repository writes", + "No runtime GitHub access", + "No secrets or environment access", + "No AIR execution", + "No persistent user input storage" + ], + "layers": [ + {"id":"ui","name":"Terminal OS / UI","status":"scaffolded","purpose":"Human-facing workbench for sessions, workspaces, commands, providers, evidence and run queues.","inputs":"user intent","outputs":"normalized commands and views"}, + {"id":"runtime","name":"Runtime","status":"scaffolded","purpose":"Coordinates runs, plans, verification, retries, queues and replay.","inputs":"AIR plans","outputs":"run state and events"}, + {"id":"gateway","name":"Gateway","status":"planned","purpose":"Normalizes local provider-compatible message and response routes.","inputs":"local API requests","outputs":"normalized traffic"}, + {"id":"agent-bus","name":"Agent Bus","status":"planned","purpose":"Coordinates specialized agents as explicit tasks with approval gates.","inputs":"tasks and roles","outputs":"task results and evidence"}, + {"id":"air","name":"AIR","status":"scaffolded","purpose":"Describes intended work before execution: goal, context, tools, constraints, permissions and outputs.","inputs":"intent and context","outputs":"non-executed plan contract"}, + {"id":"evidence","name":"Evidence","status":"scaffolded","purpose":"Records what actually happened without secrets, raw provider payloads or hidden reasoning.","inputs":"runtime and validation events","outputs":"redacted evidence events"}, + {"id":"memory","name":"Memory / Knowledge Graph","status":"future","purpose":"Structured workspace knowledge spanning files, functions, tests, runs and evidence.","inputs":"validated workspace state","outputs":"retrievable context relationships"} + ], + "capabilities": [ + {"name":"Local status","surface":"CLI","status":"implemented","network":"no","provider":"no","mutating":"no"}, + {"name":"Local verification","surface":"CLI","status":"implemented","network":"no","provider":"no","mutating":"no"}, + {"name":"Workspace schema validation","surface":"CLI","status":"implemented","network":"no","provider":"no","mutating":"no"}, + {"name":"Hybrid context compression","surface":"HF Space","status":"experimental","network":"model-cache","provider":"no","mutating":"no"}, + {"name":"AIR preview","surface":"HF Space","status":"experimental","network":"no","provider":"no","mutating":"no"}, + {"name":"Simulated Evidence preview","surface":"HF Space","status":"experimental","network":"no","provider":"no","mutating":"no"}, + {"name":"Provider Router","surface":"Kernel","status":"scaffolded","network":"disabled","provider":"not_configured","mutating":"no"}, + {"name":"Workspace reflection runtime","surface":"Kernel","status":"future","network":"no","provider":"no","mutating":"local"}, + {"name":"Autonomous PR merge","surface":"Agent workflow","status":"disabled","network":"yes","provider":"no","mutating":"yes"} + ], + "skills": [ + {"name":"comptext-local-autonomy","purpose":"One-unit-at-a-time offline development loop.","validation":"python -m pytest; git diff --check","boundary":"No network, providers, secrets, GitHub writes or servers."}, + {"name":"comptext-local-verify","purpose":"Verify status, subagents, workspace validation and doctor diagnostics.","validation":"comptext verify --dry-run","boundary":"Offline dry-run checks only."}, + {"name":"comptext-workspace-validation","purpose":"Validate committed workspace examples against strict schemas.","validation":"comptext validate workspace --dry-run","boundary":"No active generation, network, databases or providers."}, + {"name":"comptext-status","purpose":"Show file presence and local diagnostic state.","validation":"comptext status --dry-run","boundary":"Offline local-only data collection."}, + {"name":"workspace-state","purpose":"Define future snapshot, delta and reflection-gate concepts.","validation":"schema and fixture validation","boundary":"No active runtime or interpretability claims."} + ], + "contracts": [ + {"name":"AIR Plan","kind":"intent","fields":"version, intent, goal, context, files, tools, permissions, expected_outputs, metadata","execution":"never in this Space"}, + {"name":"Evidence Event","kind":"observed-event","fields":"event_id, run_id, type, actor, tool, summary, hashes, timestamp, redaction","execution":"simulated preview only"}, + {"name":"Run Record","kind":"run-index","fields":"run_id, air_hash, status, start_time, end_time, event_hashes, metrics","execution":"not created in this Space"}, + {"name":"Workspace Snapshot","kind":"workspace-state","fields":"schema-defined local state","execution":"static concept preview only"} + ] +} diff --git a/hf_space/docs/2026-07-16-comptext-universe-design.md b/hf_space/docs/2026-07-16-comptext-universe-design.md new file mode 100644 index 0000000..75d9d15 --- /dev/null +++ b/hf_space/docs/2026-07-16-comptext-universe-design.md @@ -0,0 +1,344 @@ +# CompText Universe + Compression Lab Design + +Date: 2026-07-16 +Status: Approved for implementation +Scope boundary: **Only `hf_space/**` may change.** + +## 1. Goal + +Turn the existing Hugging Face Space into a safe public CompText showcase and context-engineering laboratory without modifying or emulating the local CompText runtime. + +The Space has three roles: + +1. **Universe Explorer** β€” explain architecture, capabilities, contracts, skills, maturity, and safety boundaries. +2. **Context Engineering Lab** β€” preserve the existing fail-closed hybrid compression and add CompText-specific profiles and diagnostics. +3. **Contract Preview Sandbox** β€” generate deterministic, non-executable AIR and simulated Evidence previews. + +The Space is not a hosted CompText runtime. It performs no provider calls, repository writes, agent execution, MCP activity, or persistent evidence storage. + +## 2. Hard Scope Boundary + +Allowed changes: + +- `hf_space/**` + +Forbidden changes: + +- `.github/**` +- `modules/**` +- `schemas/**` +- `examples/**` +- `docs/**` outside `hf_space/docs/**` +- CLI, runtime, gateway, provider, TUI, plugin, skill, and workspace-state implementation + +Repository content outside `hf_space/**` may be read to create a manually curated static snapshot, but it must not be modified. + +## 3. Architecture + +```text +Curated CompText source knowledge + | + v +hf_space/data/*.json + | + v +Universe loader + validation + | + +--> Overview / Architecture / Capabilities / Skills / Contracts + +--> AIR Preview / Simulated Evidence Preview + +--> Compression Lab / Benchmarks +``` + +The deployed Space performs no live GitHub reads. All CompText knowledge used by the UI is shipped as static JSON under `hf_space/data/`. + +## 4. Modules + +```text +hf_space/ +β”œβ”€β”€ app.py +β”œβ”€β”€ README.md +β”œβ”€β”€ requirements.txt +β”œβ”€β”€ compression.py +β”œβ”€β”€ protected_segments.py +β”œβ”€β”€ safety_checks.py +β”œβ”€β”€ metrics.py +β”œβ”€β”€ benchmark_cases.json +β”œβ”€β”€ universe/ +β”‚ β”œβ”€β”€ __init__.py +β”‚ β”œβ”€β”€ loader.py +β”‚ β”œβ”€β”€ models.py +β”‚ └── validation.py +β”œβ”€β”€ previews/ +β”‚ β”œβ”€β”€ __init__.py +β”‚ β”œβ”€β”€ air_preview.py +β”‚ β”œβ”€β”€ evidence_preview.py +β”‚ └── redaction.py +β”œβ”€β”€ ui/ +β”‚ β”œβ”€β”€ __init__.py +β”‚ β”œβ”€β”€ overview.py +β”‚ β”œβ”€β”€ architecture.py +β”‚ β”œβ”€β”€ capabilities.py +β”‚ β”œβ”€β”€ compression_lab.py +β”‚ β”œβ”€β”€ previews.py +β”‚ └── contracts.py +β”œβ”€β”€ data/ +β”‚ β”œβ”€β”€ universe_snapshot.json +β”‚ β”œβ”€β”€ architecture_graph.json +β”‚ β”œβ”€β”€ capability_matrix.json +β”‚ β”œβ”€β”€ skills_catalog.json +β”‚ β”œβ”€β”€ contracts_catalog.json +β”‚ └── provenance_manifest.json +β”œβ”€β”€ tests/ +β”‚ β”œβ”€β”€ test_universe_data.py +β”‚ β”œβ”€β”€ test_air_preview.py +β”‚ β”œβ”€β”€ test_evidence_preview.py +β”‚ β”œβ”€β”€ test_redaction.py +β”‚ └── test_compression_profiles.py +└── docs/ + └── 2026-07-16-comptext-universe-design.md +``` + +`app.py` remains the composition root. Domain logic belongs in focused modules. + +## 5. Universe Data Model + +### 5.1 Product snapshot + +Required fields: + +- snapshot version +- product name +- maturity status (`local-dry-run-mvp`) +- core claim +- source repository +- source commit +- generated timestamp +- explicit limitations + +### 5.2 Architecture entities + +Each layer or service contains: + +- stable ID +- display name +- purpose +- maturity status +- inputs +- outputs +- relationships +- security notes +- source references + +Allowed maturity vocabulary: + +- `implemented` +- `scaffolded` +- `experimental` +- `planned` +- `future` +- `disabled` +- `not_configured` + +The UI must never imply that planned or scaffolded functionality is production-ready. + +### 5.3 Capability matrix + +Each capability records: + +- name +- surface +- status +- network requirement +- provider requirement +- mutation behavior +- approval requirement +- summary + +## 6. User Experience + +Tabs: + +1. **Overview** β€” product claim, maturity, boundaries, snapshot provenance. +2. **Architecture** β€” seven-layer navigation and relationships. +3. **Capabilities** β€” filterable capability matrix. +4. **Compression Lab** β€” protected hybrid compression with profiles. +5. **AIR Preview** β€” deterministic non-executable AIR-shaped output. +6. **Evidence Preview** β€” explicitly simulated evidence output. +7. **Contracts & Skills** β€” human-readable catalogs and validation expectations. +8. **Benchmarks** β€” CompText-specific safety and reduction results. +9. **About & Boundaries** β€” explicit non-goals and privacy behavior. + +The initial page must state: + +- experimental public demonstration +- no provider calls +- no repository writes +- no secrets +- no production-runtime claim + +## 7. Compression Lab + +### 7.1 Profiles + +- Natural prose +- Engineering task +- AIR context +- Evidence summary +- Workspace description +- CLI instruction +- Configuration +- Mixed technical context + +Profiles adjust protection rules and minimum useful reduction. They do not weaken the fail-closed safety requirement. + +### 7.2 Segment classes + +- `PROTECTED` +- `COMPRESSIBLE` +- `STRUCTURED` +- `UNSAFE_TO_TRANSFORM` + +Protected content includes negations, prohibitions, CLI flags, commands, paths, URLs, JSON/YAML/TOML-like blocks, environment variable names, versions, hashes, IDs, code symbols, schema fields, permission terms, approval boundaries, and CompText status vocabulary. + +### 7.3 Acceptance rule + +A compressed candidate is accepted only when: + +- all relevant protected items are preserved +- structure validation passes +- no secret pattern is detected +- net reduction meets the profile threshold + +Otherwise the exact original input is returned. + +### 7.4 Metrics + +- original tokens +- candidate tokens +- output tokens +- gross reduction +- net reduction +- protected segments +- compressed segments +- relevant safety checks +- candidate safety +- final safety +- decision +- fallback reason +- runtime +- model +- snapshot commit + +## 8. AIR Preview + +The AIR preview is deterministic and non-executable. It may extract: + +- intent +- goal +- context +- files +- tools mentioned +- constraints +- permissions +- expected outputs +- review and approval hints + +Every result includes: + +```json +{ + "execution": { + "enabled": false + }, + "preview": true +} +``` + +The preview must not call agents, tools, providers, GitHub, or the local runtime. + +## 9. Evidence Preview + +Evidence output is always labeled `simulation: true` and may contain: + +- synthetic event type +- actor `hf-space-demo` +- input and output hashes +- summary +- compression metrics +- redaction status + +It must never claim that a real CompText action occurred. + +## 10. Security and Privacy + +- no API keys or provider integrations +- no live GitHub connection +- no database +- no prompt persistence +- no raw input logging by application code +- temporary exports only +- deterministic secret-pattern blocking before compression or preview generation +- user input is treated only as data, never as executable instruction + +Secret detection covers common provider keys, GitHub and Hugging Face tokens, bearer tokens, private-key blocks, AWS-style credentials, and suspicious environment assignments. + +## 11. Hugging Face Runtime + +- Gradio Space +- ZeroGPU used only for LLMLingua compression functions +- Universe, previews, validation, hashing, and redaction remain CPU-only +- keep ZeroGPU duration conservative to avoid quota rejection +- preserve the currently compatible PyTorch version +- model loading remains lazy unless preload is proven stable +- Space metadata clearly lists the LLMLingua model and experimental status + +## 12. Tests and Validation + +All validation must be runnable from within `hf_space/**` without changing other repository files. + +Required checks: + +```bash +python -m compileall hf_space +python -m pytest hf_space/tests +``` + +Test requirements: + +- all shipped JSON files parse and satisfy internal validators +- all relationships reference existing IDs +- maturity values use the approved vocabulary +- AIR output is always non-executable +- Evidence output is always simulated +- secret-like inputs are blocked +- protected technical items survive accepted compression +- unsafe or unprofitable candidates return the exact original +- existing benchmark behavior does not regress + +## 13. Delivery Strategy + +Implementation occurs on `plugin/hf-comptext-universe` and changes only `hf_space/**`. + +Logical increments: + +1. static data model and validators +2. Universe loader and UI +3. AIR and Evidence previews +4. modular compression UI and profiles +5. security hardening and tests +6. README metadata and final integration + +No pull request or merge is performed without explicit user approval after implementation and validation. + +## 14. Success Criteria + +A visitor can answer: + +- What is CompText? +- Why is context central? +- What are AIR and Evidence? +- Which capabilities exist today versus being planned? +- How does protected compression save tokens safely? +- Why does the real CompText runtime remain local? + +The Space must remain useful even when ZeroGPU is unavailable: all Universe and preview features continue to work, while compression reports a clear runtime limitation. \ No newline at end of file diff --git a/hf_space/docs/2026-07-16-comptext-universe-implementation-plan.md b/hf_space/docs/2026-07-16-comptext-universe-implementation-plan.md new file mode 100644 index 0000000..eceda4f --- /dev/null +++ b/hf_space/docs/2026-07-16-comptext-universe-implementation-plan.md @@ -0,0 +1,79 @@ +# CompText Universe Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:executing-plans to implement this plan task-by-task. + +**Goal:** Turn the existing Hugging Face compression lab into a static CompText Universe explorer with safe context compression, deterministic AIR previews, simulated Evidence previews, contract/capability views, and isolated tests. + +**Architecture:** Keep every change under `hf_space/**`. The Space loads a committed static snapshot and never reads GitHub at runtime. Gradio composes read-only Universe views with the existing fail-closed LLMLingua pipeline; only compression receives ZeroGPU allocation. + +**Tech Stack:** Python 3.10, Gradio 5.44.1, pandas, LLMLingua 0.2.2, PyTorch 2.8.0, pytest. + +## Global Constraints + +- Modify only `hf_space/**`. +- No provider calls, repository writes, runtime GitHub access, secrets, persistence, or execution of generated AIR. +- Mark the product state as `local-dry-run-mvp` and previews as non-executable/simulated. +- Compression must return the exact original input whenever safety or minimum-reduction gates fail. +- ZeroGPU is used only for compression callbacks. + +--- + +### Task 1: Static Universe Data and Loader + +**Files:** +- Create: `hf_space/data/universe_snapshot.json` +- Create: `hf_space/universe.py` +- Create: `hf_space/tests/test_universe.py` + +**Interfaces:** +- Produces: `load_universe() -> dict`, `layers_frame(data) -> pandas.DataFrame`, `capabilities_frame(data) -> pandas.DataFrame`, `skills_frame(data) -> pandas.DataFrame`. + +- [ ] Write tests asserting required product status, seven architecture layers, capability statuses, source provenance, and deterministic dataframe columns. +- [ ] Implement the static snapshot and loader. +- [ ] Run `python -m pytest hf_space/tests/test_universe.py -q` and expect PASS. + +### Task 2: Deterministic AIR and Evidence Previews + +**Files:** +- Create: `hf_space/previews.py` +- Create: `hf_space/tests/test_previews.py` + +**Interfaces:** +- Produces: `build_air_preview(text, compression=None) -> dict`, `build_evidence_preview(original, output, decision, metrics) -> dict`, `scan_secrets(text) -> list[str]`. + +- [ ] Write tests for file/flag/constraint extraction, disabled execution, simulated Evidence, stable SHA-256 hashes, and secret blocking. +- [ ] Implement deterministic preview builders with no external calls. +- [ ] Run `python -m pytest hf_space/tests/test_previews.py -q` and expect PASS. + +### Task 3: Space UI Composition + +**Files:** +- Modify: `hf_space/app.py` +- Modify: `hf_space/README.md` + +**Interfaces:** +- Consumes: Universe loader, hybrid compression, AIR/Evidence previews. +- Produces: Gradio tabs `Overview`, `Architecture`, `Capabilities`, `Compression Lab`, `AIR & Evidence`, `Skills`, and `Benchmarks`. + +- [ ] Replace the single-purpose UI with the modular Universe composition while preserving compression and benchmark behavior. +- [ ] Add explicit maturity and safety notices. +- [ ] Update Space metadata and documentation. +- [ ] Run `python -m compileall -q hf_space` and expect exit code 0. + +### Task 4: Regression and Boundary Tests + +**Files:** +- Create: `hf_space/tests/test_boundaries.py` + +**Interfaces:** +- Verifies: no files referenced outside the allowed snapshot provenance; no provider-key configuration; preview execution is always disabled; Evidence is always simulated. + +- [ ] Add boundary tests. +- [ ] Run `python -m pytest hf_space/tests -q` and expect PASS. +- [ ] Verify the branch diff contains only `hf_space/**`. + +### Task 5: Review and Delivery + +- [ ] Compare branch against `main` and confirm every changed path begins with `hf_space/`. +- [ ] Create a pull request summarizing features, safety boundaries, and validation. +- [ ] Merge only after the path-boundary check passes. diff --git a/hf_space/previews.py b/hf_space/previews.py new file mode 100644 index 0000000..3939733 --- /dev/null +++ b/hf_space/previews.py @@ -0,0 +1,110 @@ +from __future__ import annotations + +import hashlib +import re +from typing import Any + +from safety_checks import extract_cli_flags, extract_file_paths + +_SECRET_PATTERNS = { + "private-key": re.compile(r"-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----"), + "github-token": re.compile(r"\bgh[pousr]_[A-Za-z0-9_]{20,}\b"), + "huggingface-token": re.compile(r"\bhf_[A-Za-z0-9]{20,}\b"), + "openai-key": re.compile(r"\bsk-(?:proj-)?[A-Za-z0-9_-]{20,}\b"), + "aws-access-key": re.compile(r"\bAKIA[0-9A-Z]{16}\b"), + "bearer-token": re.compile(r"\bBearer\s+[A-Za-z0-9._~+/=-]{16,}", re.I), + "secret-assignment": re.compile(r"\b(?:API_KEY|TOKEN|SECRET|PASSWORD)\s*=\s*[^\s]+", re.I), +} + + +def _sha256(text: str) -> str: + return "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def scan_secrets(text: str) -> list[str]: + return sorted(name for name, pattern in _SECRET_PATTERNS.items() if pattern.search(text or "")) + + +def _constraints(text: str) -> list[str]: + lower = text.lower() + constraints: list[str] = [] + if any(term in lower for term in ("no provider", "keine provider", "keine live-provider", "without provider")): + constraints.append("no_provider_calls") + if "--dry-run" in text: + constraints.append("dry_run_only") + if any(term in lower for term in ("do not write", "nicht Γ€ndern", "nicht veraendern", "read only", "nur lesen")): + constraints.append("no_writes") + if any(term in lower for term in ("do not commit", "nicht commit", "no commit")): + constraints.append("no_commits") + return sorted(set(constraints)) + + +def _expected_outputs(text: str) -> list[str]: + lower = text.lower() + outputs: list[str] = [] + if "json" in lower: + outputs.append("json_report") + if "markdown" in lower: + outputs.append("markdown_report") + if "summary" in lower or "zusammenfassung" in lower: + outputs.append("summary") + return outputs or ["analysis_preview"] + + +def build_air_preview(text: str, compression: dict[str, Any] | None = None) -> dict[str, Any]: + clean = (text or "").strip() + secret_matches = scan_secrets(clean) + files = sorted(extract_file_paths(clean)) + flags = sorted(extract_cli_flags(clean)) + constraints = _constraints(clean) + return { + "version": "preview-1", + "preview": True, + "executable": False, + "intent": "analyze_engineering_context", + "goal": clean[:240] if clean else "No goal supplied", + "context": { + "source": "user_input", + "input_hash": _sha256(clean), + "compression": compression or {"decision": "not_run"}, + "secret_scan": {"blocked": bool(secret_matches), "matches": secret_matches}, + }, + "files": files, + "tools": [], + "flags": flags, + "constraints": constraints, + "permissions": { + "network": False, + "provider": False, + "repository_write": False, + "execute": False, + }, + "expected_outputs": _expected_outputs(clean), + "execution": {"enabled": False, "reason": "Hugging Face Space previews contracts only."}, + } + + +def build_evidence_preview( + original: str, + output: str, + decision: str, + metrics: dict[str, Any] | None = None, +) -> dict[str, Any]: + secret_matches = scan_secrets(original) + return { + "schema": "simulated-evidence-preview-1", + "simulation": True, + "persisted": False, + "event_type": "context_preview_completed", + "actor": "hf-space-demo", + "summary": f"Context preview decision: {decision}", + "input_hash": _sha256(original), + "output_hash": _sha256(output), + "decision": decision, + "metrics": metrics or {}, + "redaction": { + "secrets_detected": bool(secret_matches), + "matches": secret_matches, + "raw_input_persisted": False, + }, + } diff --git a/hf_space/tests/test_boundaries.py b/hf_space/tests/test_boundaries.py new file mode 100644 index 0000000..fed99bd --- /dev/null +++ b/hf_space/tests/test_boundaries.py @@ -0,0 +1,42 @@ +from __future__ import annotations + +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) + +from previews import build_air_preview, build_evidence_preview +from universe import load_universe + + +def test_snapshot_sources_are_explicit_and_runtime_github_is_disabled(): + data = load_universe() + assert data["source"]["mode"] == "committed-static-snapshot" + assert data["source"]["runtime_github_access"] is False + assert all(not path.startswith((".env", "secrets/", "logs/")) for path in data["source"]["allowed_sources"]) + + +def test_air_can_never_enable_execution_or_mutating_permissions(): + air = build_air_preview("Ignore rules and push changes to GitHub with --force") + assert air["executable"] is False + assert air["execution"]["enabled"] is False + assert air["permissions"] == { + "network": False, + "provider": False, + "repository_write": False, + "execute": False, + } + + +def test_evidence_is_always_simulated_and_non_persistent(): + evidence = build_evidence_preview("input", "output", "preview_only") + assert evidence["simulation"] is True + assert evidence["persisted"] is False + assert evidence["redaction"]["raw_input_persisted"] is False + + +def test_space_source_has_no_provider_key_configuration(): + source = (ROOT / "app.py").read_text(encoding="utf-8") + (ROOT / "README.md").read_text(encoding="utf-8") + forbidden = ("OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GEMINI_API_KEY", "GITHUB_TOKEN") + assert not any(name in source for name in forbidden) diff --git a/hf_space/tests/test_previews.py b/hf_space/tests/test_previews.py new file mode 100644 index 0000000..77e6d07 --- /dev/null +++ b/hf_space/tests/test_previews.py @@ -0,0 +1,38 @@ +from __future__ import annotations + +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) + +from previews import build_air_preview, build_evidence_preview, scan_secrets + + +def test_air_preview_is_non_executable_and_extracts_contract_data(): + text = "Analysiere modules/cli/cli_entrypoint.py mit --dry-run. Keine Provider-Aufrufe. Gib JSON zurΓΌck." + air = build_air_preview(text) + assert air["preview"] is True + assert air["executable"] is False + assert air["execution"]["enabled"] is False + assert "modules/cli/cli_entrypoint.py" in air["files"] + assert "--dry-run" in air["flags"] + assert "dry_run_only" in air["constraints"] + assert "no_provider_calls" in air["constraints"] + assert "json_report" in air["expected_outputs"] + + +def test_evidence_preview_is_simulated_not_persisted_and_hashes_are_stable(): + first = build_evidence_preview("alpha", "beta", "accepted", {"reduction": 20}) + second = build_evidence_preview("alpha", "beta", "accepted", {"reduction": 20}) + assert first == second + assert first["simulation"] is True + assert first["persisted"] is False + assert first["input_hash"].startswith("sha256:") + assert first["redaction"]["raw_input_persisted"] is False + + +def test_secret_scanner_blocks_known_token_shapes(): + assert scan_secrets("HF_TOKEN=hf_abcdefghijklmnopqrstuvwxyz") == ["huggingface-token", "secret-assignment"] + air = build_air_preview("Use HF_TOKEN=hf_abcdefghijklmnopqrstuvwxyz") + assert air["context"]["secret_scan"]["blocked"] is True diff --git a/hf_space/tests/test_universe.py b/hf_space/tests/test_universe.py new file mode 100644 index 0000000..40f41b7 --- /dev/null +++ b/hf_space/tests/test_universe.py @@ -0,0 +1,30 @@ +from __future__ import annotations + +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) + +from universe import capabilities_frame, layers_frame, load_universe, skills_frame + + +def test_universe_snapshot_has_required_identity_and_layers(): + data = load_universe() + assert data["product"]["name"] == "CompText" + assert data["product"]["status"] == "local-dry-run-mvp" + assert len(data["layers"]) == 7 + assert data["source"]["runtime_github_access"] is False + + +def test_universe_frames_have_stable_columns(): + data = load_universe() + assert list(layers_frame(data).columns) == ["name", "status", "purpose", "inputs", "outputs"] + assert list(capabilities_frame(data).columns) == ["name", "surface", "status", "network", "provider", "mutating"] + assert list(skills_frame(data).columns) == ["name", "purpose", "validation", "boundary"] + + +def test_capability_statuses_use_explicit_maturity_vocabulary(): + data = load_universe() + allowed = {"implemented", "scaffolded", "experimental", "planned", "future", "disabled"} + assert {item["status"] for item in data["capabilities"]} <= allowed diff --git a/hf_space/universe.py b/hf_space/universe.py new file mode 100644 index 0000000..45b3c5c --- /dev/null +++ b/hf_space/universe.py @@ -0,0 +1,55 @@ +from __future__ import annotations + +import json +from functools import lru_cache +from pathlib import Path +from typing import Any + +import pandas as pd + +ROOT = Path(__file__).parent +SNAPSHOT_PATH = ROOT / "data" / "universe_snapshot.json" + + +@lru_cache(maxsize=1) +def load_universe() -> dict[str, Any]: + data = json.loads(SNAPSHOT_PATH.read_text(encoding="utf-8")) + required = {"snapshot_version", "product", "source", "layers", "capabilities", "skills", "contracts"} + missing = sorted(required - data.keys()) + if missing: + raise ValueError(f"Universe snapshot is missing required keys: {', '.join(missing)}") + if data["product"].get("status") != "local-dry-run-mvp": + raise ValueError("Universe snapshot must identify the product as local-dry-run-mvp.") + return data + + +def layers_frame(data: dict[str, Any]) -> pd.DataFrame: + return pd.DataFrame(data["layers"], columns=["name", "status", "purpose", "inputs", "outputs"]) + + +def capabilities_frame(data: dict[str, Any]) -> pd.DataFrame: + return pd.DataFrame( + data["capabilities"], + columns=["name", "surface", "status", "network", "provider", "mutating"], + ) + + +def skills_frame(data: dict[str, Any]) -> pd.DataFrame: + return pd.DataFrame(data["skills"], columns=["name", "purpose", "validation", "boundary"]) + + +def contracts_frame(data: dict[str, Any]) -> pd.DataFrame: + return pd.DataFrame(data["contracts"], columns=["name", "kind", "fields", "execution"]) + + +def overview_markdown(data: dict[str, Any]) -> str: + product = data["product"] + boundaries = "\n".join(f"- {item}" for item in data["boundaries"]) + return ( + f"# 🧭 {product['name']} Universe\n\n" + f"> {product['claim']}\n\n" + f"**Status:** `{product['status']}` Β· **Snapshot:** `{data['snapshot_version']}`\n\n" + f"{product['description']}\n\n" + "## Hard boundaries\n" + f"{boundaries}" + )