From eabfc9aa6b4a3a84313f157ceef46705399ca067 Mon Sep 17 00:00:00 2001 From: Christopher Sim Date: Fri, 10 Jul 2026 17:45:46 -0700 Subject: [PATCH] =?UTF-8?q?feat:=20Keeper=20v1=20=E2=80=94=20AI-assisted?= =?UTF-8?q?=20culling=20for=20photos=20and=20videos?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pivot the Aperture video-studio scaffold into the culling product: a local date-sorted library with a checksummed/deduped import pipeline (EXIF, RAW/HEIC thumbnails, video posters + scrub strips), a pure-JS CV pass (blur/exposure/ black-frame/accidental-clip) with explainable verdicts, pHash burst grouping with best-pick suggestions, RAW+JPEG/Live Photo pairing, local CLIP semantic search, a budgeted GPT-5.5 judge for borderline items, and a taste profile that learns from every override. Catalog is SQLite (node:sqlite) validated end-to-end by the new @keeper/schema zod contracts, brokered to the app by a spawned catalog service; the renderer is rebuilt on the inherited tokens/UI kit (lazy date grid, review queues with evidence, loupe with burst strip, keyboard-first culling, undo). Export writes Lightroom-readable XMP sidecars; the only deletion path is a two-step empty-rejects that moves originals to the OS Trash. Removes the video-editor surface (Remotion, timeline, EDL, style library, TTS/ElevenLabs) and rewrites docs, rules, and skills for Keeper. Co-authored-by: Cursor --- .claude/skills/auto-tune/SKILL.md | 26 - .claude/skills/create-social-video/SKILL.md | 28 - .claude/skills/critique-video/SKILL.md | 62 - .claude/skills/cull-shoot/SKILL.md | 56 + .claude/skills/find-media/SKILL.md | 26 + .claude/skills/learn-aesthetic/SKILL.md | 31 - .claude/skills/organize-library/SKILL.md | 33 + .cursor/rules/design-system.mdc | 12 +- .cursor/rules/edl-schema.mdc | 21 - .cursor/rules/engine-scripts.mdc | 28 +- .cursor/rules/ipc-conventions.mdc | 32 +- .cursor/rules/schema.mdc | 20 + .cursor/rules/workflow.mdc | 4 +- .github/pull_request_template.md | 3 +- .github/workflows/ci.yml | 4 +- .gitignore | 15 +- AGENTS.md | 52 +- CHANGELOG.md | 121 +- CLAUDE.md | 2 +- README.md | 159 +- app/.env.local.example | 21 +- app/package.json | 15 +- app/resources/music/README.md | 8 - app/scripts/analyze-benchmarks.mjs | 110 - app/scripts/analyze-collection.mjs | 261 -- app/scripts/analyze-style.mjs | 149 - app/scripts/analyze.mjs | 166 - app/scripts/autotune-llm.mjs | 157 - app/scripts/autotune.mjs | 199 -- app/scripts/catalog-service.mjs | 279 ++ app/scripts/critique-llm.mjs | 125 - app/scripts/edl-util.mjs | 141 - app/scripts/edl-util.test.mjs | 141 - app/scripts/export.mjs | 103 + app/scripts/extract-frames.mjs | 64 - app/scripts/generate-llm.mjs | 261 -- app/scripts/import.mjs | 245 ++ app/scripts/judge-llm.mjs | 182 ++ app/scripts/lib/catalog.mjs | 538 ++++ app/scripts/lib/catalog.test.mjs | 132 + app/scripts/lib/embeddings.mjs | 103 + app/scripts/lib/grouping.mjs | 131 + app/scripts/lib/grouping.test.mjs | 97 + app/scripts/lib/llm-util.mjs | 69 + app/scripts/lib/llm-util.test.mjs | 61 + app/scripts/lib/media.mjs | 316 ++ app/scripts/lib/paths.mjs | 82 + app/scripts/lib/phash.mjs | 105 + app/scripts/lib/phash.test.mjs | 50 + app/scripts/lib/pipeline.mjs | 249 ++ app/scripts/lib/quality.mjs | 159 + app/scripts/lib/quality.test.mjs | 119 + app/scripts/lib/taste.mjs | 117 + app/scripts/lib/taste.test.mjs | 87 + app/scripts/lib/xmp.mjs | 61 + app/scripts/lib/xmp.test.mjs | 39 + app/scripts/llm.mjs | 24 +- app/scripts/query.mjs | 118 + app/scripts/render.mjs | 79 - app/scripts/reprocess.mjs | 70 + app/scripts/transcribe.mjs | 92 - app/scripts/tts-util.mjs | 81 - app/scripts/tts-util.test.mjs | 65 - app/scripts/tts.mjs | 181 -- app/scripts/verdict.mjs | 58 + app/scripts/write-narration.mjs | 97 - app/src/main/audio-sources.ts | 54 - app/src/main/index.ts | 1616 +++------- app/src/preload/index.ts | 323 +- app/src/renderer/src/App.tsx | 312 +- app/src/renderer/src/components/AssetCard.tsx | 59 + .../renderer/src/components/CritiquePanel.tsx | 352 -- .../renderer/src/components/EditorHeader.tsx | 114 - .../renderer/src/components/ErrorBoundary.tsx | 8 +- .../renderer/src/components/ExportModal.tsx | 163 +- app/src/renderer/src/components/Header.tsx | 113 + app/src/renderer/src/components/Home.test.tsx | 51 - app/src/renderer/src/components/Home.tsx | 364 --- .../src/components/InspectorPanel.test.tsx | 27 - .../src/components/InspectorPanel.tsx | 526 --- .../renderer/src/components/LeftRail.test.tsx | 26 - app/src/renderer/src/components/LeftRail.tsx | 649 +--- .../renderer/src/components/LibraryGrid.tsx | 136 + app/src/renderer/src/components/Loupe.tsx | 238 ++ .../renderer/src/components/PreviewStage.tsx | 90 - .../renderer/src/components/RejectsModal.tsx | 80 + .../renderer/src/components/ReviewMode.tsx | 188 ++ .../renderer/src/components/RightPanel.tsx | 36 - .../renderer/src/components/SettingsModal.tsx | 667 +--- .../src/components/StylePanel.test.tsx | 20 - .../renderer/src/components/StylePanel.tsx | 221 -- .../renderer/src/components/TasteModal.tsx | 98 + app/src/renderer/src/components/Timeline.tsx | 644 ---- app/src/renderer/src/env.d.ts | 4 +- app/src/renderer/src/lib/critique.test.ts | 50 - app/src/renderer/src/lib/critique.ts | 150 - app/src/renderer/src/lib/edl-edit.test.ts | 89 - app/src/renderer/src/lib/edl-edit.ts | 153 - app/src/renderer/src/lib/format.ts | 94 + app/src/renderer/src/main.tsx | 3 +- app/src/renderer/src/motion/Root.tsx | 44 - app/src/renderer/src/motion/SocialVideo.tsx | 418 --- .../renderer/src/motion/animations.test.ts | 32 - app/src/renderer/src/motion/animations.ts | 124 - app/src/renderer/src/motion/index.ts | 5 - app/src/renderer/src/store.test.ts | 198 +- app/src/renderer/src/store.ts | 624 ++-- app/src/renderer/src/styles.css | 800 ----- app/src/renderer/src/styles/editor.css | 1134 ------- app/src/renderer/src/styles/library.css | 1094 +++++++ .../renderer/src/styles/visual-styles.json | 98 - app/src/renderer/src/styles/visual-styles.ts | 24 - docs/notes/automated-benchmark-sourcing.md | 76 - fixtures/sample-project/analysis.json | 5 - fixtures/sample-project/assets/.gitkeep | 1 - fixtures/sample-project/critique.json | 17 - fixtures/sample-project/edl.json | 112 - fixtures/sample-project/meta.json | 5 - fixtures/sample-project/prompt.md | 12 - fixtures/sample-project/renders/.gitkeep | 1 - fixtures/sample-project/transcripts/.gitkeep | 1 - package-lock.json | 2850 ++++++----------- package.json | 25 +- packages/edl/src/index.ts | 69 - packages/edl/src/schema.test.ts | 135 - packages/edl/src/schema.ts | 329 -- packages/edl/src/types.ts | 53 - packages/{edl => schema}/package.json | 2 +- .../{edl => schema}/scripts/postbuild.mjs | 2 +- packages/schema/src/index.ts | 99 + packages/schema/src/schema.test.ts | 181 ++ packages/schema/src/schema.ts | 355 ++ packages/schema/src/types.ts | 62 + packages/{edl => schema}/tsconfig.cjs.json | 0 packages/{edl => schema}/tsconfig.esm.json | 0 packages/{edl => schema}/tsconfig.json | 0 test/setup.ts | 68 +- vitest.config.ts | 4 +- 138 files changed, 9290 insertions(+), 14229 deletions(-) delete mode 100644 .claude/skills/auto-tune/SKILL.md delete mode 100644 .claude/skills/create-social-video/SKILL.md delete mode 100644 .claude/skills/critique-video/SKILL.md create mode 100644 .claude/skills/cull-shoot/SKILL.md create mode 100644 .claude/skills/find-media/SKILL.md delete mode 100644 .claude/skills/learn-aesthetic/SKILL.md create mode 100644 .claude/skills/organize-library/SKILL.md delete mode 100644 .cursor/rules/edl-schema.mdc create mode 100644 .cursor/rules/schema.mdc delete mode 100644 app/resources/music/README.md delete mode 100644 app/scripts/analyze-benchmarks.mjs delete mode 100644 app/scripts/analyze-collection.mjs delete mode 100644 app/scripts/analyze-style.mjs delete mode 100644 app/scripts/analyze.mjs delete mode 100644 app/scripts/autotune-llm.mjs delete mode 100644 app/scripts/autotune.mjs create mode 100644 app/scripts/catalog-service.mjs delete mode 100644 app/scripts/critique-llm.mjs delete mode 100644 app/scripts/edl-util.mjs delete mode 100644 app/scripts/edl-util.test.mjs create mode 100644 app/scripts/export.mjs delete mode 100644 app/scripts/extract-frames.mjs delete mode 100644 app/scripts/generate-llm.mjs create mode 100644 app/scripts/import.mjs create mode 100644 app/scripts/judge-llm.mjs create mode 100644 app/scripts/lib/catalog.mjs create mode 100644 app/scripts/lib/catalog.test.mjs create mode 100644 app/scripts/lib/embeddings.mjs create mode 100644 app/scripts/lib/grouping.mjs create mode 100644 app/scripts/lib/grouping.test.mjs create mode 100644 app/scripts/lib/llm-util.mjs create mode 100644 app/scripts/lib/llm-util.test.mjs create mode 100644 app/scripts/lib/media.mjs create mode 100644 app/scripts/lib/paths.mjs create mode 100644 app/scripts/lib/phash.mjs create mode 100644 app/scripts/lib/phash.test.mjs create mode 100644 app/scripts/lib/pipeline.mjs create mode 100644 app/scripts/lib/quality.mjs create mode 100644 app/scripts/lib/quality.test.mjs create mode 100644 app/scripts/lib/taste.mjs create mode 100644 app/scripts/lib/taste.test.mjs create mode 100644 app/scripts/lib/xmp.mjs create mode 100644 app/scripts/lib/xmp.test.mjs create mode 100644 app/scripts/query.mjs delete mode 100644 app/scripts/render.mjs create mode 100644 app/scripts/reprocess.mjs delete mode 100644 app/scripts/transcribe.mjs delete mode 100644 app/scripts/tts-util.mjs delete mode 100644 app/scripts/tts-util.test.mjs delete mode 100644 app/scripts/tts.mjs create mode 100644 app/scripts/verdict.mjs delete mode 100644 app/scripts/write-narration.mjs delete mode 100644 app/src/main/audio-sources.ts create mode 100644 app/src/renderer/src/components/AssetCard.tsx delete mode 100644 app/src/renderer/src/components/CritiquePanel.tsx delete mode 100644 app/src/renderer/src/components/EditorHeader.tsx create mode 100644 app/src/renderer/src/components/Header.tsx delete mode 100644 app/src/renderer/src/components/Home.test.tsx delete mode 100644 app/src/renderer/src/components/Home.tsx delete mode 100644 app/src/renderer/src/components/InspectorPanel.test.tsx delete mode 100644 app/src/renderer/src/components/InspectorPanel.tsx delete mode 100644 app/src/renderer/src/components/LeftRail.test.tsx create mode 100644 app/src/renderer/src/components/LibraryGrid.tsx create mode 100644 app/src/renderer/src/components/Loupe.tsx delete mode 100644 app/src/renderer/src/components/PreviewStage.tsx create mode 100644 app/src/renderer/src/components/RejectsModal.tsx create mode 100644 app/src/renderer/src/components/ReviewMode.tsx delete mode 100644 app/src/renderer/src/components/RightPanel.tsx delete mode 100644 app/src/renderer/src/components/StylePanel.test.tsx delete mode 100644 app/src/renderer/src/components/StylePanel.tsx create mode 100644 app/src/renderer/src/components/TasteModal.tsx delete mode 100644 app/src/renderer/src/components/Timeline.tsx delete mode 100644 app/src/renderer/src/lib/critique.test.ts delete mode 100644 app/src/renderer/src/lib/critique.ts delete mode 100644 app/src/renderer/src/lib/edl-edit.test.ts delete mode 100644 app/src/renderer/src/lib/edl-edit.ts create mode 100644 app/src/renderer/src/lib/format.ts delete mode 100644 app/src/renderer/src/motion/Root.tsx delete mode 100644 app/src/renderer/src/motion/SocialVideo.tsx delete mode 100644 app/src/renderer/src/motion/animations.test.ts delete mode 100644 app/src/renderer/src/motion/animations.ts delete mode 100644 app/src/renderer/src/motion/index.ts delete mode 100644 app/src/renderer/src/styles.css delete mode 100644 app/src/renderer/src/styles/editor.css create mode 100644 app/src/renderer/src/styles/library.css delete mode 100644 app/src/renderer/src/styles/visual-styles.json delete mode 100644 app/src/renderer/src/styles/visual-styles.ts delete mode 100644 docs/notes/automated-benchmark-sourcing.md delete mode 100644 fixtures/sample-project/analysis.json delete mode 100644 fixtures/sample-project/assets/.gitkeep delete mode 100644 fixtures/sample-project/critique.json delete mode 100644 fixtures/sample-project/edl.json delete mode 100644 fixtures/sample-project/meta.json delete mode 100644 fixtures/sample-project/prompt.md delete mode 100644 fixtures/sample-project/renders/.gitkeep delete mode 100644 fixtures/sample-project/transcripts/.gitkeep delete mode 100644 packages/edl/src/index.ts delete mode 100644 packages/edl/src/schema.test.ts delete mode 100644 packages/edl/src/schema.ts delete mode 100644 packages/edl/src/types.ts rename packages/{edl => schema}/package.json (95%) rename packages/{edl => schema}/scripts/postbuild.mjs (94%) create mode 100644 packages/schema/src/index.ts create mode 100644 packages/schema/src/schema.test.ts create mode 100644 packages/schema/src/schema.ts create mode 100644 packages/schema/src/types.ts rename packages/{edl => schema}/tsconfig.cjs.json (100%) rename packages/{edl => schema}/tsconfig.esm.json (100%) rename packages/{edl => schema}/tsconfig.json (100%) diff --git a/.claude/skills/auto-tune/SKILL.md b/.claude/skills/auto-tune/SKILL.md deleted file mode 100644 index f2e0c10..0000000 --- a/.claude/skills/auto-tune/SKILL.md +++ /dev/null @@ -1,26 +0,0 @@ ---- -name: auto-tune -description: Iteratively improve a short-form cut by looping generate/adjust -> critique -> apply fixes -> re-critique, logging each iteration's score to results.tsv. Use when the user wants the agent to "auto-improve" or "keep iterating until it's good". ---- - -# Auto-Tune - -A critique-in-the-loop optimizer for a single video, modeled on the design `auto-skills` pattern. It treats the `critique-video` score as the fitness function and stops when it plateaus or hits the target. - -Operates on one project: `projects//edl.json`, scored against `benchmarks.json` when present. - -## Loop - -1. Baseline: run the `critique-video` skill, record the score. Append a header row to `projects//results.tsv` if missing (`iter\tscore\tdelta\tchange`), then log iteration 0. -2. Pick the lowest subscore with a concrete, safe fix. Prefer fixes that move a metric toward the creator's `benchmarks.json` distribution (pacing toward `cutsPer10s.mean`, length toward `durationSec.mean`). -3. Apply exactly one change to `edl.json` (e.g. tighten pacing by trimming/reordering clips, add a stronger hook in the first 2s, enable/fix captions, set safe margins, add or duck a music bed, sharpen the ending). Keep it schema-valid (`EdlSchema`). -4. Re-run `critique-video`. If the score improved, keep the change and log the iteration (`i score +delta change`); if it regressed, revert. -5. Repeat until: the target score is reached, the score plateaus (no improving change), or you hit the iteration cap (default 4-6). - -## Rules - -- One change per iteration so each delta is attributable. -- Never emit an `edl.json` that fails `EdlSchema`; never reference assets that don't exist. -- Learn toward general principles, not one-off hacks; the goal is a genuinely better cut, not a gamed score. -- Always leave `results.tsv` and `edl.json` in a consistent, valid state (the editor live-reloads `edl.json`). -- Be honest in the final summary: report the start score, end score, and what changed. diff --git a/.claude/skills/create-social-video/SKILL.md b/.claude/skills/create-social-video/SKILL.md deleted file mode 100644 index 0b66a27..0000000 --- a/.claude/skills/create-social-video/SKILL.md +++ /dev/null @@ -1,28 +0,0 @@ ---- -name: create-social-video -description: Turn a prompt + uploaded clips into a first-cut short-form vertical video by writing a validated edl.json. Use when the user wants to generate or assemble a social video (Instagram / TikTok / Reels) from raw footage and an English description of intent. ---- - -# Create Social Video - -End-to-end first-cut generator. Input: `projects//prompt.md` plus clips in `projects//assets/`. Output: a validated `projects//edl.json` and word-level transcripts in `projects//transcripts/`. - -## Steps - -1. Read the brief: `projects//prompt.md` (intent, vibe, target length, platform). -1b. Read the creator's style, if present: `projects//style.json` (a StyleProfile written by `learn-aesthetic`) and/or a named preset in `app/src/renderer/src/styles/visual-styles.json` referenced by `meta.json.styleProfileId`. Let it drive palette, font, captionStyle, pacing (cuts-per-10s / avg shot length), target length, hook pattern, and the do/avoid rules. The prompt still wins where they conflict. -2. Inventory + baseline assembly: run `node app/scripts/analyze.mjs --slug `. It probes every clip with `@remotion/media-parser` and writes a deterministic first-cut `edl.json` (clips overlapped for crossfades) you then refine. -3. Captions (optional, needs speech): run `node app/scripts/transcribe.mjs --slug ` to extract audio + run whisper.cpp and write word-level `words[]` into the caption track. -4. Plan the edit from the intent: pick a hook clip for the first ~2s, reorder/trim clips, set pacing. Edit `edl.json` directly. -5. Add text overlays (title / subtitle) and pick transitions + text animations: `anim.name` is an animate-text spec (see `app/src/renderer/src/motion/animations.ts`); `transitionOut.preset` is `fade | slide | wipe`. -6. If music is provided, add it as an audio asset + an audio track; set `gain` and `duckUnderVoice`. -7. Every `edl.json` MUST pass `EdlSchema` from `packages/edl/src/schema.ts`. Preview via the editor (`npm run dev`), export via the Export button or `node app/scripts/render.mjs --slug `. - -## Rules - -- Format defaults to 1080x1920 @ 30fps. -- Respect `theme.safeMargins` — keep text/captions out of the platform UI zones. -- Make the first ~2 seconds a strong hook (follow `style.json.hookPattern` when present). -- When a style profile exists, set `theme.palette`, `theme.fontFamily`, `theme.captionStyle`, and `theme.stylePreset` from it, and match its pacing (cut roughly to its `cutsPer10s` / `avgShotSec`) and `targetLengthSec`. -- Only reference assets that exist in `assets/` and are declared in `edl.assets`. -- Never emit an edl.json that fails schema validation. diff --git a/.claude/skills/critique-video/SKILL.md b/.claude/skills/critique-video/SKILL.md deleted file mode 100644 index 74c6115..0000000 --- a/.claude/skills/critique-video/SKILL.md +++ /dev/null @@ -1,62 +0,0 @@ ---- -name: critique-video -description: Score a short-form video cut against short-form best practices (and the creator's own high-performers when available) and write critique.json with a 0-100 score and specific fixes. Use after a first cut exists, when the user wants feedback or a quality read on a video. ---- - -# Critique Video - -Reads `projects//edl.json` (and optional rendered stills + `benchmarks.json`) and writes `projects//critique.json`. - -## Calibration - -Grade to the standard of a demanding short-form editor: "would this earn a stop-scroll and a rewatch?" Score what's actually there, not the intent. Most AI-assembled first cuts land 45-65. Above 80 means you'd confidently post it. Bands: 0-39 broken, 40-59 functional, 60-74 good, 75-89 excellent, 90-100 world-class. - -## Rubric (0-100, weighted) - -- Hook strength (first ~2s) — 25 -- Pacing / cut frequency — 15 -- Caption coverage + legibility — 15 -- Vertical safe-area compliance — 10 -- Length vs platform norm — 10 -- Audio presence / quality — 15 -- Ending / payoff — 10 - -## Benchmark calibration (preferred) - -If `projects//benchmarks.json` exists (built by `node app/scripts/analyze-benchmarks.mjs` from the creator's uploaded high-performers), score Pacing and Length RELATIVE to that distribution, not fixed thresholds: - -- Pull `distribution.cutsPer10s` and `distribution.durationSec` (each has `mean`, `std`, `min`, `max`). -- Full marks when the cut is within ~1 std of the mean; decay toward 0 by ~3 std. -- In the matching subscore, set `benchmark: { yours, theirs, unit }` (theirs = the benchmark mean) and make the fix comparative ("your top videos average ~6 cuts/10s; this has 3"). - -If `benchmarks.json` is absent, score on heuristics and say so in `summary`. - -## Common short-form pitfalls to flag - -Slow/ambiguous first second, no captions (most viewers are muted), text inside the platform UI safe zones, monotone pacing (evenly spaced cuts), dead air / no audio bed, and an ending that just stops instead of paying off. - -## Output: critique.json - -Write this exact shape (it powers the editor's Critique panel): - -```json -{ - "score": 0, - "subscores": [ - { "key": "hook", "label": "Hook (first 2s)", "max": 25, "score": 0, "note": "" }, - { "key": "pacing", "label": "Pacing", "max": 15, "score": 0, "note": "", "benchmark": { "yours": 0, "theirs": 0, "unit": "cuts/10s" } }, - { "key": "captions", "label": "Captions", "max": 15, "score": 0, "note": "" }, - { "key": "safe", "label": "Safe areas", "max": 10, "score": 0, "note": "" }, - { "key": "length", "label": "Length", "max": 10, "score": 0, "note": "", "benchmark": { "yours": 0, "theirs": 0, "unit": "s" } }, - { "key": "audio", "label": "Audio", "max": 15, "score": 0, "note": "" }, - { "key": "ending", "label": "Ending", "max": 10, "score": 0, "note": "" } - ], - "fixes": [{ "issue": "", "fix": "" }], - "benchmarksUsed": false, - "summary": "" -} -``` - -## Honesty - -This is a craft + fit score, not a virality guarantee. When calibrated against `benchmarks.json`, say how many of the creator's videos it was compared to in `summary`. Otherwise state plainly that it's heuristic. diff --git a/.claude/skills/cull-shoot/SKILL.md b/.claude/skills/cull-shoot/SKILL.md new file mode 100644 index 0000000..d9838cb --- /dev/null +++ b/.claude/skills/cull-shoot/SKILL.md @@ -0,0 +1,56 @@ +--- +name: cull-shoot +description: Import a folder of photos/videos into the Keeper library (or work on what's already there) and cull it — confirm the pipeline's verdicts, visually review the borderline items, pick the best frame of each burst, and leave a clean picks/rejects split. Use when the user wants a shoot, trip, or SD-card dump culled. +--- + +# Cull a shoot + +You are culling for a real person: the goal is that their **picks are postable/editable keepers** and their **rejects are safely discardable junk**. You can look at images directly — use that superpower where the local heuristics can't decide. + +## Ground rules + +- The library home is `~/Pictures/Keeper` unless `KEEPER_LIBRARY_DIR` is set. Pass `--library ` to every script if the user gave a custom location. +- **Never delete or move media files.** Verdicts are flags; the user empties rejects themselves in the app. +- Every CLI below prints JSON. Scripts live in `app/scripts/`. + +## Workflow + +1. **Import** (skip if the media is already in the library): + + ```bash + node app/scripts/import.mjs --source /path/to/dump + ``` + + This copies (checksummed, deduped), extracts metadata, generates thumbnails, measures quality, suggests verdicts, groups bursts, and builds the search index. Resume an interrupted run with `node app/scripts/reprocess.mjs`. + +2. **Read the queues**: + + ```bash + node app/scripts/query.mjs --review --limit 100 + ``` + + Three buckets: `sure-reject` (high-confidence junk), `sure-keep`, `needs-eye` (borderline). Each item includes `thumbPath` — an absolute path to its thumbnail. + +3. **Spot-check the sure queues.** View a handful of `sure-reject` thumbnails with the Read tool. If they are genuinely junk (blur, black frames, misfires), confirm the whole queue: + + ```bash + node app/scripts/verdict.mjs --ids --flag reject + ``` + + Same for `sure-keep` with `--flag pick`. If a spot-check reveals a wrong call, DO NOT bulk-confirm — review that queue item by item. + +4. **Review `needs-eye` visually.** Read each thumbnail. Judge like a photo editor: moment > technical perfection. A soft photo of a real moment (laughter, a kiss, the peak of action) is a **pick**; a tack-sharp photo of nothing is not automatically one. Set verdicts in batches. + +5. **Resolve bursts.** Items sharing a `groupId` are one moment. View all frames of a group, pick the best (eyes open, expression, framing), then: + + ```bash + node app/scripts/verdict.mjs --group --best + node app/scripts/verdict.mjs --ids --flag pick + node app/scripts/verdict.mjs --ids --flag reject + ``` + +6. **Report.** Summarize honestly: counts per verdict, anything you were unsure about (leave those unrated rather than guessing), and any pattern worth adding as a taste rule. + +## Taste + +Your verdicts feed `taste.json` exactly like the user's own (contradictions of AI suggestions become learning exemplars). Read the profile first if it exists — `/taste.json` — and respect its `rules`. diff --git a/.claude/skills/find-media/SKILL.md b/.claude/skills/find-media/SKILL.md new file mode 100644 index 0000000..fc46fdf --- /dev/null +++ b/.claude/skills/find-media/SKILL.md @@ -0,0 +1,26 @@ +--- +name: find-media +description: Find photos/videos in the Keeper library from a vague natural-language memory ("the ocean at sunset", "the kids at that dinner in June") using semantic search, then visually verify the hits. Use when the user is looking for specific media they half-remember. +--- + +# Find media + +The library ships a local CLIP index — semantic search runs offline and returns ranked candidates. Your job is to turn a fuzzy memory into verified results. + +## Workflow + +1. **Search** (scripts in `app/scripts/`, `--library ` if non-default): + + ```bash + node app/scripts/query.mjs --search "ocean at sunset" --limit 24 + ``` + + Results are ranked by similarity and include `thumbPath` + `originalPath`. If it errors with "no embeddings yet", run `node app/scripts/reprocess.mjs --stage embed` first (downloads the model on first run). + +2. **Vary the phrasing.** CLIP responds to concrete visual language. Try 2–3 reformulations: "waves at golden hour", "beach silhouette dusk". Merge candidates. + +3. **Verify visually.** Read the top thumbnails and keep only genuine matches — semantic scores are suggestive, not proof. The user asked for a memory, not a similarity list. + +4. **Narrow by metadata when the user gave constraints.** Capture dates are in each record (`capturedAt`); "second week of the trip" means filtering the date range yourself. `query.mjs --asset ` returns the full record. + +5. **Present**: absolute original paths + one-line why-it-matches each. If the user wants them exported, set `--flag pick` via `verdict.mjs` and point them at Export in the app (or run `node app/scripts/export.mjs --dest --ids --xmp`). diff --git a/.claude/skills/learn-aesthetic/SKILL.md b/.claude/skills/learn-aesthetic/SKILL.md deleted file mode 100644 index 778521d..0000000 --- a/.claude/skills/learn-aesthetic/SKILL.md +++ /dev/null @@ -1,31 +0,0 @@ ---- -name: learn-aesthetic -description: Study a creator's own past videos and write a reusable aesthetic profile (style.json) that seeds generation. Use when the user wants the agent to "learn my style/vibe" from reference clips they've uploaded into a project's references/ folder. ---- - -# Learn Aesthetic - -Turn a creator's uploaded reference videos into a portable style profile. Input: clips in `projects//references/`. Output: a validated `projects//style.json` (StyleProfileSchema in `packages/edl/src/schema.ts`) plus a human-readable `projects//aesthetic.md`. - -This is the multimodal complement to the deterministic `analyze-style.mjs` script: the script measures palette + pacing; you watch the frames and capture the taste. - -## Steps - -1. Confirm references exist: list `projects//references/`. If empty, ask the user to upload 2-5 of their own past videos first. -2. Run the deterministic baseline: `node app/scripts/extract-frames.mjs --slug ` (samples stills into `references/.frames/`) and `node app/scripts/analyze-style.mjs --slug ` (writes a baseline `style.json` with palette, pacing, length, energy). -3. Look at the sampled frames in `references/.frames/`. Read them as images. Note: dominant colors and contrast, framing/composition, on-screen text treatment (size, position, font feel, caption style), motion energy, and any recurring hook structure in the first ~2s. -4. If transcripts exist (or you run `node app/scripts/transcribe.mjs`), note verbal hook patterns and pacing of speech. -5. Refine `style.json` on top of the baseline. Keep the script's measured `palette`, `pacing`, `targetLengthSec`, and `energy` unless your eyes strongly disagree; then add the interpretive fields: - - `fontFamily` (a CSS stack matching the vibe), `captionStyle` (`karaoke | block | word | none`). - - `hookPattern` — one sentence describing how their strongest opens work. - - `do` — 3-6 concrete, transferable style rules (e.g. "Open on the single most striking shot, no title card first"). - - `avoid` — 3-6 things that would break the vibe. - - `notes` — a short paragraph a stranger could use to reproduce the look. -6. Write `aesthetic.md`: a readable narrative of the creator's style (palette, type, motion, hook, pacing, energy) for the user to skim and the generator to reference. - -## Rules - -- Learn GENERAL, transferable principles, not the literal contents of one clip ("warm low-key palette, fast cuts on the beat" — not "the ramen shot at 0:04"). -- Never invent metrics the script measured differently; the script is the source of truth for palette/pacing/length. -- `style.json` MUST validate against `StyleProfileSchema`. -- Set `id` to something stable (e.g. `learned`) so `meta.json.styleProfileId` and generation can reference it. diff --git a/.claude/skills/organize-library/SKILL.md b/.claude/skills/organize-library/SKILL.md new file mode 100644 index 0000000..c15c2a3 --- /dev/null +++ b/.claude/skills/organize-library/SKILL.md @@ -0,0 +1,33 @@ +--- +name: organize-library +description: Organize and enrich the Keeper library — caption and tag media for better search, name events, surface duplicates across imports, and produce a library health report. Use when the user wants their library "organized", "tagged", or "cleaned up". +--- + +# Organize the library + +Keeper's catalog stores tags and captions per asset; search and export both use them. You can look at thumbnails directly and write enrichment back through the CLIs (scripts in `app/scripts/`, add `--library ` for non-default locations). + +## Workflow + +1. **Assess**: + + ```bash + node app/scripts/query.mjs --counts + node app/scripts/query.mjs --imports + ``` + + Report totals, unrated backlog, and how much of the search index is built (run `reprocess.mjs` for anything missing). + +2. **Enrich where it pays.** Captions/tags come from the LLM judge when configured: + + ```bash + node app/scripts/judge-llm.mjs --budget 200 + ``` + + It only spends budget on borderline/unjudged items and writes captions + lowercase tags. If the user wants deeper coverage, raise the budget explicitly — never silently. + +3. **Name events.** List days (`query.mjs --counts` shows the day count; `--list unrated --limit …` etc. include `capturedAt`). Cluster consecutive days with media into events, read a few thumbnails per day, and propose names ("Kyoto — days 3–5"). Write them as tags on the day's assets via the catalog CLI if the user approves. + +4. **Surface cross-import duplicates.** Exact duplicates are already blocked at import (content hash). Near-duplicates across sessions show up as high-similarity search hits — flag suspicious pairs for the user rather than auto-rejecting. + +5. **Report.** What was tagged, what events were named, what the user should review. Nothing in this skill deletes or moves media. diff --git a/.cursor/rules/design-system.mdc b/.cursor/rules/design-system.mdc index bcd0afa..171e2d8 100644 --- a/.cursor/rules/design-system.mdc +++ b/.cursor/rules/design-system.mdc @@ -10,6 +10,7 @@ alwaysApply: false - Colors come from `styles/tokens.css` custom properties (`--background-primary`, `--foreground-secondary`, `--border-primary`, ...). Both themes must work — check `:root[data-theme="dark"]` derivations before adding a raw hex. - Radii/spacing/typography: `--radius-*`, `--space-*`, `--font-ui`, `--font-brand`. - Needing a "slightly different" color? Prefer `color-mix(in srgb, var(--token) N%, var(--other-token))` over new hex values. +- Verdict semantics are the one hardcoded exception: keep-green `#2f9e63`, reject-red `#c4384f`, star-gold `#f5c451` (used identically in both themes). ## Components - Reuse the UI kit first: `Button`, `IconButton`, `Badge`, `Modal`, `Field`/`Input`/`TextArea`/`Select`, `Icon` from `components/ui`. @@ -18,10 +19,11 @@ alwaysApply: false - Dialogs/dropdowns must close on Escape (`useEscapeKey`) and outside click. ## CSS placement -- Editor chrome styles: `styles/editor.css`. UI kit: `styles/ui.css`. Home + legacy: `styles.css`. Delete dead CSS when retiring a component. -- File inputs are hidden `` + a styled `upload-area` div; snapshot `FileList` synchronously (it's live and emptied by `input.value = ""`). +- Library UI styles: `styles/library.css`. UI kit: `styles/ui.css`. Delete dead CSS when retiring a component. +- Media loads through the `keeper-asset://home/` protocol via the helpers in `lib/format.ts` (`thumbUrl`/`previewUrl`/`originalUrl`) — never build protocol URLs inline. +- Grid thumbnails use `loading="lazy"` and day sections lazy-load via IntersectionObserver — keep it that way; a 50k library must open instantly. ## Store & interaction -- Every EDL mutation goes through `updateEdl(mutate)` — it's one undo step and schedules the debounced save. Batch multi-asset operations into ONE `updateEdl` call. -- Editor-wide shortcuts live in `App.tsx` (Cmd+Z, T, Cmd+\, Space) — never inside a component that unmounts (e.g. Timeline in focus mode). Always guard: skip when typing in INPUT/TEXTAREA/SELECT/contentEditable. -- Long text/media lists: cap visible rows and scroll within (`clip-list-capped` pattern). +- Every verdict mutation goes through `useKeeper.setVerdict(ids, patch)` — it's optimistic, one undo step, and records the taste signal server-side. Batch multi-asset operations into ONE call. +- App-wide shortcuts live in `App.tsx` only (P/K keep, X/R reject, U unrated, 0–5 stars, Space loupe, arrows navigate, Cmd+Z undo, T theme, Cmd+F search) — never inside a component that unmounts. Always guard: skip when typing in INPUT/TEXTAREA/SELECT/contentEditable. +- Records are normalized in one `Map` (`records`); section/search/queue views hold ids only. Merge fetched assets with `mergeRecords`, never store copies. diff --git a/.cursor/rules/edl-schema.mdc b/.cursor/rules/edl-schema.mdc deleted file mode 100644 index 51d31fa..0000000 --- a/.cursor/rules/edl-schema.mdc +++ /dev/null @@ -1,21 +0,0 @@ ---- -description: EDL schema package rules (validation boundary, bounds, rebuild) -globs: packages/edl/** -alwaysApply: false ---- - -# EDL schema (packages/edl) - -The schema is a **security boundary**: project files are a shareable interchange format and LLMs author them. Every field must stay hostile-input-safe. - -- Numerics: `.finite()` + explicit bounds (timeline seconds cap at `MAX_TIMELINE_SEC`). Unbounded values hang the timeline/player (JSON `1e400` parses to `Infinity`). -- Media paths (`src`, `proxySrc`, caption `source`): `RelativePathSchema` — no `..`, absolute paths, drive letters, UNC, or NUL. -- Colors: `CssColorSchema` allowlist (hex / rgb() / hsl() / bare names) — arbitrary strings reach inline CSS where `url(...)` fires network requests. -- Strings and arrays get max lengths/sizes. New collection fields need caps too. - -## Change protocol -1. Add the field with a **default** (old files must keep parsing) and mirror the type in `types.ts` exports. -2. `npm run build:edl` after every schema change — the app and scripts import from `dist/`. -3. Renderer must tolerate EDLs parsed by an older schema (a stale main process across HMR): nullish-fallback new theme/track fields at the use site. -4. Add hostile-input coverage to `schema.test.ts` for anything new (reject Infinity/traversal/oversize, accept the sane case). -5. If LLM output could violate the new constraint, teach `sanitizeEdl` (app/scripts/edl-util.mjs) to repair it. diff --git a/.cursor/rules/engine-scripts.mdc b/.cursor/rules/engine-scripts.mdc index c962f2f..992f1f9 100644 --- a/.cursor/rules/engine-scripts.mdc +++ b/.cursor/rules/engine-scripts.mdc @@ -1,27 +1,31 @@ --- -description: Engine script conventions (app/scripts — protocol, LLM, safety) +description: Engine script conventions globs: app/scripts/** -alwaysApply: false --- # Engine script conventions (app/scripts) ## Contract with the app -- Plain Node ESM (`.mjs`), spawned by the main process. Communicate over stdout with line protocol: `PHASE `, `PROGRESS <0-100>`, `DONE `; errors to stderr as `ERROR ` + non-zero exit. -- Resolve the project dir from `process.env.APERTURE_PROJECTS_DIR` (fallback `/projects`), args via `--slug ` style flags. -- `edl.json` read from disk is untrusted (shareable file): re-check any path derived from it stays inside the project dir before touching the filesystem or ffmpeg. +- Plain Node ESM (`.mjs`), spawned by the main process (Node >= 22.13 for `node:sqlite`). Communicate over stdout with line protocol: `PHASE `, `PROGRESS <0-100>`, `DONE `; errors to stderr as `ERROR ` + non-zero exit. +- Resolve the library home via `lib/paths.mjs` (`resolveHome()` honors `KEEPER_LIBRARY_DIR`, default `~/Pictures/Keeper`); accept `--library ` for overrides. +- All catalog access goes through `lib/catalog.mjs` (`openCatalog`). Rows are re-validated against `@keeper/schema` on read AND write — catalog rows are untrusted (agents and LLMs write them). Any path from a row must be confined with `safeJoin(home, rel)` before touching the filesystem or ffmpeg. +- CLI/pipeline contexts open the catalog with `{ stampOnWrite: true }` so the running app refreshes; the app's own catalog service must NOT stamp (it would refresh itself in a loop). +- Pipeline stages checkpoint per asset (`markStage`) so interrupted runs resume via `reprocess.mjs`. ## LLM calls -- Always go through `llm.mjs` (`resolveModel()`, `isLlmConfigured()`, `reasoningEffort()`); never instantiate a provider directly. +- Always go through `llm.mjs` (`resolveModel()`, `isLlmConfigured()`, `reasoningEffort()`); never instantiate a provider directly. Env: `KEEPER_LLM_PROVIDER/MODEL/BASE_URL/API_KEY`. - Cap output: `maxOutputTokens` set explicitly. `providerOptions: { openai: { reasoningEffort: reasoningEffort() } }`. -- Model output is repaired, not trusted: `extractJson` → `sanitizeEdl` → `parseEdl`, with one retry carrying the validation error back to the model. -- Deterministic insurance after every whole-EDL model response: `enforceStyle` (look) and `restoreAudioTracks` (never lose music/voiceover). +- Model output is repaired, not trusted: `extractJson` -> `sanitizeJudge` -> zod parse (`lib/llm-util.mjs`), with one retry carrying the validation error back to the model. Hallucinated asset ids are dropped. +- Respect the budget: the judge never analyzes more than `--budget`/`KEEPER_AI_BUDGET` items, and only sends downscaled thumbnails, never originals. ## Media -- ffmpeg = `ffmpeg-static` import; spawn argv arrays only (no shell strings). -- Voice/TTS audio must be loudness-normalized (two-pass loudnorm, −14 LUFS / −1.5 dBTP) before landing in a project. -- Cache expensive synth/analysis keyed by content hash so re-runs don't re-burn API credits. +- ffmpeg = `ffmpeg-static` import; spawn argv arrays only (no shell strings). No ffprobe exists — durations come from exiftool or the ffmpeg stderr banner trick. +- Pixel math (blur, exposure, pHash) happens in pure JS over grayscale buffers piped from ffmpeg (`extractGray`) — no native image libraries. +- Decode chain for stills: jpeg/png/webp/tiff via ffmpeg; RAW via exiftool embedded-preview extraction; HEIC via `sips` on macOS (ffmpeg fallback). Extend `lib/media.mjs`, don't inline decoders. + +## Safety +- Scripts never delete or move user media. The only deletion path is the app's "empty rejects" (OS trash, user-confirmed). ## Testing -- Pure logic lives in `*-util.mjs` with a sibling `*.test.mjs` (vitest node project). Scripts themselves stay thin. +- Pure logic lives in `lib/*.mjs` with a sibling `*.test.mjs` (vitest node project). Scripts themselves stay thin. - `node --check app/scripts/.mjs` before committing. diff --git a/.cursor/rules/ipc-conventions.mdc b/.cursor/rules/ipc-conventions.mdc index 90e73a6..f744218 100644 --- a/.cursor/rules/ipc-conventions.mdc +++ b/.cursor/rules/ipc-conventions.mdc @@ -1,34 +1,32 @@ --- -description: Main-process + IPC conventions (path guards, result shapes, keys) +description: Main process & IPC conventions globs: app/src/main/**,app/src/preload/** -alwaysApply: false --- # Main process & IPC conventions ## Every renderer-supplied path/id is untrusted -- Filesystem access keyed by a slug/id MUST go through `safeProjectPath()` / `safeStylePath()` (they throw on escape; they compare against `root + sep`). -- Validate id shape at the IPC boundary before spawning scripts: `/^[a-z0-9][a-z0-9_-]{0,63}$/i` (see `runScript`). -- Filenames from the renderer: reject anything where `basename(file) !== file`. +- Filesystem access under the library MUST go through `safeHomePath()` (throws on escape; compares against `root + sep`). +- Asset ids are validated at the IPC boundary with `sanitizeIds()` (`/^[a-zA-Z0-9][a-zA-Z0-9._-]{0,127}$/`); enum-ish params (flags, media types) are allowlisted before forwarding anywhere. +- The main process never trusts catalog contents either — paths coming back from the catalog service still go through `safeHomePath()` before `shell.trashItem`/`showItemInFolder`. ## Result shape -Handlers never throw across IPC. Return `{ ok: boolean, error?: string }` (plus payload fields): +Handlers never throw across IPC. Return `{ ok: boolean, error?: string }` (plus payload fields). Service-backed handlers use `serviceResult()` which wraps catalog-service calls into `{ ok, result?, error? }`. -```ts -ipcMain.handle("thing:do", (_e, slug: string) => { - try { ... return { ok: true }; } - catch (err) { return { ok: false, error: String(err) }; } -}); -``` +## The catalog service +- SQLite access lives in a spawned Node child (`app/scripts/catalog-service.mjs`, line-delimited JSON-RPC over stdio) — the Electron main process must stay free of native/DB modules. Talk to it via `callService(method, params)`; it lazily respawns on exit. +- New catalog queries: add a method to the service + a thin validated IPC handler + a preload mirror. Don't put query logic in main. ## Long-running work -- Spawn engine scripts via `runScript`/`runScriptArgs` — they stream the script's `PHASE`/`PROGRESS` lines to `${prefix}:phase` / `${prefix}:progress` channels. -- Guard event senders: check `event.sender.isDestroyed()` before `.send()` in delayed callbacks. +- Spawn engine scripts via `runScriptArgs` — it streams the script's `PHASE`/`PROGRESS` lines to `${prefix}:progress` / `${prefix}:phase` channels. Guard `event.sender.isDestroyed()` before `.send()`. +- External writes (agent CLIs, pipeline scripts) touch `/.keeper/.stamp`; the stamp watcher debounces and pushes `library:changed` so the renderer refreshes. The app's own service never stamps. ## Secrets & settings -- API keys live in `AppSettings` (settings.json) AND are injected into `process.env` via `applyAgentEnv` so spawned scripts inherit them. Env vars from `.env.local`/shell always win — capture the lock in `envLocked` once at startup and never overwrite locked vars. -- New settings fields: parse defensively in `readSettings` (unknown JSON on disk), update the mirror `AppSettings` interface in `app/src/preload/index.ts`. +- API keys live in `AppSettings` (settings.json) AND are injected into `process.env` via `applyAgentEnv` (`KEEPER_LLM_*`) so spawned scripts inherit them. Env vars from `.env.local`/shell always win — the lock is captured in `envLocked` once at startup; never overwrite locked vars. +- New settings fields: parse defensively in `readSettings` (unknown JSON on disk), mirror the `AppSettings` interface in `app/src/preload/index.ts`. + +## Destructive operations +- The ONLY deletion path is `rejects:empty`: user-confirmed in the UI, moves originals to the OS trash via `shell.trashItem` (recoverable), then forgets rows. Never add a hard-delete IPC channel; never let a script or AI verdict trigger deletion. ## Remember - Main-process changes need a dev-app restart (renderer HMR does not reload main/preload). Say so in the PR/summary. -- Writes to a project's `edl.json` from the app must set `lastSelfWrite` (via `writeEdl`) or the file watcher echoes a reload that clobbers editor state. diff --git a/.cursor/rules/schema.mdc b/.cursor/rules/schema.mdc new file mode 100644 index 0000000..65c6293 --- /dev/null +++ b/.cursor/rules/schema.mdc @@ -0,0 +1,20 @@ +--- +description: Shared schema package conventions +globs: packages/schema/** +--- + +# Shared schemas (packages/schema) + +The schemas are a **security boundary**: catalog rows, taste.json, and LLM judge output are written by agents and models and re-read everywhere. Every field must stay hostile-input-safe. + +- Numerics: `.finite()` + explicit bounds (`MAX_DURATION_SEC`; JSON `1e400` parses to `Infinity` and wedges sorting/layout). +- Paths (`relPath`, `thumbRel`, …): `RelativePathSchema` — no `..`, absolute paths, drive letters, UNC, or NUL. Consumers still confine with `safeJoin`/`safeHomePath` before filesystem use. +- Ids: `IdSchema` (alphanumeric + `._-`, capped) — ids appear in SQL parameters, file names, and URLs. +- Strings and arrays get max lengths/sizes. New collection fields need caps too. + +## Change protocol +1. Add the field with a **default** (old catalogs must keep parsing) and mirror the type in `types.ts`. +2. `npm run build:schema` after every change — the app and scripts import from `dist/`. +3. If it's a catalog column worth filtering on, add it to the hot columns in `app/scripts/lib/catalog.mjs` (with a migration guarded by `PRAGMA user_version`); otherwise it rides in a JSON side-column. +4. Add hostile-input coverage to `schema.test.ts` (reject Infinity/traversal/oversize, accept the sane case). +5. If LLM judge output could violate the new constraint, teach `sanitizeJudge` (`app/scripts/lib/llm-util.mjs`) to repair it. diff --git a/.cursor/rules/workflow.mdc b/.cursor/rules/workflow.mdc index 5c671e5..d20d4cf 100644 --- a/.cursor/rules/workflow.mdc +++ b/.cursor/rules/workflow.mdc @@ -8,7 +8,7 @@ alwaysApply: true - Branch per change off `main` (`fix/…`, `feat/…`, `polish/…`, `docs/…`). Never commit to `main` directly; never stack PRs unless explicitly agreed (merge-order accidents have bitten us twice). - Gate before every push: `npm run typecheck && npm test && npm run build` (build matters — it resolves CSS/asset imports that typecheck and vitest miss). - Push over HTTPS with the gh credential helper (SSH agent is flaky here): - `git -c credential.helper= -c credential.helper='!gh auth git-credential' push https://github.com/thisiscsim/aperture.git HEAD:` + `git -c credential.helper= -c credential.helper='!gh auth git-credential' push https://github.com/thisiscsim/keeper.git HEAD:` - PR flow: open with a Summary + Test plan, wait for `gh pr checks --watch`, merge with `--merge --delete-branch`, then ff-sync local `main` from the same HTTPS remote. - Call out in the summary when a change requires restarting `npm run dev` (main-process/preload/engine-script changes — HMR only covers the renderer). -- Generated artifacts live under the user's `~/Documents/Aperture` (never the repo). `/projects/` and `/styles/` in the repo root are gitignored, root-anchored on purpose. +- User data (the media library) lives under `~/Pictures/Keeper` (never the repo). `/library/`, `/.keeper/`, and `/taste.json` in the repo root are gitignored, root-anchored on purpose. diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index 25e1f2d..b620f74 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -18,6 +18,7 @@ ## Checklist -- [ ] Any `edl.json` produced still validates against `EdlSchema` +- [ ] Catalog rows / taste.json still validate against `@keeper/schema` (hostile-input tests updated if the schema changed) - [ ] No secrets committed (`app/.env.local`, keys) and no large/uploaded media - [ ] `CHANGELOG.md` updated under "Unreleased" if user-facing +- [ ] Called out if the change needs a `npm run dev` restart (main/preload/engine scripts) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c7cdf46..e214c5a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -19,8 +19,8 @@ jobs: - name: Install dependencies run: npm ci - - name: Build shared EDL package - run: npm run build:edl + - name: Build shared schema package + run: npm run build:schema - name: Typecheck run: npm run typecheck diff --git a/.gitignore b/.gitignore index 85a4da8..4022ffb 100644 --- a/.gitignore +++ b/.gitignore @@ -4,13 +4,14 @@ dist/ .DS_Store *.log -# User data lives in ~/Documents/Aperture (configurable), never in the repo. -# Any local projects/ or styles/ (e.g. from a dev APERTURE_HOME override) are ignored. -# The tracked sample project lives in fixtures/sample-project instead. -# NOTE: root-anchored — an unanchored "styles/" also matched -# app/src/renderer/src/styles/ and silently kept the design system untracked. -/projects/ -/styles/ +# User data (the media library) lives in ~/Pictures/Keeper (configurable), +# never in the repo. A local library/ from a dev KEEPER_LIBRARY_DIR override +# pointed at the repo is ignored too. +# NOTE: root-anchored — an unanchored pattern would also match +# app/src/renderer/src/styles/ and silently untrack the design system. +/library/ +/.keeper/ +/taste.json .env .env.local diff --git a/AGENTS.md b/AGENTS.md index 2db3340..96f01b5 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,50 +1,38 @@ # Keeper -> NOTE: scaffolded from Aperture — the conventions below describe the inherited codebase; update as Keeper diverges. - -AI-assisted short-form video studio. You (the agent) turn a prompt + raw clips into a finished vertical social video by writing a declarative timeline (`edl.json`), which a local Electron editor previews and exports via Remotion. +AI-assisted culling for photos and videos. The user dumps an SD card / phone folder in; Keeper copies it into a local library, flags the junk with explainable verdicts, groups bursts, indexes everything for natural-language search, and learns the user's taste from every override. You (the agent) are the richest tier of that pipeline: you can see images, so you can judge what heuristics can't. ## How it works -The full creator journey (front to back): - -1. The user opens the editor on a project **homepage** and creates a project under `projects//` (scaffolds `meta.json`, `prompt.md`, an empty `edl.json`, and `assets/ references/ benchmarks/ transcripts/ renders/`). -2. In the editor they provide input: upload clips into `assets/`, write intent in `prompt.md`, attach music, and add or record a voiceover (which auto-transcribes to word-level captions). -3. Optionally they teach the agent their look: upload their own past videos into `references/` and run aesthetic learning, which writes a reusable `style.json` profile (+ `aesthetic.md`). -4. You generate the first cut by writing `projects//edl.json` — the single source of truth — conditioned on `prompt.md` and `style.json`. -5. The Electron editor live-previews `edl.json` and live-reloads it when you (or the user) change it. The user refines on the timeline; their edits autosave back to `edl.json`. -6. You critique the cut into `critique.json`, calibrated against the creator's own high-performers in `benchmarks.json` when present. The `auto-tune` loop iterates generate -> critique -> fix, logging `results.tsv`. - -## The contract: edl.json (+ sidecar files) - -`edl.json` is validated by the zod schema in `packages/edl` (`packages/edl/src/schema.ts`). Never write an `edl.json` that fails `EdlSchema`. - -Shape: `format` (vertical 1080x1920, fps), `theme` (font, palette, captionStyle, safeMargins, optional `stylePreset`), `assets[]`, `tracks[]` where each track is `video | text | caption | audio`. Audio clips carry a `role` (`music | voiceover | sfx`); music with `duckUnderVoice` is attenuated under voiceover. +1. **Import** — `import.mjs` copies files into `/library/YYYY/YYYY-MM-DD/` (SHA-256 checksummed, deduped against the whole library), reads EXIF via exiftool, generates thumbnails/previews (RAW via embedded preview, HEIC via sips, videos get posters + scrub strips), measures quality (blur/exposure via ffmpeg-piped grayscale + pure-JS math), suggests verdicts, clusters bursts (pHash + capture time), pairs RAW+JPEG / Live Photos, and embeds everything with local CLIP for search. +2. **Review** — the Electron app shows three confidence queues (sure rejects / sure keeps / needs your eye) with reason chips and evidence; the user (or you) confirms or overrides. User verdicts are separate from AI suggestions; the AI never decides destructively. +3. **Learn** — every override of an AI suggestion becomes an exemplar in `taste.json` (+ threshold tuning); standing rules and recent corrections are injected into future LLM judgments. +4. **Search** — CLIP embeddings, fully local; the LLM judge adds captions/tags for borderline items when configured. +5. **Export** — picks copy out with `.xmp` sidecars (rating/flag/keywords) that Lightroom/Capture One/Bridge read natively; rejected originals are only ever moved to the OS trash after explicit user confirmation. -Per-project sidecar files (each has its own schema + `parse*` helper in `packages/edl`): +## The contract -- `meta.json` (`MetaSchema`) — title, platform, status, `styleProfileId`. -- `style.json` (`StyleProfileSchema`) — learned/selected aesthetic: palette, font, captions, pacing, hook, energy, do/avoid. -- `benchmarks.json` (`BenchmarksSchema`) — feature distribution of the creator's high-performers, for benchmark-relative critique. +- **Catalog**: `/.keeper/catalog.db` (SQLite). All access goes through `app/scripts/lib/catalog.mjs`; every row is validated against the zod schemas in `packages/schema` on read and write. Never write the DB directly. +- **Taste**: `/taste.json` (`TasteProfileSchema`) — thresholds, standing rules, exemplars, override stats. +- Library home: `KEEPER_LIBRARY_DIR` env, else `~/Pictures/Keeper`. CLI writes touch `/.keeper/.stamp`, which live-refreshes the running app. ## Skills -- `/create-social-video ` — analyze clips + prompt (+ `style.json`), write `edl.json` (first cut). -- `/learn-aesthetic ` — study the creator's `references/`, write `style.json` + `aesthetic.md`. -- `/critique-video ` — score the cut (vs `benchmarks.json` when present), write `critique.json`. -- `/auto-tune ` — loop generate/adjust -> critique -> fix, logging `results.tsv`. +- `/cull-shoot` — import + confirm/override verdicts + resolve bursts, visually. +- `/find-media` — semantic search + visual verification of a half-remembered shot. +- `/organize-library` — captions/tags/events, dedupe report, library health. -## Helper scripts (`app/scripts/`) +## CLI surface (`app/scripts/`) -`analyze.mjs` (baseline assembly), `transcribe.mjs` (captions, prefers the voiceover clip), `render.mjs` (export), `extract-frames.mjs` + `analyze-style.mjs` (aesthetic baseline), `analyze-benchmarks.mjs` (benchmark features), `autotune.mjs` (deterministic auto-improve). +`import.mjs` (full pipeline), `reprocess.mjs` (resume/re-run stages), `query.mjs` (counts / lists / review queues / semantic search — JSON out), `verdict.mjs` (set flags/ratings/group picks — records taste like the app), `judge-llm.mjs` (budgeted GPT-5.5 vision pass over borderline items), `export.mjs` (copy + XMP sidecars), `catalog-service.mjs` (the app's own DB broker — not for direct use). ## Scoped conventions -Area-specific rules live in `.cursor/rules/` (IPC/main-process, renderer design system, engine scripts, EDL schema, delivery workflow). They activate by file glob in Cursor; other agents should skim the relevant file before working in that area. +Area rules live in `.cursor/rules/` (IPC/main process, renderer design system, engine scripts, schema package, delivery workflow). They activate by file glob in Cursor; other agents should skim the relevant file before working in that area. ## Boundaries -- Generated artifacts live under `projects//`. In the app these resolve to the user's Aperture home (`~/Documents/Aperture/projects/`, configurable); the scripts honor `APERTURE_PROJECTS_DIR` and fall back to the repo's `projects/` in dev. Don't write outside a project folder except code changes you were explicitly asked to make. -- Vertical 1080x1920 @ 30fps is the default format. -- Keep the design system lightweight: font, palette, caption style, simple overlays — all driven by `theme`. Don't hardcode styling that belongs in `theme`. -- Only reference assets that actually exist in the project's `assets/` and are listed in `edl.assets`. +- **Never delete, move, or rewrite media files.** Verdicts are flags. The only deletion path is the app's user-confirmed "empty rejects" (OS trash). +- Media stays local. The LLM judge sends downscaled thumbnails only, capped by an explicit budget; embeddings and quality metrics never leave the machine. +- Requires Node >= 22.13 (`node:sqlite`). Scripts are spawned with the system `node`. +- Don't write outside the library home except code changes you were explicitly asked to make. diff --git a/CHANGELOG.md b/CHANGELOG.md index ceb4a03..1e941c5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,6 @@ # Changelog -All notable changes to Aperture are documented here. The format is based on +All notable changes to Keeper are documented here. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); the project is pre-release, so everything lives under "Unreleased" until we start tagging versions. @@ -9,75 +9,66 @@ versions. ### Added -- **Editor redesign (Figma V0, 4 phases)** — new shell (header with centered - filename, undo/redo, presets dropdown, export; left input rail with prompt + - generate, clips and audio upload/record; floating device preview; pill-tab - right panel), a combined Inspector (Design: alignment/padding/typography/ - palette/captions; Format: fps, aspect ratio, resolution; Back-headed clip - subflows for text/video/audio), Style tab with reference modes - (literal/inspired) and read-only style guide, Critique tab with score card + - detail subflow and Auto-improve, timeline rework (dynamic renamable layers, - music/voiceover split tracks, Layer button, transport bar with Space - play/pause, action-token chips, text-sketch and click-to-add empty lanes, - drag assets from the rail), EDL undo/redo history (Cmd+Z / Shift+Cmd+Z) and - a tabbed Settings modal (General / Export / Agent). Schema gains - `theme.textAlignment`, `styleProfile.referenceMode`, and optional track - `name`; generation prompts are format-aware. - -- **Visual language refactor (Home + New Project dialog)** — new design-token - system (`styles/tokens.css`, light authoritative + derived dark), Bradford - brand font scaffolding with SF Pro as the UI stack, a custom icon set - harvested from Figma (38 SVGs, `currentColor`-normalized) behind a typed - `Icon` component, componentized UI primitives (`Button`, `IconButton`, - `Badge`, `Modal`, `Field`/`Input`/`TextArea`/`Select`), a Figma-faithful Home - page (header, welcome hero, 240x300 project cards with status badges + - relative time, dashed new-project card) and restyled New Project dialog. The - editor is retinted onto the new tokens pending its own redesign. - -- **Testing foundation** — Vitest with two projects (Node for `packages/edl` + - `app/scripts`, jsdom for the renderer), covering the EDL schema, the - `sanitizeEdl`/`enforceStyle`/`metrics` helpers, critique scoring, EDL edits, - text animations, the store (autosave/routing/theme), and a `LeftPanel` render - regression guard for the rules-of-hooks crash. `npm test` / `npm run test:watch`. -- **CI** — GitHub Actions workflow running `npm ci`, build, typecheck, and tests - on every pull request and push to `main`. -- **Repo hygiene** — this changelog and a pull request template. -- **Style Library** — a reusable, creator-level look: bulk-import a folder of - reference videos (native picker), analyze once (`analyze-collection.mjs`, - GPT-5.5 vision distills a style guide + per-reference exemplars, with a - deterministic fallback), and reuse across projects; per-project `references/` - override when present. -- **Color grade** — `theme.grade` (brightness/contrast/saturation/temperature/ - vignette) rendered as a CSS filter on clips in preview and export. -- **LLM everywhere** — provider-agnostic layer (Vercel AI SDK, default OpenAI - GPT-5.5, env-configurable) powering Generate, Critique, and Auto-improve, each - with an offline deterministic fallback. Local, gitignored `app/.env.local`. -- **Creator pipeline** — project homepage (create/open/delete, thumbnails), - clip upload, editable prompt, music attach + bundled library, voiceover - upload/record with auto-transcribed captions and music ducking, per-project - aesthetic learning, named style presets, and benchmark-aware critique. -- **Editor/platform** — `edl.json` autosave + file-watch live reload, light/dark - theme, root error boundary, and toasts. +- **Keeper v1 — AI-assisted culling for photos and videos.** The repo pivots + from the inherited Aperture video-studio scaffold to a media-culling app: + - **Library**: one date-sorted local library (`~/Pictures/Keeper` by + default); imports copy in checksummed + deduplicated, with EXIF via + exiftool, thumbnails/previews for JPEG/PNG/HEIC/RAW (embedded preview), + and posters + hover-scrub strips for video. + - **Auto-cull pipeline**: local blur/exposure/black-frame/corrupt/accidental + -clip/screenshot detection (ffmpeg-piped grayscale + pure-JS math), burst + and near-duplicate grouping (pHash + capture time) with best-frame picks, + RAW+JPEG and Live Photo pairing — every suggestion with reasons and + confidence, checkpointed and resumable. + - **Review mode**: three confidence queues (sure rejects / sure keeps / + needs your eye) with evidence (3x focus crop, sharper-twin comparison), + bulk confirm, keyboard-first culling (P/X/U, 0–5 stars, Space loupe), + and full undo/redo. + - **Natural-language search**: local CLIP embeddings (transformers.js / + onnxruntime, model cached in the library), instant offline cosine search + with date-word narrowing; metadata fallback before the index is built. + - **LLM judge** (optional): budgeted, batched GPT-5.5 vision over borderline + items only — downscaled thumbnails, repaired-then-validated output, + captions + tags, burst best-pick refinement; provider-agnostic via + `KEEPER_LLM_*` env or Settings. + - **Taste profile**: overrides become exemplars + threshold tuning in + `taste.json`; standing rules injected into every AI run; fully + user-inspectable in the app. + - **Export/interop**: copy picks (with RAW/Live siblings) to a folder with + Lightroom-readable `.xmp` sidecars, or write sidecars in place; reveal in + Finder; two-step "empty rejects" that only ever moves originals to the OS + Trash. + - **Agent tier**: `/cull-shoot`, `/find-media`, `/organize-library` skills + + JSON CLIs (`query.mjs`, `verdict.mjs`) that read/write the same catalog + with the same taste feedback. +- **Catalog**: SQLite (`node:sqlite`, no native build step) at + `.keeper/catalog.db`, brokered to the app by a spawned catalog service + (line-delimited JSON-RPC) so the Electron main process stays DB-free; all + rows validated against the new `@keeper/schema` zod contracts on read and + write. ### Changed -- Rebranded **Reel Studio -> Aperture** (window title, macOS app/dock name, icon, - docs). -- Generation is style-faithful: injects the style guide + top exemplars and - deterministically stamps palette/font/captions/grade. -- EDL package: added `meta`, `style`, and `benchmark` schemas; audio-clip `role`; - `theme.stylePreset` and `theme.grade`; fixed the ESM build so the schema - imports cleanly from Node scripts. +- `packages/edl` (`@reel/edl`) is now `packages/schema` (`@keeper/schema`): + asset records, verdicts, groups, taste profile, import manifests, search and + judge I/O — same hostile-input discipline (bounded numerics, confined + relative paths, capped collections). +- Electron main process rebuilt around the library: `keeper-asset://` byte-range + streaming protocol, stamp-file watcher for agent live-refresh, `KEEPER_*` + env names, settings (library location, AI budget, auto-judge, hardware + decode). +- Renderer rebuilt on the same design tokens + UI kit: date-sectioned lazy + grid, loupe with burst strip, review queues, search, taste/export/rejects/ + settings modals; zustand store with optimistic verdicts and undo history. -### Fixed +### Removed -- Rules-of-hooks crash that blanked the editor when a project loaded. -- Generation silently falling back to the baseline when the model omitted a - required `anim.name` (now repaired by `sanitizeEdl`). -- Reasoning-model incompatibility (dropped unsupported `temperature`). +- The video-editor surface area: Remotion preview/export, the timeline editor, + EDL schema and engine scripts (generate/critique/autotune/transcribe/TTS), + style library, ElevenLabs integration, and the bundled music resources. ## [0.1.0] - Initial commit -- Scaffold: Electron + Vite + React editor, shared `packages/edl` schema, - Remotion preview/export spine, and the `create-social-video` / - `critique-video` Claude Code skills. +- Scaffolded from the Aperture video studio (Electron + Vite + React editor, + shared schema package, engine-script + skills architecture, design tokens, + Vitest + CI). diff --git a/CLAUDE.md b/CLAUDE.md index 1da6dc6..0f0670a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,3 +1,3 @@ -# Aperture — Claude Code +# Keeper — Claude Code > Shared project rules live in `AGENTS.md`. diff --git a/README.md b/README.md index f07f08e..6d05f09 100644 --- a/README.md +++ b/README.md @@ -1,157 +1,96 @@ # Keeper -> Scaffolded from the Aperture video studio. Docs below describe the inherited stack (Electron + Vite + React, design tokens + UI kit, EDL schema package, engine scripts, Vitest + CI) until Keeper's own docs replace them. +AI-assisted culling for photos and videos. Dump in an SD card, a phone folder, or two weeks of vacation chaos — Keeper copies everything into a local, date-sorted library, flags the junk with reasons you can check, groups bursts and picks the best frame, lets you search your whole library in plain English, and learns your taste from every call you overrule. When you're done, your keepers flow out to Lightroom, Capture One, or any folder — with your ratings attached. -AI-assisted short-form video studio. Create a project, drop in your clips, write what you want, let it learn your aesthetic, generate a first cut, refine it on a timeline, critique it against your own best posts, auto-improve it, and export a vertical MP4. +Keeper is **local-first, not local-only**: importing, quality analysis, burst grouping, semantic search, preview, and export all run on your machine. The one optional cloud step is the AI judge (GPT-5.5 by default, provider-configurable), which only ever sees downscaled thumbnails of the borderline items, capped by a budget you set. With no API key, everything still works — you just review the borderline pile yourself. -Aperture is **local-first, not local-only**: your media, editing, transcription, and export all run on your machine, but the AI steps (generate, critique, auto-improve) call a configurable LLM API (OpenAI GPT-5.5 by default). If no model is configured, those steps fall back to fully-offline deterministic versions. +## The flow -## The end-to-end flow - -1. **Home** — a project dashboard. Create a project (name + prompt + platform) or open/delete an existing one (per-card ⋯ menu). -2. **Input** — upload clips (drag/drop), edit the prompt, attach music, and upload or record a voiceover (auto-transcribed to word-level captions; music ducks under voice). -3. **Learn aesthetic** (optional) — build a reusable **Style Library**: bulk-import a whole folder of your past videos once, and Aperture distills their palette, grade, pacing, hook, and text treatment into a profile (with a prose style guide + per-reference exemplars). Point any project at a library profile, or learn from a project's own references as an override. Built-in named style presets are also available. -4. **Generate** — produces a real first cut (hook, reordering, titles, transitions, palette + color grade) conditioned on your prompt and the active style profile. LLM-powered when configured; deterministic baseline otherwise. -5. **Refine** — a timeline editor with live Remotion preview. Every edit autosaves to `edl.json`; external writes live-reload. -6. **Critique** — score the cut, calibrated against your own uploaded high-performers ("you vs your best"). LLM critique or instant offline heuristic. -7. **Auto-improve** — a generate → critique → improve loop that iterates the edit and logs the score trajectory. -8. **Export** — render a vertical 1080×1920 MP4 locally via Remotion. +1. **Import** — pick files or a whole folder. Keeper copies (checksum-verified, deduplicated against your whole library) into `library/YYYY/YYYY-MM-DD/`, reads EXIF, and generates thumbnails and previews — including RAW (embedded previews), HEIC, and video posters with hover-scrub strips. +2. **Auto-cull** — a local pass measures sharpness and exposure, catches black frames, corrupt files, sub-second accidental clips, and screenshots; bursts and near-duplicates are grouped (perceptual hash + capture time) with the sharpest frame suggested as the pick; RAW+JPEG pairs and Live Photos are treated as one photo. Every suggestion carries a reason and a confidence. +3. **AI review** (optional) — the configured model looks at just the borderline items and judges what heuristics can't: eyes closed, bad framing, or the opposite — a technically imperfect shot of a moment that matters. Your standing rules and past corrections ride along in the prompt. +4. **Review** — three queues: *sure rejects* (spot-check, confirm in bulk), *sure keeps*, and *needs your eye*. Evidence is attached — a 3× focus crop, the sharper twin side-by-side. Keyboard-first: P/K keep, X/R reject, 0–5 stars, Space for the loupe, arrows to move. Everything is undoable (Cmd+Z). +5. **Search** — "ocean at sunset", "the kids at dinner" — a local CLIP index answers instantly and offline; month/year words narrow the range. AI captions and tags (when the judge has run) make results explainable. +6. **Taste** — every override becomes a labeled example: thresholds adapt (keep overriding blur rejects and the blur bar loosens), and recent corrections + your standing rules ("never auto-reject photos of my kids") steer future AI runs. Inspect it all under *Taste profile* — it's just `taste.json` in your library. +7. **Export** — copy picks to a folder with `.xmp` sidecars (rating, reject flag, keywords) that Lightroom, Capture One, and Bridge read on import — or write sidecars in place and point your editor at the library. RAW twins and Live Photo videos travel with their picks. +8. **Empty rejects** — the only destructive action in the app: explicit, two-step, shows count and size, and moves originals to the OS Trash (recoverable). The AI can never delete anything. ## Architecture -Three layers, bridged by one file per project: - -- **Electron editor** (`app/`) — homepage + timeline UI + live preview (Remotion Player) + export (Remotion renderer). -- **Node scripts** (`app/scripts/`) — the engine: clip probing/assembly, transcription, frame/style/benchmark analysis, the LLM generate/critique/auto-improve calls, and rendering. -- **Agent skills** (`.claude/skills/`, `AGENTS.md`) — the richest path: skills run from a Claude/Cursor harness that read/write the same project files. +Three layers around one contract: -**The contract:** each video is a folder under `projects//` whose `edl.json` (validated by the zod schema in `packages/edl`) is the single source of truth. Generators write it; the editor previews/edits/autosaves it; the renderer exports it. +- **Electron app** (`app/`) — the library grid, review queues, loupe, search, and settings. The main process owns IPC, a streaming `keeper-asset://` protocol for media, and spawns everything else. +- **Engine scripts** (`app/scripts/`) — plain Node: the import pipeline (copy → derive → CV → group → embed), the LLM judge, search, export, and CLI tools for agents. A long-lived catalog service brokers SQLite access for the app. +- **Agent skills** (`.claude/skills/`) — `/cull-shoot`, `/find-media`, `/organize-library`: an agent harness (Claude/Cursor) works the same catalog through the same CLIs, and can actually look at your photos when judgment is needed. -Every "smart" step exists at three tiers and the app picks the best available: deterministic script (offline, free) → single LLM call (cost-predictable) → agent skill (richest). +**The contract:** the catalog (`.keeper/catalog.db`, SQLite) plus `taste.json`, both validated by the zod schemas in `packages/schema`. Every "smart" step has three tiers — deterministic local pass → single budgeted LLM call → agent skill — and the app uses the best one available. ``` -aperture/ +keeper/ app/ src/ Electron main + preload + React renderer - scripts/ analyze, transcribe, render, extract-frames, - analyze-style, analyze-collection, analyze-benchmarks, - generate-llm, critique-llm, autotune(-llm), llm - resources/ app icon, bundled music - packages/edl/ Shared EDL + meta/style/benchmark schemas (zod) - .claude/skills/ create-social-video, learn-aesthetic, - critique-video, auto-tune - projects// meta.json, prompt.md, assets/, edl.json, - style.json, references/, benchmarks/, - benchmarks.json, transcripts/, critique.json, renders/ - styles// Global Style Library (gitignored): profile.json, - style-guide.md, sources/, .frames/ + scripts/ import, reprocess, judge-llm, export, + query, verdict, catalog-service, llm + + lib/ (catalog, media, quality, phash, + grouping, embeddings, taste, xmp) + packages/schema/ Shared zod contracts (assets, verdicts, + groups, taste, imports, judge I/O) + .claude/skills/ cull-shoot, find-media, organize-library AGENTS.md Agent operating manual ``` -## Style Library - -Rather than re-uploading references per project, build a creator-level look once and reuse it everywhere (Style tab): +## Where your media lives -- **Bulk import** a whole folder (or multi-select) via a native picker. -- **Analyze once** — samples frames and computes editing metrics, then (with a model configured) distills a prose style guide + per-reference exemplars the generator imitates in-context. Without a model it still writes a solid deterministic profile. -- **Reuse** — a project points at a library profile via `meta.styleProfileId`; a project's own `references/` override the library when present. -- **Faithful generation** — generation injects the style guide + top exemplars and then deterministically stamps the measurable look (palette, font, caption style, and a light color grade rendered as a CSS filter on your clips). +The library is **user data, never the repo**: `~/Pictures/Keeper` by default (change it in Settings; dev override `KEEPER_LIBRARY_DIR`). -Scope note: this matches edit structure, captions, text, transitions, palette, and a light grade — not footage transformation (no LUT/effects/AI restyle of the source pixels). +``` +~/Pictures/Keeper/ + library/2026/2026-07-04/ your originals, immutable, date-sorted + .keeper/ catalog.db, thumbs, previews, models + taste.json your learned culling profile +``` ## AI configuration -The AI steps use the Vercel AI SDK behind a provider-agnostic layer (`app/scripts/llm.mjs`). Configure it with a local, gitignored env file — copy `app/.env.local.example` to `app/.env.local`: +The judge and search enrichment use the Vercel AI SDK behind a provider-agnostic layer (`app/scripts/llm.mjs`). Add a key in **Settings → AI**, or copy `app/.env.local.example` to `app/.env.local`: ```bash -# app/.env.local (default: OpenAI GPT-5.5) OPENAI_API_KEY=sk-... ``` -The main process loads this at startup, so spawned scripts inherit it (no shell exporting needed). Restart `npm run dev` after changing it. - Escape hatch — rotate models/providers with no code change: ```bash -APERTURE_LLM_PROVIDER=openai # openai | anthropic | openai-compatible -APERTURE_LLM_MODEL=gpt-5.5 -APERTURE_LLM_BASE_URL= # e.g. http://localhost:11434/v1 for a self-hosted model -APERTURE_LLM_API_KEY= # generic; overrides the provider-specific key -``` - -With no key set, Generate falls back to the deterministic baseline assembly and the Critique panel uses the offline heuristic; Auto-improve uses the deterministic fix loop. - -### Voiceover (ElevenLabs) - -Generated voiceovers (left rail → Audio → Generate voiceover) and in-app voice cloning use ElevenLabs. Add a key to the same env file (or paste it under Settings → Voices — the env var wins when both are set): - -```bash -ELEVENLABS_API_KEY=sk_... +KEEPER_LLM_PROVIDER=openai # openai | anthropic | openai-compatible +KEEPER_LLM_MODEL=gpt-5.5 +KEEPER_LLM_BASE_URL= # e.g. http://localhost:11434/v1 for a local model +KEEPER_LLM_API_KEY= # generic; overrides the provider-specific key ``` -When creating the key in the ElevenLabs dashboard, scope it minimally — Aperture only needs: - -| Scope | Level | Used for | -| --- | --- | --- | -| Text to Speech | Access | Narration synthesis (also returns the word timings used for captions) | -| Voices | Read | Listing your voices in the picker | -| Voices | Write | Only if you clone/delete voices in-app | - -Everything else (Dubbing, Projects, Voice Generation, Forced Alignment, Speech to Text, …) can stay on No Access. Voice *cloning* requires a paid ElevenLabs plan and the consent of the person being cloned; without a key, recorded voiceovers and whisper captions still work fully offline. +Env vars always win over Settings (the UI shows locked fields). ### What leaves your machine -Generation, critique, and auto-improve send **text** — the `edl.json` edit plan, your `prompt.md`, the resolved style profile, and benchmark feature stats — never your clips. The one exception is **Style Library analysis**, which sends *sampled still frames* of your reference videos to the model (once per profile) so it can see the aesthetic; your source video/audio files themselves are never uploaded. Clip probing, frame sampling, transcription (whisper.cpp), preview, and export all run locally. With no model configured, nothing leaves your machine. +With no key configured: **nothing**. With a key: only downscaled thumbnails of the borderline items the judge analyzes (capped by *AI budget per run*), plus your standing rules and recent corrections as text. Originals, previews, embeddings, and quality metrics never leave your machine. The CLIP search model (~100 MB) downloads once into `.keeper/models/` and runs offline thereafter. -## Where your work is stored +## RAW / HEIC / video notes -Projects and the style library are **user data**, not part of the repo. By default they live in **`~/Documents/Aperture/`** (`projects/` and `styles/`), so they're never committed or bundled into the app. You can change the location in Settings (gear icon → Projects folder); a restart applies it. Dev overrides: `APERTURE_HOME` (root), `REEL_PROJECTS_DIR`, `APERTURE_STYLES_DIR`. A sample project for development lives in `fixtures/sample-project/`. +- RAW support in v1 means metadata + the embedded JPEG preview (fast, no demosaic); the RAW file itself is treated as the sidecar of its JPEG twin when both exist, and always travels with exports. +- HEIC decodes via `sips` on macOS (ffmpeg fallback elsewhere); Live Photos (HEIC+MOV) are paired into one logical photo. +- Videos are first-class in the pipeline: posters, scrub strips, junk detection (sub-second misfires, screen recordings), embedding of the poster frame for search, and playback in the loupe. ## Develop ```bash -npm install # install workspaces -npm run dev # launch the Electron editor (electron-vite) +npm install # workspaces (Node >= 22.13 required — node:sqlite) +npm run dev # launch the Electron app (electron-vite) npm run typecheck # type-check all workspaces +npm test # vitest (schema, pipeline libs, catalog, renderer store) npm run build # production build ``` -## Status - -V1. The full creator pipeline (homepage → input → aesthetic learning → generate → refine → benchmark critique → auto-improve → export) is implemented, with deterministic offline fallbacks and an LLM path for generation, critique, and auto-improvement. - -## Changelog - -This project is pre-release; all notable changes are grouped below until we start tagging versions. Newest first. +Engine scripts run standalone too — see `AGENTS.md` for the CLI surface (`import.mjs`, `query.mjs`, `verdict.mjs`, …). -### Unreleased - -**Style Library + faithful generation** -- Global, reusable **Style Library** (`styles//`): bulk-import a folder (native picker), analyze once, reuse across projects; per-project `references/` still override. -- Rich multimodal style capture (`analyze-collection.mjs`): frame sampling + editing metrics distilled by the LLM into a prose style guide + per-reference exemplars (with a deterministic fallback). -- Style-faithful generation: `generate-llm.mjs` injects the style guide + top exemplars and deterministically stamps palette, font, caption style, and color grade. -- Color grade: `theme.grade` (brightness/contrast/saturation/temperature/vignette) rendered as a CSS filter on clips in preview and export. - -**LLM everywhere (with offline fallbacks)** -- Provider-agnostic LLM layer on the Vercel AI SDK (`llm.mjs`), default OpenAI GPT-5.5, env-configurable provider/model/base-URL. -- LLM-backed **Generate** (`generate-llm.mjs`), **Critique** (`critique-llm.mjs`), and **Auto-improve** (`autotune-llm.mjs`, a critique-in-the-loop optimizer), each falling back to deterministic scripts when no model is set. -- Local, gitignored `app/.env.local` loading at startup; graceful, surfaced errors (toasts) instead of silent fallback. - -**Creator pipeline (front + back of the journey)** -- Project **homepage/dashboard**: create, open, delete (per-card ⋯ menu), thumbnails, view routing. -- **Creator input**: clip upload (drag/drop), editable prompt (`prompt.md`), attach music (upload + bundled library), voiceover upload or in-app recording with auto-transcribed captions and music ducking under voice. -- **Aesthetic learning** (per-project) + named visual-style presets. -- **Benchmark-aware critique**: upload your high-performers, analyze features, and score "you vs your best". - -**Editor & platform** -- `edl.json` autosave + file-watch live reload (editor ↔ agent round-trip). -- Root React error boundary (no more blank-screen crashes); empty-project preview placeholder. -- Light/dark theme toggle (system-aware, persisted). -- Rebrand **Reel Studio → Aperture**: window title, macOS app/dock name, app icon, and docs. - -**EDL package** -- New `meta`, `style`, and `benchmark` zod schemas; audio-clip `role` (music/voiceover/sfx); `theme.stylePreset` and `theme.grade`. -- Fixed the ESM build so the schema imports cleanly from Node scripts. +## Status -### Initial commit -- Scaffold: Electron + Vite + React editor, shared `packages/edl` schema, Remotion preview/export spine, and the `create-social-video` / `critique-video` Claude Code skills. +V1. The full loop — import → auto-cull → AI review → human review → taste learning → search → export/XMP → trash-with-consent — is implemented, with the AI tier optional and budgeted, and offline deterministic behavior as the baseline. diff --git a/app/.env.local.example b/app/.env.local.example index 1fd70be..7297dd7 100644 --- a/app/.env.local.example +++ b/app/.env.local.example @@ -1,19 +1,18 @@ # Copy this file to `app/.env.local` (gitignored) and fill in your key. -# The Aperture main process loads it at startup; spawned scripts inherit it. +# The Keeper main process loads it at startup; spawned scripts inherit it. +# Keys set here are "locked": the Settings UI shows them as env-managed. -# --- LLM generation (the "Generate" button) --- +# --- AI judge + search refinement --- # Default provider is OpenAI with model gpt-5.5. OPENAI_API_KEY=sk-your-rotated-key-here # Optional escape hatch — rotate models/providers with no code change: -# APERTURE_LLM_PROVIDER=openai # openai | anthropic | openai-compatible -# APERTURE_LLM_MODEL=gpt-5.5 -# APERTURE_LLM_BASE_URL= # e.g. http://localhost:11434/v1 for local Gemma -# APERTURE_LLM_API_KEY= # generic; overrides the provider-specific key +# KEEPER_LLM_PROVIDER=openai # openai | anthropic | openai-compatible +# KEEPER_LLM_MODEL=gpt-5.5 +# KEEPER_LLM_BASE_URL= # e.g. http://localhost:11434/v1 for local models +# KEEPER_LLM_API_KEY= # generic; overrides the provider-specific key +# KEEPER_REASONING_EFFORT=low # low | medium | high # ANTHROPIC_API_KEY= -# --- Voiceover synthesis + voice cloning (optional) --- -# Scope the key minimally: Text to Speech = Access, Voices = Read -# (+ Voices = Write only for in-app cloning/deleting). Takes precedence -# over the key field in Settings -> Voices and locks it while set. -# ELEVENLABS_API_KEY=sk_... +# --- Storage overrides (dev) --- +# KEEPER_LIBRARY_DIR=/tmp/keeper-dev-library diff --git a/app/package.json b/app/package.json index 1d6254f..b38cdbb 100644 --- a/app/package.json +++ b/app/package.json @@ -1,7 +1,7 @@ { "name": "keeper-app", "version": "0.1.0", - "description": "Keeper Electron editor.", + "description": "Keeper Electron app — AI-assisted culling for photos and videos.", "main": "out/main/index.js", "scripts": { "dev": "electron-vite dev", @@ -13,20 +13,13 @@ "@ai-sdk/anthropic": "^4.0.4", "@ai-sdk/openai": "^4.0.4", "@ai-sdk/openai-compatible": "^3.0.2", - "@reel/edl": "0.1.0", - "@remotion/bundler": "^4.0.0", - "@remotion/install-whisper-cpp": "^4.0.0", - "@remotion/media-parser": "^4.0.0", - "@remotion/player": "^4.0.0", - "@remotion/renderer": "^4.0.0", - "@remotion/transitions": "^4.0.0", + "@huggingface/transformers": "^3.5.1", + "@keeper/schema": "0.1.0", "ai": "^7.0.9", + "exiftool-vendored": "^30.0.0", "ffmpeg-static": "^5.2.0", - "img-fx": "^0.4.1", "react": "^18.3.1", "react-dom": "^18.3.1", - "remotion": "^4.0.0", - "three": "^0.185.1", "zustand": "^5.0.2" }, "devDependencies": { diff --git a/app/resources/music/README.md b/app/resources/music/README.md deleted file mode 100644 index d8a69f1..0000000 --- a/app/resources/music/README.md +++ /dev/null @@ -1,8 +0,0 @@ -# Bundled music - -Drop royalty-free audio tracks (`.mp3`, `.m4a`, `.wav`, `.ogg`) here to make them -selectable from the editor's Audio → Music → Library control. - -These ship with the app, so only include tracks you have the right to distribute -(e.g. CC0 / public-domain). Tracks a user uploads themselves are copied into -their own `projects//assets/` instead and are never added here. diff --git a/app/scripts/analyze-benchmarks.mjs b/app/scripts/analyze-benchmarks.mjs deleted file mode 100644 index 78c8c45..0000000 --- a/app/scripts/analyze-benchmarks.mjs +++ /dev/null @@ -1,110 +0,0 @@ -// Extract structural features from the creator's uploaded high-performing -// videos (projects//benchmarks/) so the critic can score the current cut -// against what actually works for THIS creator, not generic heuristics. -// -// Run: `node app/scripts/analyze-benchmarks.mjs --slug ` -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { spawnSync } from "node:child_process"; -import ffmpegPath from "ffmpeg-static"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); -const VIDEO_EXT = new Set([".mp4", ".mov", ".webm", ".m4v"]); - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} -const round = (n) => Math.round(n * 100) / 100; - -function durationSec(file) { - const res = spawnSync(ffmpegPath, ["-i", file], { encoding: "utf8" }); - const m = (res.stderr || "").match(/Duration:\s*(\d+):(\d+):(\d+(?:\.\d+)?)/); - return m ? Number(m[1]) * 3600 + Number(m[2]) * 60 + Number(m[3]) : 0; -} - -// Scene-change timestamps (seconds) via ffmpeg detection. -function sceneTimes(file) { - const res = spawnSync( - ffmpegPath, - ["-i", file, "-vf", "select='gt(scene,0.4)',showinfo", "-f", "null", "-"], - { encoding: "utf8", maxBuffer: 1 << 26 }, - ); - return [...(res.stderr || "").matchAll(/pts_time:(\d+(?:\.\d+)?)/g)].map((m) => Number(m[1])); -} - -// Integrated loudness (LUFS) via ebur128. -function loudnessLufs(file) { - const res = spawnSync(ffmpegPath, ["-i", file, "-af", "ebur128", "-f", "null", "-"], { - encoding: "utf8", - maxBuffer: 1 << 26, - }); - const matches = [...(res.stderr || "").matchAll(/I:\s*(-?\d+(?:\.\d+)?)\s*LUFS/g)]; - return matches.length ? Number(matches[matches.length - 1][1]) : undefined; -} - -function stats(xs) { - const vals = xs.filter((x) => typeof x === "number" && Number.isFinite(x)); - if (vals.length === 0) return undefined; - const mean = vals.reduce((a, b) => a + b, 0) / vals.length; - const variance = vals.reduce((a, b) => a + (b - mean) ** 2, 0) / vals.length; - return { mean: round(mean), std: round(Math.sqrt(variance)), min: round(Math.min(...vals)), max: round(Math.max(...vals)) }; -} - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const benchDir = path.join(projectDir, "benchmarks"); - const metaPath = path.join(benchDir, "benchmarks.meta.json"); - const metrics = fs.existsSync(metaPath) ? JSON.parse(fs.readFileSync(metaPath, "utf8")) : {}; - - const files = (fs.existsSync(benchDir) ? fs.readdirSync(benchDir) : []).filter((f) => - VIDEO_EXT.has(path.extname(f).toLowerCase()), - ); - if (files.length === 0) throw new Error("no benchmark videos in benchmarks/"); - - console.log(`PHASE analyzing ${files.length} benchmark clips`); - const videos = []; - let done = 0; - for (const f of files) { - const file = path.join(benchDir, f); - const dur = durationSec(file); - const scenes = sceneTimes(file); - const cuts = scenes.length; - videos.push({ - file: f, - durationSec: round(dur), - cutsPer10s: dur > 0 ? round((cuts / dur) * 10) : undefined, - hookSec: scenes.length ? round(scenes[0]) : undefined, - loudnessLufs: loudnessLufs(file), - views: metrics[f]?.views, - likes: metrics[f]?.likes, - }); - done++; - console.log(`PROGRESS ${Math.round((done / files.length) * 100)}`); - } - - const distribution = {}; - for (const key of ["durationSec", "cutsPer10s", "hookSec", "loudnessLufs"]) { - const s = stats(videos.map((v) => v[key])); - if (s) distribution[key] = s; - } - - const out = { - generatedAt: new Date().toISOString(), - count: videos.length, - videos, - distribution, - }; - fs.writeFileSync(path.join(projectDir, "benchmarks.json"), `${JSON.stringify(out, null, 2)}\n`); - console.log(`DONE ${videos.length} benchmarks analyzed`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/analyze-collection.mjs b/app/scripts/analyze-collection.mjs deleted file mode 100644 index c8b24b3..0000000 --- a/app/scripts/analyze-collection.mjs +++ /dev/null @@ -1,261 +0,0 @@ -// Analyze a COLLECTION of reference videos into one rich, reusable style profile. -// Works on either the global library (--styleDir ) or a project's own -// references (--slug ). Samples frames, computes deterministic editing -// metrics, then (when an LLM is configured) distills frames + metrics into a -// prose style guide + per-reference exemplars the generator imitates. Without a -// model it still writes a solid deterministic profile. -// -// Run: node app/scripts/analyze-collection.mjs --styleDir /abs/styles/ -// node app/scripts/analyze-collection.mjs --slug -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { execFileSync, spawnSync } from "node:child_process"; -import ffmpegPath from "ffmpeg-static"; -import { generateText } from "ai"; -import { isLlmConfigured, llmConfig, resolveModel, reasoningEffort } from "./llm.mjs"; -import { extractJson } from "./edl-util.mjs"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); -const VIDEO_EXT = new Set([".mp4", ".mov", ".webm", ".m4v"]); -const FRAMES_PER_CLIP = 6; -const MAX_VISION_FRAMES = 12; - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} -const round = (n) => Math.round(n * 100) / 100; -const median = (xs) => { - if (xs.length === 0) return undefined; - const s = [...xs].sort((a, b) => a - b); - const m = Math.floor(s.length / 2); - return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2; -}; -const toHex = (r, g, b) => - "#" + [r, g, b].map((v) => Math.max(0, Math.min(255, Math.round(v))).toString(16).padStart(2, "0")).join(""); - -function durationSec(file) { - const res = spawnSync(ffmpegPath, ["-i", file], { encoding: "utf8" }); - const m = (res.stderr || "").match(/Duration:\s*(\d+):(\d+):(\d+(?:\.\d+)?)/); - return m ? Number(m[1]) * 3600 + Number(m[2]) * 60 + Number(m[3]) : 0; -} -function sceneTimes(file) { - const res = spawnSync(ffmpegPath, ["-i", file, "-vf", "select='gt(scene,0.4)',showinfo", "-f", "null", "-"], { - encoding: "utf8", - maxBuffer: 1 << 26, - }); - return [...(res.stderr || "").matchAll(/pts_time:(\d+(?:\.\d+)?)/g)].map((m) => Number(m[1])); -} -function paletteFor(file, atSec) { - try { - const res = spawnSync( - ffmpegPath, - ["-ss", String(atSec), "-i", file, "-frames:v", "1", "-vf", "scale=4:1", "-f", "rawvideo", "-pix_fmt", "rgb24", "-"], - { maxBuffer: 1 << 20 }, - ); - const buf = res.stdout; - if (!buf || buf.length < 12) return []; - return [0, 3, 6, 9].map((o) => toHex(buf[o], buf[o + 1], buf[o + 2])); - } catch { - return []; - } -} -function avgPalette(palettes) { - const valid = palettes.filter((p) => p.length >= 3); - if (valid.length === 0) return []; - const slots = Math.min(...valid.map((p) => p.length), 4); - const acc = Array.from({ length: slots }, () => [0, 0, 0]); - for (const p of valid) { - for (let i = 0; i < slots; i++) { - const n = parseInt(p[i].slice(1), 16); - acc[i][0] += (n >> 16) & 255; - acc[i][1] += (n >> 8) & 255; - acc[i][2] += n & 255; - } - } - return acc.map(([r, g, b]) => toHex(r / valid.length, g / valid.length, b / valid.length)); -} - -function sampleFrames(file, framesDir, base) { - execFileSync( - ffmpegPath, - ["-y", "-i", file, "-vf", "fps=1/2,scale=512:-1", "-frames:v", String(FRAMES_PER_CLIP), path.join(framesDir, `${base}-%02d.jpg`)], - { stdio: "ignore" }, - ); -} - -function resolvePaths() { - const styleDir = arg("styleDir"); - const slug = arg("slug"); - if (styleDir) { - return { sourcesDir: path.join(styleDir, "sources"), outDir: styleDir, profileName: "profile.json" }; - } - if (slug) { - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - return { sourcesDir: path.join(projectDir, "references"), outDir: projectDir, profileName: "style.json" }; - } - throw new Error("missing --styleDir or --slug"); -} - -function buildVisionPrompt(metricsJson) { - return [ - "You are a senior short-form (vertical 9:16) video editor reverse-engineering a creator's aesthetic from sample frames of their past videos.", - "Study the frames and the measured editing metrics, then output ONE reusable STYLE PROFILE as JSON.", - "", - "Return ONLY this JSON (no prose, no code fences):", - `{ - "palette": ["#hex","#hex","#hex"], // [text, background, accent] that matches the look - "fontFamily": "a CSS font stack matching the vibe", - "captionStyle": "karaoke|block|word|none", - "grade": { "brightness": 1.0, "contrast": 1.0, "saturation": 1.0, "temperature": 0, "vignette": 0 }, - "hookPattern": "one sentence on how their opens grab attention", - "textTreatment": "how on-screen text looks/behaves (size, position, weight, motion)", - "transitions": ["fade|slide|wipe|whip|cut", "..."], - "energy": 0.0, // 0 calm .. 1 frenetic - "musicEnergy": 0.0, // 0 none .. 1 tightly beat-synced - "styleGuide": "2-4 paragraph guide a stranger could follow to reproduce the look/feel", - "exemplars": [ { "source": "", "hook": "...", "beats": "...", "captionStyle": "...", "textTreatment": "...", "transitions": ["..."], "gradeNote": "..." } ], - "do": ["3-6 concrete, transferable rules"], - "avoid": ["3-6 things that would break the vibe"] -}`, - "", - "Learn GENERAL, transferable principles, not the literal contents of any one clip.", - "", - "=== MEASURED METRICS (ground truth for pacing/length) ===", - metricsJson, - ].join("\n"); -} - -async function main() { - const { sourcesDir, outDir, profileName } = resolvePaths(); - const framesDir = path.join(outDir, ".frames"); - fs.mkdirSync(framesDir, { recursive: true }); - - const videos = (fs.existsSync(sourcesDir) ? fs.readdirSync(sourcesDir) : []).filter((f) => - VIDEO_EXT.has(path.extname(f).toLowerCase()), - ); - if (videos.length === 0) throw new Error(`no reference videos in ${sourcesDir}`); - - console.log(`PHASE analyzing ${videos.length} reference clips`); - const perVideo = []; - let done = 0; - for (const f of videos) { - const file = path.join(sourcesDir, f); - const base = path.basename(f, path.extname(f)).replace(/[^a-zA-Z0-9_-]+/g, "-"); - const dur = durationSec(file); - const scenes = sceneTimes(file); - try { - sampleFrames(file, framesDir, base); - } catch { - // frame sampling best-effort - } - perVideo.push({ - file: f, - base, - durationSec: round(dur), - cutsPer10s: dur > 0 ? round((scenes.length / dur) * 10) : 0, - hookSec: scenes.length ? round(scenes[0]) : undefined, - palette: paletteFor(file, Math.max(0.5, dur / 2)), - }); - done++; - console.log(`PROGRESS ${Math.round((done / videos.length) * 60)}`); - } - - const cuts = perVideo.map((v) => v.cutsPer10s).filter(Boolean); - const lengths = perVideo.map((v) => v.durationSec).filter(Boolean); - const medianCuts = median(cuts) ?? 0; - const metrics = { - clips: videos.length, - palette: avgPalette(perVideo.map((v) => v.palette)), - pacing: { - cutsPer10s: round(medianCuts), - avgShotSec: round(median(perVideo.map((v) => (v.cutsPer10s ? 10 / v.cutsPer10s : 0)).filter(Boolean)) ?? 0), - }, - hookSec: round(median(perVideo.map((v) => v.hookSec).filter((x) => x != null)) ?? 0), - targetLengthSec: round(median(lengths) ?? 0), - energy: Math.max(0, Math.min(1, round((medianCuts - 1) / 11))), - perVideo: perVideo.map(({ file, durationSec, cutsPer10s, hookSec }) => ({ file, durationSec, cutsPer10s, hookSec })), - }; - - const existing = (() => { - try { - return JSON.parse(fs.readFileSync(path.join(outDir, profileName), "utf8")); - } catch { - return {}; - } - })(); - - let profile = { - id: existing.id ?? "learned", - name: existing.name ?? "My Style", - palette: metrics.palette, - captionStyle: existing.captionStyle ?? "karaoke", - pacing: metrics.pacing, - hookSec: metrics.hookSec, - energy: metrics.energy, - targetLengthSec: metrics.targetLengthSec, - exemplars: [], - do: [], - avoid: [], - notes: "Deterministic baseline (no model). Configure an LLM for a richer style guide.", - source: { clips: videos.length, generatedAt: new Date().toISOString() }, - }; - - if (isLlmConfigured()) { - const { provider, model } = llmConfig(); - console.log(`PHASE distilling style with ${provider}/${model}`); - // Gather a capped, spread-out set of frames across videos for vision. - const frameFiles = fs.readdirSync(framesDir).filter((f) => f.endsWith(".jpg")).sort(); - const step = Math.max(1, Math.floor(frameFiles.length / MAX_VISION_FRAMES)); - const chosen = frameFiles.filter((_, i) => i % step === 0).slice(0, MAX_VISION_FRAMES); - const images = chosen.map((f) => ({ type: "image", image: fs.readFileSync(path.join(framesDir, f)) })); - - try { - const { text } = await generateText({ - model: resolveModel(), - maxOutputTokens: 4000, - providerOptions: { openai: { reasoningEffort: reasoningEffort() } }, - messages: [ - { - role: "user", - content: [{ type: "text", text: buildVisionPrompt(JSON.stringify(metrics, null, 2)) }, ...images], - }, - ], - }); - console.log("PROGRESS 90"); - const llm = extractJson(text); - // LLM owns look/feel + guide + exemplars; deterministic metrics win for pacing/length. - profile = { - ...profile, - palette: Array.isArray(llm.palette) && llm.palette.length ? llm.palette : profile.palette, - fontFamily: llm.fontFamily ?? existing.fontFamily, - captionStyle: llm.captionStyle ?? profile.captionStyle, - grade: llm.grade ?? undefined, - hookPattern: llm.hookPattern, - textTreatment: llm.textTreatment, - transitions: Array.isArray(llm.transitions) ? llm.transitions : [], - energy: typeof llm.energy === "number" ? llm.energy : profile.energy, - musicEnergy: typeof llm.musicEnergy === "number" ? llm.musicEnergy : undefined, - styleGuide: llm.styleGuide, - exemplars: Array.isArray(llm.exemplars) ? llm.exemplars : [], - do: Array.isArray(llm.do) ? llm.do : [], - avoid: Array.isArray(llm.avoid) ? llm.avoid : [], - notes: "Distilled from frames + metrics by the LLM.", - }; - } catch (err) { - console.error(`ERROR LLM distillation failed, keeping deterministic profile: ${err}`); - } - } - - fs.writeFileSync(path.join(outDir, profileName), `${JSON.stringify(profile, null, 2)}\n`); - if (profile.styleGuide) fs.writeFileSync(path.join(outDir, "style-guide.md"), `${profile.styleGuide}\n`); - console.log("PROGRESS 100"); - console.log(`DONE ${path.join(outDir, profileName)}`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/analyze-style.mjs b/app/scripts/analyze-style.mjs deleted file mode 100644 index 1e05a71..0000000 --- a/app/scripts/analyze-style.mjs +++ /dev/null @@ -1,149 +0,0 @@ -// Deterministic baseline for aesthetic learning. Probes the creator's own past -// videos in projects//references/ and writes a baseline style.json -// (palette, pacing, length, energy). The richer, interpretive layer (hook -// patterns, do/avoid, narrative) is added by the `learn-aesthetic` agent skill -// on top of this, exactly as create-social-video builds on analyze.mjs. -// -// Run: `node app/scripts/analyze-style.mjs --slug ` -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { execFileSync, spawnSync } from "node:child_process"; -import ffmpegPath from "ffmpeg-static"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); -const VIDEO_EXT = new Set([".mp4", ".mov", ".webm", ".m4v"]); - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} - -const round = (n) => Math.round(n * 100) / 100; -const median = (xs) => { - if (xs.length === 0) return undefined; - const s = [...xs].sort((a, b) => a - b); - const m = Math.floor(s.length / 2); - return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2; -}; -const toHex = (r, g, b) => - "#" + [r, g, b].map((v) => Math.max(0, Math.min(255, v)).toString(16).padStart(2, "0")).join(""); - -function durationSec(file) { - const res = spawnSync(ffmpegPath, ["-i", file], { encoding: "utf8" }); - const m = (res.stderr || "").match(/Duration:\s*(\d+):(\d+):(\d+(?:\.\d+)?)/); - return m ? Number(m[1]) * 3600 + Number(m[2]) * 60 + Number(m[3]) : 0; -} - -// Count hard cuts via ffmpeg scene-change detection. -function countCuts(file) { - const res = spawnSync( - ffmpegPath, - ["-i", file, "-vf", "select='gt(scene,0.4)',showinfo", "-f", "null", "-"], - { encoding: "utf8", maxBuffer: 1 << 26 }, - ); - const matches = (res.stderr || "").match(/pts_time:/g); - return matches ? matches.length : 0; -} - -// Crude 3-swatch palette: one mid-clip frame downscaled to 3x1 raw RGB. -function paletteFor(file, atSec) { - try { - const res = spawnSync( - ffmpegPath, - ["-ss", String(atSec), "-i", file, "-frames:v", "1", "-vf", "scale=3:1", "-f", "rawvideo", "-pix_fmt", "rgb24", "-"], - { maxBuffer: 1 << 20 }, - ); - const buf = res.stdout; - if (!buf || buf.length < 9) return []; - return [0, 3, 6].map((o) => toHex(buf[o], buf[o + 1], buf[o + 2])); - } catch { - return []; - } -} - -function avgPalette(palettes) { - const valid = palettes.filter((p) => p.length === 3); - if (valid.length === 0) return []; - const acc = [ - [0, 0, 0], - [0, 0, 0], - [0, 0, 0], - ]; - for (const p of valid) { - p.forEach((hex, i) => { - const n = parseInt(hex.slice(1), 16); - acc[i][0] += (n >> 16) & 255; - acc[i][1] += (n >> 8) & 255; - acc[i][2] += n & 255; - }); - } - return acc.map(([r, g, b]) => toHex(Math.round(r / valid.length), Math.round(g / valid.length), Math.round(b / valid.length))); -} - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const refDir = path.join(projectDir, "references"); - const videos = (fs.existsSync(refDir) ? fs.readdirSync(refDir) : []).filter((f) => - VIDEO_EXT.has(path.extname(f).toLowerCase()), - ); - if (videos.length === 0) throw new Error("no reference videos in references/"); - - console.log(`PHASE analyzing ${videos.length} reference clips`); - const cutsPer10s = []; - const shotLens = []; - const lengths = []; - const palettes = []; - - let done = 0; - for (const f of videos) { - const file = path.join(refDir, f); - const dur = durationSec(file); - const cuts = countCuts(file); - if (dur > 0) { - cutsPer10s.push((cuts / dur) * 10); - shotLens.push(dur / (cuts + 1)); - lengths.push(dur); - palettes.push(paletteFor(file, Math.max(0.5, dur / 2))); - } - done++; - console.log(`PROGRESS ${Math.round((done / videos.length) * 100)}`); - } - - const medianCuts = median(cutsPer10s) ?? 0; - // energy: ~0 at 1 cut / 10s, ~1 at 12 cuts / 10s. - const energy = Math.max(0, Math.min(1, round((medianCuts - 1) / 11))); - - const stylePath = path.join(projectDir, "style.json"); - const existing = fs.existsSync(stylePath) ? JSON.parse(fs.readFileSync(stylePath, "utf8")) : {}; - const style = { - id: existing.id ?? "learned", - name: existing.name ?? "My Style", - palette: avgPalette(palettes), - fontFamily: existing.fontFamily, - captionStyle: existing.captionStyle ?? "karaoke", - pacing: { - cutsPer10s: round(medianCuts), - avgShotSec: round(median(shotLens) ?? 0), - }, - hookPattern: existing.hookPattern, - energy, - targetLengthSec: round(median(lengths) ?? 0), - do: existing.do ?? [], - avoid: existing.avoid ?? [], - notes: existing.notes ?? "Baseline from analyze-style.mjs — refine with the learn-aesthetic skill.", - source: { clips: videos.length, generatedAt: new Date().toISOString() }, - }; - - fs.writeFileSync(stylePath, `${JSON.stringify(style, null, 2)}\n`); - console.log(`DONE ${stylePath}`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/analyze.mjs b/app/scripts/analyze.mjs deleted file mode 100644 index bfc100f..0000000 --- a/app/scripts/analyze.mjs +++ /dev/null @@ -1,166 +0,0 @@ -// Clip ingest + deterministic first-cut assembly. -// Run directly: `node app/scripts/analyze.mjs --slug ` -// or via the Electron main process (the Generate button). -// -// Probes every clip in projects//assets with @remotion/media-parser, -// then writes a baseline edl.json: clips laid end-to-end (each capped), with -// existing theme + text overlays preserved. The prompt-aware "smart" assembly -// is done by the agent (the /create-social-video skill) on top of this. -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { parseMedia } from "@remotion/media-parser"; -import { nodeReader } from "@remotion/media-parser/node"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); -// .webm is ambiguous (recorded voiceovers are audio-only .webm); those files -// are classified by probing for video dimensions instead of by extension. -const VIDEO_EXT = new Set([".mp4", ".mov", ".m4v"]); -const AUDIO_EXT = new Set([".mp3", ".wav", ".m4a", ".aac", ".ogg"]); -const AMBIGUOUS_EXT = new Set([".webm"]); -const CAP_SEC = 4; - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} - -const round = (n) => Math.round(n * 100) / 100; - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const assetsDir = path.join(projectDir, "assets"); - const edlPath = path.join(projectDir, "edl.json"); - const edl = JSON.parse(fs.readFileSync(edlPath, "utf8")); - - const all = fs.readdirSync(assetsDir).filter((f) => !f.startsWith(".")).sort(); - const videoFiles = all.filter((f) => VIDEO_EXT.has(path.extname(f).toLowerCase())); - const audioFiles = all.filter((f) => AUDIO_EXT.has(path.extname(f).toLowerCase())); - for (const f of all.filter((x) => AMBIGUOUS_EXT.has(path.extname(x).toLowerCase()))) { - try { - const meta = await parseMedia({ - src: path.join(assetsDir, f), - fields: { dimensions: true }, - reader: nodeReader, - acknowledgeRemotionLicense: true, - }); - (meta.dimensions ? videoFiles : audioFiles).push(f); - } catch { - // unreadable file: skip - } - } - - console.log(`PHASE probing ${videoFiles.length} clips`); - const assets = []; - const probes = []; - for (const f of videoFiles) { - const full = path.join(assetsDir, f); - const meta = await parseMedia({ - src: full, - fields: { durationInSeconds: true, dimensions: true, fps: true }, - reader: nodeReader, - acknowledgeRemotionLicense: true, - }); - const id = path.basename(f, path.extname(f)); - assets.push({ - id, - kind: "video", - src: `assets/${f}`, - durationSec: meta.durationInSeconds ?? undefined, - width: meta.dimensions?.width, - height: meta.dimensions?.height, - }); - probes.push({ file: f, ...meta }); - console.log(`CLIP ${f} ${round(meta.durationInSeconds ?? 0)}s ${meta.dimensions?.width}x${meta.dimensions?.height}`); - } - - // Deterministic assembly: clips on one absolute timeline, overlapping by the - // transition duration so crossfades stay in sync with text/captions. Muted - // (music comes later). - const TRANS = 0.3; - let cursor = 0; - const clips = assets.map((a, i) => { - const len = Math.min(a.durationSec ?? CAP_SEC, CAP_SEC); - const start = round(cursor); - const clip = { - id: `v-${a.id}`, - assetId: a.id, - start, - in: 0, - out: round(len), - volume: 0, - }; - if (i > 0) clip.transitionIn = { preset: "fade", duration: TRANS }; - if (i < assets.length - 1) clip.transitionOut = { preset: "fade", duration: TRANS }; - cursor = start + len - (i < assets.length - 1 ? TRANS : 0); - return clip; - }); - - // Music bed: first audio file, spanning the video duration. - for (const f of audioFiles) { - const meta = await parseMedia({ - src: path.join(assetsDir, f), - fields: { durationInSeconds: true }, - reader: nodeReader, - acknowledgeRemotionLicense: true, - }); - assets.push({ - id: path.basename(f, path.extname(f)), - kind: "audio", - src: `assets/${f}`, - durationSec: meta.durationInSeconds ?? undefined, - }); - } - - // Merge probed assets with the existing list: probes win on fresh metadata, - // but keep fields only the app knows (proxySrc) and keep entries this scan - // doesn't cover (images, previously imported files). - const byId = new Map(edl.assets?.map((a) => [a.id, a]) ?? []); - for (const probed of assets) { - const prev = byId.get(probed.id); - byId.set(probed.id, prev ? { ...prev, ...probed, proxySrc: prev.proxySrc } : probed); - } - edl.assets = [...byId.values()]; - - const videoTrack = edl.tracks.find((tr) => tr.type === "video"); - if (videoTrack) videoTrack.clips = clips; - else edl.tracks.unshift({ id: "v", type: "video", clips }); - - // Music bed on the dedicated music track ("aud"). Voiceover tracks/clips are - // preserved untouched, and a voiceover's own audio file never becomes music. - const voAssetIds = new Set( - edl.tracks - .flatMap((tr) => (tr.type === "audio" ? tr.clips ?? [] : [])) - .filter((c) => c.role === "voiceover") - .map((c) => c.assetId), - ); - const videoDur = round(cursor); - const music = edl.assets.find((a) => a.kind === "audio" && !voAssetIds.has(a.id)); - if (music) { - const span = videoDur > 0 ? videoDur : (music.durationSec ?? 1); - const out = round(Math.min(music.durationSec ?? span, span)) || 1; - const duckUnderVoice = voAssetIds.size > 0; - const audioClips = [ - { id: `a-${music.id}`, assetId: music.id, start: 0, in: 0, out, gain: -12, duckUnderVoice, role: "music" }, - ]; - const audioTrack = edl.tracks.find((tr) => tr.type === "audio" && tr.id !== "vo"); - if (audioTrack) audioTrack.clips = audioClips; - else edl.tracks.push({ id: "aud", type: "audio", clips: audioClips }); - } - - fs.writeFileSync(edlPath, `${JSON.stringify(edl, null, 2)}\n`); - fs.writeFileSync( - path.join(projectDir, "analysis.json"), - `${JSON.stringify({ slug, generatedAt: new Date().toISOString(), probes }, null, 2)}\n`, - ); - console.log(`DONE assembled ${clips.length} clips, ${round(cursor)}s of video`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/autotune-llm.mjs b/app/scripts/autotune-llm.mjs deleted file mode 100644 index 8317d26..0000000 --- a/app/scripts/autotune-llm.mjs +++ /dev/null @@ -1,157 +0,0 @@ -// LLM-backed auto-improve loop (the Claudia auto-skills pattern): each iteration -// the model critiques the current cut and returns an improved edl.json plus a -// 0-100 score and a one-line change note. We validate, write, log the score -// trajectory to results.tsv, and stop on target/plateau/iteration cap. -// Provider-agnostic (default GPT-5.5; see llm.mjs). -// -// Run: `OPENAI_API_KEY=... node app/scripts/autotune-llm.mjs --slug [--iterations 3] [--target 88]` -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { generateText } from "ai"; -import { parseEdl } from "@reel/edl"; -import { isLlmConfigured, llmConfig, resolveModel, reasoningEffort } from "./llm.mjs"; -import { ANIM_NAMES, extractJson, metrics, restoreAudioTracks, sanitizeEdl } from "./edl-util.mjs"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} -function readMaybe(file) { - try { - return fs.readFileSync(file, "utf8"); - } catch { - return ""; - } -} - -function buildPrompt({ edlJson, metricsJson, promptMd, styleJson, benchmarksJson }) { - return [ - "You are an expert short-form (vertical 9:16) video editor improving a cut, one focused pass at a time.", - "Critique the CURRENT edit, then return an IMPROVED version that fixes the 1-2 highest-impact problems", - "(weakest of: hook in first 2s, pacing/cuts, captions, safe margins, length, audio, ending).", - "If a benchmarks distribution is given, move pacing/length toward its means.", - "", - "Return ONLY this JSON (no prose, no code fences):", - '{ "score": , "change": "", "edl": }', - "", - "Constraints for `edl`:", - "- Same shape/keys as the input; only use assets already present (same id/src).", - `- Every text clip's anim, if present, must be {"name":"","from":"animate-text"} (name REQUIRED) — or omit anim.`, - "- Keep it schema-valid; respect theme.safeMargins.", - "", - "=== DETERMINISTIC METRICS (ground truth) ===", - metricsJson, - "", - "=== CREATOR PROMPT ===", - promptMd || "(none)", - "", - "=== STYLE PROFILE ===", - styleJson || "(none)", - "", - "=== BENCHMARKS ===", - benchmarksJson || "(none)", - "", - "=== CURRENT edl.json ===", - edlJson, - ].join("\n"); -} - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - if (!isLlmConfigured()) { - console.error("ERROR no LLM credentials configured (set OPENAI_API_KEY or APERTURE_LLM_API_KEY)."); - process.exit(3); - } - const maxIterations = Number(arg("iterations") ?? 3); - const target = Number(arg("target") ?? 88); - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const edlPath = path.join(projectDir, "edl.json"); - if (!fs.existsSync(edlPath)) { - console.error("ERROR no edl.json to improve"); - process.exit(2); - } - const promptMd = readMaybe(path.join(projectDir, "prompt.md")); - const styleJson = readMaybe(path.join(projectDir, "style.json")); - const benchmarksJson = readMaybe(path.join(projectDir, "benchmarks.json")) || "(none)"; - - const resultsPath = path.join(projectDir, "results.tsv"); - if (!fs.existsSync(resultsPath)) fs.writeFileSync(resultsPath, "iter\tscore\tdelta\tchange\n"); - - const { provider, model } = llmConfig(); - const llm = resolveModel(); - let edl = JSON.parse(readMaybe(edlPath)); - let prev = null; - let stagnant = 0; - - for (let i = 1; i <= maxIterations; i++) { - console.log(`PHASE iteration ${i}/${maxIterations} (${provider}/${model})`); - const prompt = buildPrompt({ - edlJson: JSON.stringify(edl, null, 2), - metricsJson: JSON.stringify(metrics(edl), null, 2), - promptMd, - styleJson, - benchmarksJson, - }); - - let out; - try { - const { text } = await generateText({ - model: llm, - prompt, - maxOutputTokens: 16000, - providerOptions: { openai: { reasoningEffort: reasoningEffort() } }, - }); - out = extractJson(text); - } catch (err) { - console.error(`ERROR iteration ${i} model call failed: ${err}`); - break; - } - - const candidate = sanitizeEdl(out.edl); - const parsed = parseEdl(candidate); - if (!parsed.ok || !parsed.edl) { - console.log(`PHASE iteration ${i} produced invalid edl, stopping`); - break; - } - const score = typeof out.score === "number" ? Math.round(out.score) : prev ?? 0; - const change = typeof out.change === "string" ? out.change.slice(0, 120) : "revised edit"; - - // Never let an improvement pass silently drop the music bed / voiceover. - edl = restoreAudioTracks(parsed.edl, edl); - fs.writeFileSync(edlPath, `${JSON.stringify(edl, null, 2)}\n`); - const delta = prev == null ? 0 : score - prev; - fs.appendFileSync(resultsPath, `${i}\t${score}\t${delta >= 0 ? "+" : ""}${delta}\t${change}\n`); - console.log(`PHASE ${change} -> ${score}`); - console.log(`PROGRESS ${Math.round((i / maxIterations) * 100)}`); - - if (score >= target) { - console.log("PHASE target reached"); - prev = score; - break; - } - if (prev != null && delta < 2) { - stagnant++; - if (stagnant >= 2) { - console.log("PHASE score plateaued"); - prev = score; - break; - } - } else { - stagnant = 0; - } - prev = score; - } - - console.log(`DONE final score ${prev ?? "n/a"}`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/autotune.mjs b/app/scripts/autotune.mjs deleted file mode 100644 index 61c79e2..0000000 --- a/app/scripts/autotune.mjs +++ /dev/null @@ -1,199 +0,0 @@ -// Deterministic auto-improve loop: score the cut, apply the single highest-value -// safe fix, re-score, repeat — logging each iteration's score delta to -// projects//results.tsv. This is the in-app analogue of the agent-driven -// `auto-tune` skill (which applies smarter, content-aware fixes). -// -// Run: `node app/scripts/autotune.mjs --slug [--iterations 4]` -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} -const round = (n) => Math.round(n * 100) / 100; - -function videoClips(edl) { - return edl.tracks.filter((t) => t.type === "video").flatMap((t) => t.clips); -} -function textClips(edl) { - return edl.tracks.filter((t) => t.type === "text").flatMap((t) => t.clips); -} -function audioClips(edl) { - return edl.tracks.filter((t) => t.type === "audio").flatMap((t) => t.clips); -} - -function durationSeconds(edl) { - let max = 0; - for (const t of edl.tracks) { - if (t.type === "video" || t.type === "audio") { - for (const c of t.clips) max = Math.max(max, c.start + (c.out - c.in)); - } else if (t.type === "text") { - for (const c of t.clips) max = Math.max(max, c.end); - } - } - return max; -} - -function closeness(value, mean, std, max) { - const z = Math.abs(value - mean) / Math.max(std, 1e-6); - return Math.round(max * Math.max(0, Math.min(1, 1 - (z - 1) / 2))); -} - -function score(edl, benchmarks) { - const dur = durationSeconds(edl); - const vids = videoClips(edl); - const txt = textClips(edl); - const hasCaptions = edl.tracks.some( - (t) => t.type === "caption" && (t.source || (t.words?.length ?? 0) > 0), - ); - const hasAudio = audioClips(edl).length > 0; - const margins = edl.theme.safeMargins ?? {}; - const hasMargins = (margins.top ?? 0) > 0 && (margins.bottom ?? 0) > 0; - const hook = vids.some((c) => c.start <= 0.1) || txt.some((c) => c.start <= 2); - const ending = vids.some((c) => c.start + (c.out - c.in) >= dur - 1.5) || txt.some((c) => c.end >= dur - 1.5); - const cutsPer10s = dur > 0 ? (vids.length / dur) * 10 : 0; - const dist = benchmarks?.distribution; - - const pacing = - dist?.cutsPer10s && vids.length > 0 - ? closeness(cutsPer10s, dist.cutsPer10s.mean, dist.cutsPer10s.std, 15) - : vids.length >= 4 && vids.length <= 12 - ? 14 - : vids.length === 0 - ? 4 - : 9; - const length = dist?.durationSec - ? closeness(dur, dist.durationSec.mean, dist.durationSec.std, 10) - : dur >= 7 && dur <= 35 - ? 10 - : dur < 7 - ? 5 - : 6; - - return ( - (hook ? 22 : 6) + - pacing + - (hasCaptions ? 15 : 4) + - (hasMargins ? 10 : 3) + - length + - (hasAudio ? 14 : 5) + - (ending ? 9 : 4) - ); -} - -// Ordered, conservative improvements. Each returns a label if it changed edl. -function improvements(edl, benchmarks) { - return [ - () => { - if (edl.theme.captionStyle === "none") { - edl.theme.captionStyle = "karaoke"; - return "enable captions (karaoke)"; - } - return null; - }, - () => { - const m = (edl.theme.safeMargins ??= {}); - if ((m.top ?? 0) <= 0 || (m.bottom ?? 0) <= 0) { - m.top = m.top || 220; - m.bottom = m.bottom || 320; - m.left = m.left || 64; - m.right = m.right || 64; - return "set vertical safe margins"; - } - return null; - }, - () => { - const hasAudio = audioClips(edl).length > 0; - const audioAsset = edl.assets.find((a) => a.kind === "audio"); - if (!hasAudio && audioAsset) { - const span = Math.max(1, round(durationSeconds(edl))) || (audioAsset.durationSec ?? 1); - let track = edl.tracks.find((t) => t.type === "audio"); - if (!track) { - track = { id: "aud", type: "audio", clips: [] }; - edl.tracks.push(track); - } - track.clips.push({ - id: `a-${audioAsset.id}`, - assetId: audioAsset.id, - start: 0, - in: 0, - out: round(Math.min(audioAsset.durationSec ?? span, span)) || 1, - gain: -12, - duckUnderVoice: true, - role: "music", - }); - return "add music bed"; - } - return null; - }, - () => { - // Trim toward the creator's benchmark length when we're well over it. - const mean = benchmarks?.distribution?.durationSec?.mean; - const dur = durationSeconds(edl); - const vids = videoClips(edl); - if (mean && dur > mean * 1.25 && vids.length > 0) { - const last = vids[vids.length - 1]; - const trim = Math.min(last.out - last.in - 0.5, dur - mean); - if (trim > 0.2) { - last.out = round(last.out - trim); - return `trim to ~${round(mean)}s (benchmark length)`; - } - } - return null; - }, - ]; -} - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - const iterations = Number(arg("iterations") ?? 4); - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const edlPath = path.join(projectDir, "edl.json"); - const edl = JSON.parse(fs.readFileSync(edlPath, "utf8")); - const benchPath = path.join(projectDir, "benchmarks.json"); - const benchmarks = fs.existsSync(benchPath) ? JSON.parse(fs.readFileSync(benchPath, "utf8")) : null; - - const resultsPath = path.join(projectDir, "results.tsv"); - if (!fs.existsSync(resultsPath)) fs.writeFileSync(resultsPath, "iter\tscore\tdelta\tchange\n"); - - let prev = score(edl, benchmarks); - fs.appendFileSync(resultsPath, `0\t${prev}\t0\tbaseline\n`); - console.log(`PHASE baseline score ${prev}`); - - for (let i = 1; i <= iterations; i++) { - let changed = null; - for (const improve of improvements(edl, benchmarks)) { - const label = improve(); - if (label) { - const next = score(edl, benchmarks); - if (next >= prev) { - changed = { label, next }; - break; - } - } - } - if (!changed) { - console.log("PHASE no further improvement"); - break; - } - fs.writeFileSync(edlPath, `${JSON.stringify(edl, null, 2)}\n`); - fs.appendFileSync(resultsPath, `${i}\t${changed.next}\t${changed.next - prev >= 0 ? "+" : ""}${changed.next - prev}\t${changed.label}\n`); - console.log(`PHASE ${changed.label} -> ${changed.next}`); - console.log(`PROGRESS ${Math.round((i / iterations) * 100)}`); - prev = changed.next; - } - - console.log(`DONE final score ${prev}`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/catalog-service.mjs b/app/scripts/catalog-service.mjs new file mode 100644 index 0000000..541a175 --- /dev/null +++ b/app/scripts/catalog-service.mjs @@ -0,0 +1,279 @@ +// Long-lived catalog service. The Electron main process spawns one of these +// per library and speaks line-delimited JSON over stdio: +// +// -> {"id":1,"method":"librarySummary","params":{}} +// <- {"id":1,"ok":true,"result":{...}} +// +// Why a child process: the catalog driver (node:sqlite) runs against the +// system Node ABI, keeping the Electron main process free of native/module +// version concerns, and heavy queries stay off the UI event loop. Search's +// text-embedding model is lazy-loaded here on first use. +import fs from "node:fs"; +import readline from "node:readline"; +import { reviewQueue } from "@keeper/schema"; +import { openCatalog } from "./lib/catalog.mjs"; +import { embedText, modelCached, rankBySimilarity } from "./lib/embeddings.mjs"; +import { ensureLayout, resolveHome, safeJoin } from "./lib/paths.mjs"; +import { readTaste, recordVerdict, writeTaste } from "./lib/taste.mjs"; + +function arg(name) { + const i = process.argv.indexOf(`--${name}`); + return i >= 0 ? process.argv[i + 1] : undefined; +} + +const home = ensureLayout(resolveHome(arg("library"))); +// The app's own service must not stamp, or the stamp-watcher would loop. +const catalog = openCatalog(home, { stampOnWrite: false }); + +let embeddingMatrix = null; +function embeddings() { + if (!embeddingMatrix) embeddingMatrix = catalog.allEmbeddings(); + return embeddingMatrix; +} +function invalidateEmbeddings() { + embeddingMatrix = null; +} + +/** Naive date-range extraction so "june photos" narrows without an LLM. */ +function parseDateHints(query) { + const months = [ + "january", "february", "march", "april", "may", "june", + "july", "august", "september", "october", "november", "december", + ]; + const lower = query.toLowerCase(); + const yearMatch = lower.match(/\b(20\d{2})\b/); + const monthIdx = months.findIndex((m) => lower.includes(m)); + if (monthIdx < 0 && !yearMatch) return {}; + const year = yearMatch ? Number(yearMatch[1]) : new Date().getFullYear(); + if (monthIdx >= 0) { + const from = `${year}-${String(monthIdx + 1).padStart(2, "0")}-01`; + const to = `${year}-${String(monthIdx + 1).padStart(2, "0")}-31`; + return { from, to }; + } + return { from: `${year}-01-01`, to: `${year}-12-31` }; +} + +function metadataSearch(query, limit = 120) { + const terms = query + .toLowerCase() + .split(/\s+/) + .filter((t) => t.length > 2); + if (terms.length === 0) return []; + const all = catalog.listAssets({ limit: 50_000 }); + const results = []; + for (const asset of all) { + const haystack = [asset.fileName, asset.caption ?? "", ...asset.tags, asset.exif.make ?? "", asset.exif.model ?? ""] + .join(" ") + .toLowerCase(); + const matched = terms.filter((t) => haystack.includes(t)); + if (matched.length > 0) { + results.push({ assetId: asset.id, score: Math.min(1, matched.length / terms.length), matched }); + } + } + results.sort((a, b) => b.score - a.score); + return results.slice(0, limit); +} + +const methods = { + ping: () => "pong", + + librarySummary() { + return { + home, + counts: catalog.countsSummary(), + imports: catalog.listImports(10), + embeddings: catalog.embeddingCount(), + modelCached: modelCached(home), + }; + }, + + listDays: (p) => catalog.listDays(p ?? {}), + + listAssets: (p) => catalog.listAssets(p ?? {}), + + getAssets: (p) => catalog.getAssets(p.ids ?? []), + + getGroup: (p) => catalog.getGroup(p.id), + + /** + * Set a user verdict on assets. Also folds the decision into taste.json — + * contradictions become exemplars, confirmations bump stats. Returns the + * prior records so the renderer can build undo entries. + */ + setVerdict(p) { + const ids = (p.ids ?? []).slice(0, 5000); + const before = catalog.setUserVerdict(ids, { flag: p.flag, rating: p.rating }); + if (p.flag && p.recordTaste !== false) { + let taste = readTaste(home); + let changed = false; + for (const prior of before) { + if (prior.ai && prior.user.flag === "unrated") { + taste = recordVerdict(taste, { asset: prior, userFlag: p.flag }); + changed = true; + } + } + if (changed) writeTaste(home, taste); + } + return { before: before.map((b) => ({ id: b.id, user: b.user })) }; + }, + + /** Direct restore for undo — no taste side effects. */ + restoreVerdicts(p) { + for (const entry of (p.entries ?? []).slice(0, 5000)) { + catalog.setUserVerdict([entry.id], { flag: entry.user.flag, rating: entry.user.rating }); + } + return { restored: (p.entries ?? []).length }; + }, + + setGroupPick(p) { + catalog.setGroupPick(p.groupId, p.assetId, "user"); + return { ok: true }; + }, + + reviewQueues(p) { + const sure = p?.sureConfidence ?? readTaste(home).thresholds.sureConfidence; + const unrated = catalog.listAssets({ flag: "unrated", hasAi: true, limit: 50_000 }); + const queues = { "sure-reject": [], "sure-keep": [], "needs-eye": [] }; + for (const asset of unrated) { + const queue = reviewQueue(asset, sure); + if (queue in queues) queues[queue].push(asset); + } + return queues; + }, + + async search(p) { + const query = String(p.query ?? "").slice(0, 500); + const filters = { ...parseDateHints(query), ...(p.filters ?? {}) }; + const matrix = embeddings(); + if (matrix.length > 0 && modelCached(home)) { + try { + const queryVec = await embedText(home, query); + let ranked = rankBySimilarity(queryVec, matrix, p.limit ?? 80); + if (filters.from || filters.to || filters.mediaType) { + const records = catalog.getAssets(ranked.map((r) => r.id)); + const byId = new Map(records.map((r) => [r.id, r])); + ranked = ranked.filter((r) => { + const a = byId.get(r.id); + if (!a) return false; + const day = (a.capturedAt ?? a.importedAt ?? "").slice(0, 10); + if (filters.from && day && day < filters.from) return false; + if (filters.to && day && day > filters.to) return false; + if (filters.mediaType && a.mediaType !== filters.mediaType) return false; + return true; + }); + } + return { + query, + mode: "semantic", + results: ranked.map((r) => ({ assetId: r.id, score: Math.max(-1, Math.min(1, r.score)), matched: ["semantic"] })), + }; + } catch (err) { + console.error(`ERROR semantic search failed: ${err?.message ?? err}`); + } + } + return { query, mode: "metadata", results: metadataSearch(query, p.limit ?? 80) }; + }, + + refreshEmbeddings() { + invalidateEmbeddings(); + return { count: embeddings().length }; + }, + + taste() { + return readTaste(home); + }, + + addTasteRule(p) { + const taste = readTaste(home); + const rule = { + id: `rule-${Date.now().toString(36)}`, + text: String(p.text ?? "").slice(0, 500), + createdAt: new Date().toISOString(), + }; + taste.rules = [...taste.rules.slice(-63), rule]; + writeTaste(home, taste); + return rule; + }, + + removeTasteRule(p) { + const taste = readTaste(home); + taste.rules = taste.rules.filter((r) => r.id !== p.id); + writeTaste(home, taste); + return { ok: true }; + }, + + /** Rejected assets + total size — the "empty rejects" confirmation payload. */ + rejectsSummary() { + const rejects = catalog.listAssets({ flag: "reject", limit: 100_000 }); + return { + count: rejects.length, + bytes: rejects.reduce((sum, a) => sum + a.byteSize, 0), + ids: rejects.map((a) => a.id), + }; + }, + + /** Paths for assets about to be trashed (main does the actual shell.trashItem). */ + assetPaths(p) { + const out = []; + for (const asset of catalog.getAssets((p.ids ?? []).slice(0, 100_000))) { + out.push({ id: asset.id, relPath: asset.relPath }); + for (const sibling of catalog.listAssets({ includePaired: true, pairPrimaryId: asset.id, limit: 10 })) { + out.push({ id: sibling.id, relPath: sibling.relPath }); + } + } + return out; + }, + + /** Remove catalog rows + regenerable derived files after originals were trashed. */ + forgetAssets(p) { + const ids = (p.ids ?? []).slice(0, 100_000); + for (const id of ids) { + const members = [ + catalog.getAsset(id), + ...catalog.listAssets({ includePaired: true, pairPrimaryId: id, limit: 10 }), + ].filter(Boolean); + for (const member of members) { + for (const rel of [member.thumbRel, member.previewRel, member.posterRel, member.scrubRel]) { + if (!rel) continue; + try { + fs.rmSync(safeJoin(home, rel), { force: true }); + } catch { + // derived files are regenerable; ignore + } + } + } + catalog.db.prepare("DELETE FROM assets WHERE id = ? OR pair_primary_id = ?").run(id, id); + } + invalidateEmbeddings(); + return { removed: ids.length }; + }, +}; + +const rl = readline.createInterface({ input: process.stdin, terminal: false }); +rl.on("line", async (line) => { + let msg; + try { + msg = JSON.parse(line); + } catch { + return; + } + const { id, method, params } = msg; + const fn = methods[method]; + if (!fn) { + process.stdout.write(`${JSON.stringify({ id, ok: false, error: `unknown method ${method}` })}\n`); + return; + } + try { + const result = await fn(params ?? {}); + process.stdout.write(`${JSON.stringify({ id, ok: true, result })}\n`); + } catch (err) { + process.stdout.write(`${JSON.stringify({ id, ok: false, error: String(err?.message ?? err) })}\n`); + } +}); + +rl.on("close", () => { + catalog.close(); + process.exit(0); +}); + +process.stdout.write(`${JSON.stringify({ ready: true, home })}\n`); diff --git a/app/scripts/critique-llm.mjs b/app/scripts/critique-llm.mjs deleted file mode 100644 index b0352ba..0000000 --- a/app/scripts/critique-llm.mjs +++ /dev/null @@ -1,125 +0,0 @@ -// LLM-backed critique. Reads the cut + prompt + style + benchmarks, scores it -// (grounded in deterministic metrics so numbers aren't hallucinated), and writes -// projects//critique.json in the shape the editor's Critique panel renders. -// Provider-agnostic (default GPT-5.5; see llm.mjs). -// -// Run: `OPENAI_API_KEY=... node app/scripts/critique-llm.mjs --slug ` -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { generateText } from "ai"; -import { isLlmConfigured, llmConfig, resolveModel, reasoningEffort } from "./llm.mjs"; -import { extractJson, metrics } from "./edl-util.mjs"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} -function readMaybe(file) { - try { - return fs.readFileSync(file, "utf8"); - } catch { - return ""; - } -} - -const SHAPE = `{ - "score": <0-100 int = sum of subscore scores>, - "subscores": [ - { "key": "hook", "label": "Hook (first 2s)", "max": 25, "score": , "note": "" }, - { "key": "pacing", "label": "Pacing", "max": 15, "score": , "note": "", "benchmark": { "yours": , "theirs": , "unit": "cuts/10s" } }, - { "key": "captions", "label": "Captions", "max": 15, "score": , "note": "" }, - { "key": "safe", "label": "Safe areas", "max": 10, "score": , "note": "" }, - { "key": "length", "label": "Length", "max": 10, "score": , "note": "", "benchmark": { "yours": , "theirs": , "unit": "s" } }, - { "key": "audio", "label": "Audio", "max": 15, "score": , "note": "" }, - { "key": "ending", "label": "Ending", "max": 10, "score": , "note": "" } - ], - "fixes": [ { "issue": "", "fix": "" } ], - "benchmarksUsed": , - "summary": "" -}`; - -function buildPrompt({ metricsJson, edlJson, promptMd, styleJson, benchmarksJson }) { - const hasBench = benchmarksJson && benchmarksJson !== "(none)"; - return [ - "You are a demanding short-form (vertical 9:16) video editor critiquing a cut.", - 'Grade to the standard of "would this stop the scroll and earn a rewatch?". Most AI first cuts are 45-65; above 80 means genuinely strong. Score what is actually present, not intent.', - "", - "Weighting: hook 25, pacing 15, captions 15, safe-area 10, length 10, audio 15, ending 10 (sums to 100).", - hasBench - ? "A benchmarks distribution of the creator's own high-performers is provided. Score PACING and LENGTH relative to it (full marks within ~1 std of the mean) and fill each `benchmark` with {yours, theirs=mean, unit}. Set benchmarksUsed=true." - : "No benchmarks provided — score on best-practice heuristics, omit the `benchmark` fields, and set benchmarksUsed=false.", - "", - "Return ONLY this JSON object, no prose, no code fences:", - SHAPE, - "", - "=== DETERMINISTIC METRICS (ground truth — do not contradict) ===", - metricsJson, - "", - "=== CREATOR PROMPT (prompt.md) ===", - promptMd || "(none)", - "", - "=== STYLE PROFILE (style.json) ===", - styleJson || "(none)", - "", - "=== BENCHMARKS (benchmarks.json) ===", - hasBench ? benchmarksJson : "(none)", - "", - "=== edl.json ===", - edlJson, - ].join("\n"); -} - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - if (!isLlmConfigured()) { - console.error("ERROR no LLM credentials configured (set OPENAI_API_KEY or APERTURE_LLM_API_KEY)."); - process.exit(3); - } - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const edlRaw = readMaybe(path.join(projectDir, "edl.json")); - if (!edlRaw) { - console.error("ERROR no edl.json to critique"); - process.exit(2); - } - const edl = JSON.parse(edlRaw); - const metricsJson = JSON.stringify(metrics(edl), null, 2); - const promptMd = readMaybe(path.join(projectDir, "prompt.md")); - const styleJson = readMaybe(path.join(projectDir, "style.json")); - const benchmarksJson = readMaybe(path.join(projectDir, "benchmarks.json")) || "(none)"; - - const { provider, model } = llmConfig(); - console.log(`PHASE critiquing with ${provider}/${model}`); - - const { text } = await generateText({ - model: resolveModel(), - prompt: buildPrompt({ metricsJson, edlJson: edlRaw, promptMd, styleJson, benchmarksJson }), - maxOutputTokens: 6000, - providerOptions: { openai: { reasoningEffort: reasoningEffort() } }, - }); - - let critique; - try { - critique = extractJson(text); - } catch (err) { - console.error(`ERROR could not parse critique JSON: ${err}`); - process.exit(2); - } - if (typeof critique.score !== "number" || !Array.isArray(critique.subscores)) { - console.error("ERROR critique JSON missing score/subscores"); - process.exit(2); - } - - fs.writeFileSync(path.join(projectDir, "critique.json"), `${JSON.stringify(critique, null, 2)}\n`); - console.log(`DONE critique score ${critique.score}`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/edl-util.mjs b/app/scripts/edl-util.mjs deleted file mode 100644 index d455304..0000000 --- a/app/scripts/edl-util.mjs +++ /dev/null @@ -1,141 +0,0 @@ -// Shared helpers for the LLM-backed scripts (generate / critique / autotune): -// JSON extraction, schema-deviation repair, and a deterministic metrics summary -// used to ground the model so it scores/edit against real numbers. -import { durationSeconds } from "@reel/edl"; - -export const ANIM_NAMES = [ - "soft-blur-in", - "per-character-rise", - "per-word-crossfade", - "spring-scale-in", - "mask-reveal-up", - "blur-out-up", - "scale-down-fade", - "typewriter", -]; - -const round = (n) => Math.round(n * 100) / 100; - -/** Pull the first {...} JSON object out of a model response (handles code fences). */ -export function extractJson(text) { - const fenced = text.match(/```(?:json)?\s*([\s\S]*?)```/i); - const body = fenced ? fenced[1] : text; - const start = body.indexOf("{"); - const end = body.lastIndexOf("}"); - if (start < 0 || end <= start) throw new Error("no JSON object in model output"); - return JSON.parse(body.slice(start, end + 1)); -} - -const MAX_SEC = 14_400; // mirrors MAX_TIMELINE_SEC in the EDL schema -const SAFE_COLOR = /^(#[0-9a-fA-F]{3,8}|(rgb|rgba|hsl|hsla)\(\s*[\d.,%\s/-]*\)|[a-zA-Z]+)$/; - -function clampSec(v, fallback) { - if (typeof v !== "number" || !Number.isFinite(v)) return fallback; - return Math.min(Math.max(v, 0), MAX_SEC); -} - -/** Repair common harmless model deviations before strict EdlSchema validation. */ -export function sanitizeEdl(obj) { - if (!obj || !Array.isArray(obj.tracks)) return obj; - if (Array.isArray(obj.theme?.palette)) { - // Colors reach inline CSS; anything not a plain color literal is unsafe. - obj.theme.palette = obj.theme.palette.filter((c) => typeof c === "string" && SAFE_COLOR.test(c)); - if (obj.theme.palette.length === 0) delete obj.theme.palette; - } - for (const track of obj.tracks) { - if (!Array.isArray(track?.clips)) continue; - for (const clip of track.clips) { - if (!clip) continue; - // Non-finite / absurd timings hang the timeline and player. - if ("start" in clip) clip.start = clampSec(clip.start, 0); - if ("in" in clip) clip.in = clampSec(clip.in, 0); - if ("out" in clip) clip.out = clampSec(clip.out, 1) || 1; - if ("end" in clip) clip.end = clampSec(clip.end, 1) || 1; - if (track.type === "text") { - if (typeof clip.text === "string" && clip.text.length > 2000) clip.text = clip.text.slice(0, 2000); - if (clip.anim != null) { - if (typeof clip.anim !== "object") { - delete clip.anim; - } else if (typeof clip.anim.name !== "string" || !clip.anim.name) { - clip.anim.name = "soft-blur-in"; - clip.anim.from = clip.anim.from ?? "animate-text"; - } - } - if (clip.style && clip.style !== "title" && clip.style !== "subtitle") { - clip.style = "subtitle"; - } - } - } - } - return obj; -} - -/** - * Re-attach audio the model dropped. LLM refine/improve passes return a whole - * EDL and routinely omit the audio tracks the baseline laid down (music bed, - * voiceover). For every baseline audio track that is missing or emptied in - * the new cut, restore the baseline clips — re-capping music beds to the new - * cut's video length. Deterministic insurance, same philosophy as enforceStyle. - */ -export function restoreAudioTracks(edl, baseline) { - if (!edl?.tracks || !Array.isArray(edl.tracks)) return edl; - if (!baseline?.tracks || !Array.isArray(baseline.tracks)) return edl; - - const videoLen = round( - edl.tracks - .filter((t) => t?.type === "video") - .flatMap((t) => t.clips ?? []) - .reduce((m, c) => Math.max(m, (c.start ?? 0) + (c.out ?? 0) - (c.in ?? 0)), 0), - ); - - for (const baseTrack of baseline.tracks) { - if (baseTrack?.type !== "audio" || (baseTrack.clips?.length ?? 0) === 0) continue; - let track = edl.tracks.find((t) => t?.type === "audio" && t.id === baseTrack.id); - if (!track) { - track = { id: baseTrack.id, type: "audio", clips: [] }; - if (baseTrack.name) track.name = baseTrack.name; - edl.tracks.push(track); - } - if ((track.clips?.length ?? 0) > 0) continue; // the model kept this track's audio - track.clips = baseTrack.clips.map((c) => { - const clip = { ...c }; - if ((clip.role ?? "music") === "music" && videoLen > 0) { - const maxOut = round((clip.in ?? 0) + Math.max(0.1, videoLen - (clip.start ?? 0))); - clip.out = Math.min(clip.out ?? maxOut, maxOut); - } - return clip; - }); - } - return edl; -} - -/** - * Deterministically stamp the measurable look from a style profile onto an EDL, - * so the style shows even when the model under-applies it. - */ -export function enforceStyle(edl, profile) { - if (!profile || !edl?.theme) return edl; - if (profile.palette?.length) edl.theme.palette = profile.palette.slice(0, 3); - if (profile.fontFamily) edl.theme.fontFamily = profile.fontFamily; - if (profile.captionStyle) edl.theme.captionStyle = profile.captionStyle; - if (profile.grade) edl.theme.grade = profile.grade; - if (profile.id) edl.theme.stylePreset = profile.id; - return edl; -} - -/** Deterministic structural metrics for grounding the LLM critic/editor. */ -export function metrics(edl) { - const vids = edl.tracks.filter((t) => t.type === "video").flatMap((t) => t.clips ?? []); - const txt = edl.tracks.filter((t) => t.type === "text").flatMap((t) => t.clips ?? []); - const dur = durationSeconds(edl); - return { - durationSec: round(dur), - videoClips: vids.length, - textClips: txt.length, - cutsPer10s: dur > 0 ? round((vids.length / dur) * 10) : 0, - hasCaptions: edl.tracks.some((t) => t.type === "caption" && (t.source || (t.words?.length ?? 0) > 0)), - hasAudio: edl.tracks.some((t) => t.type === "audio" && (t.clips?.length ?? 0) > 0), - hasMargins: (edl.theme?.safeMargins?.top ?? 0) > 0 && (edl.theme?.safeMargins?.bottom ?? 0) > 0, - hookPresent: vids.some((c) => c.start <= 0.1) || txt.some((c) => c.start <= 2), - }; -} diff --git a/app/scripts/edl-util.test.mjs b/app/scripts/edl-util.test.mjs deleted file mode 100644 index e3cfef0..0000000 --- a/app/scripts/edl-util.test.mjs +++ /dev/null @@ -1,141 +0,0 @@ -import { describe, expect, it } from "vitest"; -import { enforceStyle, extractJson, metrics, restoreAudioTracks, sanitizeEdl } from "./edl-util.mjs"; - -describe("sanitizeEdl", () => { - it("fills a missing anim.name (the baseline-fallback bug) and defaults from", () => { - const out = sanitizeEdl({ - tracks: [{ type: "text", clips: [{ id: "t1", anim: { from: "x" } }] }], - }); - expect(out.tracks[0].clips[0].anim.name).toBe("soft-blur-in"); - expect(out.tracks[0].clips[0].anim.from).toBe("x"); - }); - - it("drops a non-object anim and fixes an invalid style", () => { - const out = sanitizeEdl({ - tracks: [{ type: "text", clips: [{ id: "t1", anim: "nope", style: "headline" }] }], - }); - expect(out.tracks[0].clips[0].anim).toBeUndefined(); - expect(out.tracks[0].clips[0].style).toBe("subtitle"); - }); - - it("leaves non-text tracks and valid anims untouched", () => { - const input = { tracks: [{ type: "video", clips: [{ id: "v1" }] }] }; - expect(sanitizeEdl(input)).toEqual(input); - }); -}); - -describe("restoreAudioTracks", () => { - const baseline = { - tracks: [ - { id: "v", type: "video", clips: [{ id: "v1", assetId: "a", start: 0, in: 0, out: 12 }] }, - { - id: "aud", - type: "audio", - name: "Music", - clips: [{ id: "a-m", assetId: "m", start: 0, in: 0, out: 12, gain: -12, duckUnderVoice: false, role: "music" }], - }, - { - id: "vo", - type: "audio", - clips: [{ id: "a-v", assetId: "v", start: 0, in: 0, out: 6, gain: 0, duckUnderVoice: false, role: "voiceover" }], - }, - ], - }; - - it("re-attaches audio tracks the model dropped, capping music to the new video length", () => { - const modelCut = { - tracks: [{ id: "v", type: "video", clips: [{ id: "v1", assetId: "a", start: 0, in: 0, out: 8 }] }], - }; - const out = restoreAudioTracks(modelCut, baseline); - const aud = out.tracks.find((t) => t.id === "aud"); - const vo = out.tracks.find((t) => t.id === "vo"); - expect(aud.name).toBe("Music"); - expect(aud.clips[0]).toMatchObject({ assetId: "m", role: "music", out: 8 }); - // Voiceover keeps its natural length (captions may outlast the last cut). - expect(vo.clips[0]).toMatchObject({ assetId: "v", role: "voiceover", out: 6 }); - }); - - it("restores clips onto an emptied track without duplicating a kept one", () => { - const modelCut = { - tracks: [ - { id: "v", type: "video", clips: [{ id: "v1", assetId: "a", start: 0, in: 0, out: 12 }] }, - { id: "aud", type: "audio", clips: [] }, - { id: "vo", type: "audio", clips: [{ id: "a-v", assetId: "v", start: 1, in: 0, out: 6, role: "voiceover" }] }, - ], - }; - const out = restoreAudioTracks(modelCut, baseline); - expect(out.tracks.find((t) => t.id === "aud").clips).toHaveLength(1); - // The model's own placement of the kept voiceover is untouched. - expect(out.tracks.find((t) => t.id === "vo").clips[0].start).toBe(1); - }); - - it("is a no-op when the model kept all audio", () => { - const modelCut = JSON.parse(JSON.stringify(baseline)); - const out = restoreAudioTracks(modelCut, baseline); - expect(out).toEqual(baseline); - }); -}); - -describe("extractJson", () => { - it("parses a fenced ```json block", () => { - expect(extractJson('prose\n```json\n{"a":1}\n```\nmore')).toEqual({ a: 1 }); - }); - - it("parses bare JSON surrounded by prose", () => { - expect(extractJson('here it is {"b":2} thanks')).toEqual({ b: 2 }); - }); - - it("throws when there is no JSON object", () => { - expect(() => extractJson("no json here")).toThrow(); - }); -}); - -describe("metrics", () => { - it("computes cuts and duration from tracks", () => { - const edl = { - theme: { safeMargins: { top: 10, bottom: 10 } }, - tracks: [ - { - type: "video", - clips: [ - { start: 0, in: 0, out: 2 }, - { start: 2, in: 0, out: 2 }, - ], - }, - { type: "caption", words: [{ text: "hi", start: 0, end: 1 }] }, - ], - }; - const m = metrics(edl); - expect(m.durationSec).toBe(4); - expect(m.videoClips).toBe(2); - expect(m.cutsPer10s).toBe(5); - expect(m.hasCaptions).toBe(true); - expect(m.hasMargins).toBe(true); - expect(m.hookPresent).toBe(true); - }); -}); - -describe("enforceStyle", () => { - it("stamps palette, font, caption, grade, and preset id", () => { - const edl = { theme: { palette: ["#000"], captionStyle: "karaoke" } }; - const profile = { - id: "p1", - palette: ["#111111", "#eeeeee", "#ff0000"], - fontFamily: "Inter", - captionStyle: "block", - grade: { brightness: 1.1 }, - }; - enforceStyle(edl, profile); - expect(edl.theme.palette).toEqual(["#111111", "#eeeeee", "#ff0000"]); - expect(edl.theme.fontFamily).toBe("Inter"); - expect(edl.theme.captionStyle).toBe("block"); - expect(edl.theme.grade).toEqual({ brightness: 1.1 }); - expect(edl.theme.stylePreset).toBe("p1"); - }); - - it("is a no-op without a profile", () => { - const edl = { theme: { palette: ["#000"] } }; - expect(enforceStyle(edl, null)).toBe(edl); - expect(edl.theme.palette).toEqual(["#000"]); - }); -}); diff --git a/app/scripts/export.mjs b/app/scripts/export.mjs new file mode 100644 index 0000000..64194be --- /dev/null +++ b/app/scripts/export.mjs @@ -0,0 +1,103 @@ +// Export selects out of the library — the handoff to Lightroom, Capture One, +// Aperture, or a plain folder. +// +// node app/scripts/export.mjs --dest [--ids a,b,c | --picks] [--xmp] +// [--in-place-xmp] [--library ] +// +// Copies originals (plus RAW siblings and Live Photo videos), optionally +// writing .xmp sidecars with ratings/flags/keywords next to the copies. +// --in-place-xmp instead writes sidecars beside the originals inside the +// library (for apps pointed straight at the library folder). +import fs from "node:fs"; +import path from "node:path"; +import { openCatalog } from "./lib/catalog.mjs"; +import { ensureLayout, resolveHome, safeJoin } from "./lib/paths.mjs"; +import { buildXmp, sidecarPath } from "./lib/xmp.mjs"; + +function arg(name) { + const i = process.argv.indexOf(`--${name}`); + return i >= 0 ? process.argv[i + 1] : undefined; +} +const has = (name) => process.argv.includes(`--${name}`); + +const phase = (name) => console.log(`PHASE ${name}`); +const progress = (pct) => console.log(`PROGRESS ${Math.round(pct)}`); + +function siblingsOf(catalog, primary) { + return catalog.listAssets({ includePaired: true, pairPrimaryId: primary.id, limit: 100 }); +} + +async function main() { + const home = ensureLayout(resolveHome(arg("library"))); + const catalog = openCatalog(home, { stampOnWrite: false }); + const inPlace = has("in-place-xmp"); + const writeXmpSidecars = has("xmp") || inPlace; + const dest = arg("dest"); + if (!inPlace && !dest) throw new Error("missing --dest"); + + try { + const ids = arg("ids")?.split(",").filter(Boolean); + const assets = ids ? catalog.getAssets(ids) : catalog.listAssets({ flag: "pick", limit: 100_000 }); + if (assets.length === 0) { + console.log(`DONE ${JSON.stringify({ exported: 0, sidecars: 0 })}`); + return; + } + + phase(inPlace ? `writing sidecars for ${assets.length}` : `exporting ${assets.length} selects`); + if (!inPlace) fs.mkdirSync(dest, { recursive: true }); + + let exported = 0; + let sidecars = 0; + + for (const [index, asset] of assets.entries()) { + const files = [asset, ...siblingsOf(catalog, asset)]; + for (const file of files) { + const src = safeJoin(home, file.relPath); + if (!fs.existsSync(src)) { + console.error(`ERROR original offline: ${file.relPath}`); + continue; + } + if (inPlace) { + if (file.id === asset.id && writeXmpSidecars) { + fs.writeFileSync( + sidecarPath(src), + buildXmp({ + rating: asset.user.rating, + flag: asset.user.flag, + tags: asset.tags, + caption: asset.caption, + }), + ); + sidecars++; + } + } else { + const out = path.join(dest, file.fileName); + fs.copyFileSync(src, out); + exported++; + if (file.id === asset.id && writeXmpSidecars) { + fs.writeFileSync( + sidecarPath(out), + buildXmp({ + rating: asset.user.rating, + flag: asset.user.flag, + tags: asset.tags, + caption: asset.caption, + }), + ); + sidecars++; + } + } + } + progress(((index + 1) / assets.length) * 100); + } + + console.log(`DONE ${JSON.stringify({ exported, sidecars, dest: inPlace ? home : dest })}`); + } finally { + catalog.close(); + } +} + +main().catch((err) => { + console.error(`ERROR ${err?.stack ?? err}`); + process.exit(1); +}); diff --git a/app/scripts/extract-frames.mjs b/app/scripts/extract-frames.mjs deleted file mode 100644 index ecff4a6..0000000 --- a/app/scripts/extract-frames.mjs +++ /dev/null @@ -1,64 +0,0 @@ -// Sample still frames from each reference clip so the agent (learn-aesthetic -// skill) can literally see the creator's aesthetic. Frames land in -// projects//references/.frames/. -// -// Run: `node app/scripts/extract-frames.mjs --slug ` (or --dir benchmarks) -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { execFileSync } from "node:child_process"; -import ffmpegPath from "ffmpeg-static"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); -const VIDEO_EXT = new Set([".mp4", ".mov", ".webm", ".m4v"]); -const FRAMES_PER_CLIP = 8; - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - const sub = arg("dir") ?? "references"; - - const dir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug, sub); - const framesDir = path.join(dir, ".frames"); - fs.mkdirSync(framesDir, { recursive: true }); - - const videos = (fs.existsSync(dir) ? fs.readdirSync(dir) : []).filter((f) => - VIDEO_EXT.has(path.extname(f).toLowerCase()), - ); - console.log(`PHASE sampling ${videos.length} reference clips`); - - let done = 0; - for (const f of videos) { - const base = path.basename(f, path.extname(f)); - // Evenly sample across the clip: 1 frame every few seconds, capped. - execFileSync( - ffmpegPath, - [ - "-y", - "-i", - path.join(dir, f), - "-vf", - "fps=1/2,scale=480:-1", - "-frames:v", - String(FRAMES_PER_CLIP), - path.join(framesDir, `${base}-%02d.jpg`), - ], - { stdio: "ignore" }, - ); - done++; - console.log(`PROGRESS ${Math.round((done / Math.max(1, videos.length)) * 100)}`); - } - - console.log(`DONE ${framesDir}`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/generate-llm.mjs b/app/scripts/generate-llm.mjs deleted file mode 100644 index 2948b8c..0000000 --- a/app/scripts/generate-llm.mjs +++ /dev/null @@ -1,261 +0,0 @@ -// LLM-backed first cut via a single structured-output call (cost-predictable), -// provider-agnostic (default GPT-5.5; see llm.mjs for the escape hatch). -// -// Flow: run the cheap local baseline (analyze.mjs) to probe clips and produce a -// valid edl.json skeleton, then ask the model to REFINE it from prompt.md + -// style.json. One model call (with one repair retry on invalid JSON), validated -// against EdlSchema before writing. -// -// Run: `OPENAI_API_KEY=... node app/scripts/generate-llm.mjs --slug ` -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { spawnSync } from "node:child_process"; -import { generateText } from "ai"; -import { parseEdl } from "@reel/edl"; -import { isLlmConfigured, llmConfig, resolveModel, reasoningEffort } from "./llm.mjs"; -import { ANIM_NAMES, enforceStyle, extractJson, restoreAudioTracks, sanitizeEdl } from "./edl-util.mjs"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); -// Reasoning models spend tokens on hidden reasoning before the JSON, so give -// the completion generous headroom. -const MAX_OUTPUT_TOKENS = 16000; - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} - -function readMaybe(file) { - try { - return fs.readFileSync(file, "utf8"); - } catch { - return ""; - } -} -function readJsonMaybe(file) { - try { - return JSON.parse(fs.readFileSync(file, "utf8")); - } catch { - return null; - } -} - -function stylesDir() { - return process.env.APERTURE_STYLES_DIR || path.join(repoRoot, "styles"); -} - -// A profile is "analyzed" (rich) once the LLM has distilled a guide/exemplars; -// the deterministic baseline has neither. -function isAnalyzed(p) { - return Boolean(p && (p.styleGuide || (Array.isArray(p.exemplars) && p.exemplars.length > 0))); -} - -// Style ids are opaque slugs we generate ourselves (slugify). meta.json is -// part of a shareable project, so treat styleProfileId as untrusted: anything -// that could traverse out of the styles dir is ignored. -function safeStyleId(id) { - return typeof id === "string" && /^[a-z0-9][a-z0-9_-]{0,63}$/i.test(id) ? id : undefined; -} - -// Resolve the active style + where to (re)analyze it from: -// 1. project's own style.json (override) -> analyze via --slug references -// 2. meta.styleProfileId library profile -> analyze via --styleDir -// 3. exactly one library style exists -> auto-select it -function resolveActiveStyle(projectDir, slug) { - const localPath = path.join(projectDir, "style.json"); - if (fs.existsSync(localPath)) { - return { profile: readJsonMaybe(localPath), kind: "project", analyzeArgs: ["--slug", slug], profilePath: localPath }; - } - const meta = readJsonMaybe(path.join(projectDir, "meta.json")) ?? {}; - const dir = stylesDir(); - let id = safeStyleId(meta.styleProfileId); - if (!id) { - try { - const dirs = fs - .readdirSync(dir) - .filter((d) => fs.existsSync(path.join(dir, d, "sources")) || fs.existsSync(path.join(dir, d, "profile.json"))); - if (dirs.length === 1) id = dirs[0]; - } catch { - // no library - } - } - if (id) { - const styleDir = path.resolve(dir, id); - // Belt-and-braces containment: the id regex already forbids traversal. - if (!styleDir.startsWith(path.resolve(dir) + path.sep)) return { profile: null }; - return { - profile: readJsonMaybe(path.join(styleDir, "profile.json")), - kind: "library", - analyzeArgs: ["--styleDir", styleDir], - profilePath: path.join(styleDir, "profile.json"), - }; - } - return { profile: null }; -} - -// Turn a profile into concrete directives + retrieved exemplars for the prompt. -function styleBlock(profile) { - if (!profile) return "(none — use the prompt and general best practices)"; - const parts = []; - parts.push( - profile.referenceMode === "inspired" - ? "REFERENCE MODE: inspired — treat the references as a loose vibe; capture the mood and energy but freely depart from their exact structure." - : "REFERENCE MODE: literal — imitate the references' edit structure closely (hook shape, pacing, caption/text treatment, transitions).", - ); - if (profile.styleGuide) parts.push(`STYLE GUIDE:\n${profile.styleGuide}`); - const d = []; - if (profile.palette?.length) d.push(`palette [text,bg,accent] = ${profile.palette.slice(0, 3).join(", ")}`); - if (profile.fontFamily) d.push(`fontFamily = ${profile.fontFamily}`); - if (profile.captionStyle) d.push(`captionStyle = ${profile.captionStyle}`); - if (profile.grade) d.push(`grade = ${JSON.stringify(profile.grade)}`); - if (profile.pacing?.cutsPer10s) d.push(`pacing ~= ${profile.pacing.cutsPer10s} cuts/10s`); - if (profile.targetLengthSec) d.push(`target length ~= ${profile.targetLengthSec}s`); - if (profile.hookPattern) d.push(`hook = ${profile.hookPattern}`); - if (profile.textTreatment) d.push(`text treatment = ${profile.textTreatment}`); - if (profile.transitions?.length) d.push(`transitions = ${profile.transitions.join(", ")}`); - if (d.length) parts.push(`DIRECTIVES:\n- ${d.join("\n- ")}`); - if (profile.do?.length) parts.push(`DO: ${profile.do.join("; ")}`); - if (profile.avoid?.length) parts.push(`AVOID: ${profile.avoid.join("; ")}`); - const ex = (profile.exemplars ?? []).slice(0, 3); - if (ex.length) parts.push(`EXEMPLARS (imitate this edit structure):\n${JSON.stringify(ex, null, 2)}`); - return parts.join("\n\n"); -} - -function formatLabel(baselineJson) { - try { - const f = JSON.parse(baselineJson).format ?? {}; - const w = f.width ?? 1080; - const h = f.height ?? 1920; - const orient = w === h ? "square 1:1" : w > h ? "landscape 16:9" : "vertical 9:16"; - return `${orient}, ${w}x${h} @ ${f.fps ?? 30}fps`; - } catch { - return "vertical 9:16, 1080x1920 @ 30fps"; - } -} - -function buildPrompt(baselineJson, promptMd, profile) { - return [ - `You are an expert short-form (${formatLabel(baselineJson)}) video editor.`, - "Refine the BASELINE edit decision list (edl.json) into a polished first cut that closely matches the creator's STYLE.", - "", - "Return ONLY a single JSON object — the complete updated edl.json. No prose, no code fences.", - "", - "Rules:", - "- Keep the exact same shape/keys as the baseline (it is already schema-valid).", - "- Only use assets that exist in the baseline `assets` array (same `id` and `src`).", - "- Keep every audio track and its clips from the baseline (music bed, voiceover) — never remove or silence them unless the CREATOR PROMPT explicitly asks.", - "- Make the first ~2 seconds a strong hook; reorder/trim video clips to match the target pacing.", - "- Add a `text` track with title/subtitle overlays derived from the prompt, in the style's text treatment.", - `- Every text clip MUST have this exact shape: {"id":"t1","start":0.2,"end":2.6,"text":"...","style":"title"|"subtitle","anim":{"name":"","from":"animate-text"}}. The anim.name field is REQUIRED — never omit it. Or omit the whole "anim" key.`, - "- Set theme.palette, theme.fontFamily, theme.captionStyle, and theme.grade to match the STYLE.", - "- Respect theme.safeMargins; keep captions/text out of platform UI zones.", - "- Mirror the STYLE's pacing, hook, transitions, and text treatment as closely as the available clips allow.", - "", - "=== CREATOR PROMPT (prompt.md) ===", - promptMd || "(none)", - "", - "=== STYLE (learned from the creator's reference videos) ===", - styleBlock(profile), - "", - "=== BASELINE edl.json ===", - baselineJson, - ].join("\n"); -} - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - if (!isLlmConfigured()) { - console.error("ERROR no LLM credentials configured (set APERTURE_LLM_API_KEY or OPENAI_API_KEY)."); - process.exit(3); - } - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const edlPath = path.join(projectDir, "edl.json"); - - // 1) Deterministic baseline (probes clips, writes a valid edl.json skeleton). - console.log("PHASE assembling baseline"); - const baseline = spawnSync("node", [path.join(__dirname, "analyze.mjs"), "--slug", slug], { - cwd: repoRoot, - encoding: "utf8", - }); - if (baseline.status !== 0) { - console.error(`ERROR baseline assembly failed: ${baseline.stderr || baseline.status}`); - process.exit(2); - } - - const baselineJson = readMaybe(edlPath); - const promptMd = readMaybe(path.join(projectDir, "prompt.md")); - - // Resolve the active style and lazily analyze it (once, cached) so the user - // never has to click Analyze/Use manually. - const active = resolveActiveStyle(projectDir, slug); - if (active.analyzeArgs && !isAnalyzed(active.profile)) { - console.log("PHASE learning your style"); - const r = spawnSync("node", [path.join(__dirname, "analyze-collection.mjs"), ...active.analyzeArgs], { - cwd: repoRoot, - encoding: "utf8", - }); - if (r.status === 0) active.profile = readJsonMaybe(active.profilePath) ?? active.profile; - // If analysis fails (e.g. no source clips), continue with whatever profile exists. - } - const profile = active.profile; - - const { provider, model } = llmConfig(); - console.log(`PHASE editing with ${provider}/${model}${profile ? ` (style: ${profile.name ?? profile.id})` : ""}`); - - const llm = resolveModel(); - const base = buildPrompt(baselineJson, promptMd, profile); - let prompt = base; - let edl = null; - let lastError = ""; - - for (let attempt = 0; attempt < 2 && !edl; attempt++) { - const { text } = await generateText({ - model: llm, - prompt, - maxOutputTokens: MAX_OUTPUT_TOKENS, - // Reasoning models (e.g. gpt-5.5) reject `temperature`; keep effort modest - // for latency/cost. providerOptions is ignored by non-OpenAI providers. - providerOptions: { openai: { reasoningEffort: reasoningEffort() } }, - }); - try { - const candidate = sanitizeEdl(extractJson(text)); - const parsed = parseEdl(candidate); - if (parsed.ok && parsed.edl) { - edl = parsed.edl; - } else { - lastError = (parsed.errors ?? ["invalid edl"]).join("; "); - prompt = `${base}\n\nYour previous output was invalid: ${lastError}\nReturn corrected JSON only.`; - } - } catch (err) { - lastError = String(err); - prompt = `${base}\n\nYour previous output could not be parsed: ${lastError}\nReturn a single valid JSON object only.`; - } - } - - if (!edl) { - console.error(`ERROR model did not produce a valid edl.json (${lastError}). Baseline was kept.`); - process.exit(2); - } - - // Stamp the measurable look so the style shows even if the model under-applied it, - // and re-attach any audio the model dropped (music bed / voiceover). - edl = enforceStyle(edl, profile); - try { - edl = restoreAudioTracks(edl, JSON.parse(baselineJson)); - } catch { - // baseline unreadable — keep the model's cut as-is - } - - console.log("PHASE writing edl.json"); - fs.writeFileSync(edlPath, `${JSON.stringify(edl, null, 2)}\n`); - console.log(`DONE generated ${slug}/edl.json`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/import.mjs b/app/scripts/import.mjs new file mode 100644 index 0000000..8c526b3 --- /dev/null +++ b/app/scripts/import.mjs @@ -0,0 +1,245 @@ +// Import media into the Keeper library. +// +// node app/scripts/import.mjs --source [--source ...] [--library ] +// +// Stages (each checkpointed so a crash resumes via reprocess.mjs): +// scan -> copy (hash, dedupe, verify) -> derive (thumbs/posters) -> +// cv (quality + verdict) -> group (pairs + bursts) -> embed (CLIP) +// +// Emits the PHASE / PROGRESS / DONE line protocol the app parses; errors for +// individual files go to stderr and the import continues. +import crypto from "node:crypto"; +import fs from "node:fs"; +import path from "node:path"; +import { openCatalog } from "./lib/catalog.mjs"; +import { classifyFile, closeMetadataReader, probeDurationSec, readMetadata } from "./lib/media.mjs"; +import { ensureLayout, libraryRoot, resolveHome } from "./lib/paths.mjs"; +import { pool, stageCv, stageDerive, stageEmbed, stageGroup } from "./lib/pipeline.mjs"; + +function args(name) { + const out = []; + for (let i = 0; i < process.argv.length; i++) { + if (process.argv[i] === `--${name}` && process.argv[i + 1]) out.push(process.argv[i + 1]); + } + return out; +} + +const progress = (pct) => console.log(`PROGRESS ${Math.round(pct)}`); +const phase = (name) => console.log(`PHASE ${name}`); + +/** Range-mapped progress: stage share of the overall bar. */ +const span = (from, to) => (frac) => progress(from + (to - from) * Math.min(1, Math.max(0, frac))); + +function walk(dir, out) { + let entries; + try { + entries = fs.readdirSync(dir, { withFileTypes: true }); + } catch { + return out; + } + for (const entry of entries) { + if (entry.name.startsWith(".")) continue; + const full = path.join(dir, entry.name); + if (entry.isDirectory()) walk(full, out); + else if (entry.isFile() && classifyFile(entry.name)) out.push(full); + } + return out; +} + +function sha256(file) { + return new Promise((resolve, reject) => { + const hash = crypto.createHash("sha256"); + const stream = fs.createReadStream(file); + stream.on("error", reject); + stream.on("data", (chunk) => hash.update(chunk)); + stream.on("end", () => resolve(hash.digest("hex"))); + }); +} + +function dayFolder(capturedAt, mtime) { + const iso = capturedAt ?? new Date(mtime).toISOString(); + const day = iso.slice(0, 10); + const year = day.slice(0, 4); + return path.join(year, day); +} + +/** Copy with a unique name if a different file already sits at the target. */ +function uniqueDest(destDir, fileName) { + let candidate = path.join(destDir, fileName); + if (!fs.existsSync(candidate)) return candidate; + const dot = fileName.lastIndexOf("."); + const stem = dot > 0 ? fileName.slice(0, dot) : fileName; + const ext = dot > 0 ? fileName.slice(dot) : ""; + for (let i = 2; i < 10_000; i++) { + candidate = path.join(destDir, `${stem}-${i}${ext}`); + if (!fs.existsSync(candidate)) return candidate; + } + throw new Error(`cannot find a unique name for ${fileName}`); +} + +async function main() { + const sources = args("source"); + if (sources.length === 0) throw new Error("missing --source"); + const home = ensureLayout(resolveHome(args("library")[0])); + const catalog = openCatalog(home, { stampOnWrite: true }); + + const importId = `imp-${Date.now().toString(36)}`; + const manifest = catalog.saveImport({ + id: importId, + sources: sources.slice(0, 64), + startedAt: new Date().toISOString(), + stage: "scan", + status: "running", + counts: {}, + }); + + const saveManifest = (patch) => { + Object.assign(manifest, patch, { counts: { ...manifest.counts, ...(patch.counts ?? {}) } }); + catalog.saveImport(manifest); + }; + + try { + // -- scan --------------------------------------------------------------- + phase("scanning"); + const files = []; + for (const source of sources) { + const stat = fs.statSync(source, { throwIfNoEntry: false }); + if (!stat) continue; + if (stat.isDirectory()) walk(source, files); + else if (stat.isFile() && classifyFile(source)) files.push(source); + } + saveManifest({ stage: "copy", counts: { found: files.length } }); + span(0, 5)(1); + if (files.length === 0) { + saveManifest({ stage: "done", status: "done", finishedAt: new Date().toISOString() }); + console.log(`DONE ${JSON.stringify({ importId, found: 0, copied: 0, duplicates: 0, failed: 0 })}`); + return; + } + + // -- copy + metadata ---------------------------------------------------- + phase(`copying ${files.length} files`); + const imported = []; + let duplicates = 0; + let failed = 0; + const copyProgress = span(5, 40); + + await pool( + files, + async (source) => { + const fileName = path.basename(source); + const kind = classifyFile(fileName); + const stat = fs.statSync(source); + const contentHash = await sha256(source); + if (catalog.findByHash(contentHash)) { + duplicates++; + return; + } + + let meta; + try { + meta = await readMetadata(source); + } catch { + meta = { exif: {} }; + } + if (kind.mediaType === "video" && meta.durationSec === undefined) { + meta.durationSec = await probeDurationSec(source); + } + + // No EXIF date -> fall back to file mtime so date bucketing and burst + // clustering still work; dateSource records the weaker provenance. + const capturedAt = meta.capturedAt ?? new Date(stat.mtimeMs).toISOString(); + + const destDir = path.join(libraryRoot(home), dayFolder(meta.capturedAt, stat.mtimeMs)); + fs.mkdirSync(destDir, { recursive: true }); + const dest = uniqueDest(destDir, fileName); + fs.copyFileSync(source, dest); + const copiedStat = fs.statSync(dest); + if (copiedStat.size !== stat.size) { + fs.rmSync(dest, { force: true }); + throw new Error(`size mismatch copying ${fileName} (card removed mid-copy?)`); + } + + const asset = { + id: contentHash.slice(0, 16), + relPath: path.relative(home, dest), + fileName: path.basename(dest), + byteSize: copiedStat.size, + contentHash, + mediaType: kind.mediaType, + format: kind.format, + width: meta.width, + height: meta.height, + durationSec: meta.durationSec, + capturedAt, + dateSource: meta.capturedAt ? "exif" : "mtime", + exif: meta.exif ?? {}, + importId, + importedAt: new Date().toISOString(), + stages: { copy: new Date().toISOString() }, + }; + catalog.upsertAsset(asset); + imported.push(asset); + }, + { + concurrency: 3, + onDone: (n) => { + copyProgress(n / files.length); + if (n % 25 === 0) saveManifest({ counts: { copied: imported.length, duplicates } }); + }, + }, + ).then((errors) => { + failed = errors.length; + }); + + saveManifest({ + stage: "derive", + counts: { copied: imported.length, duplicates, failed }, + }); + + // -- derive ------------------------------------------------------------- + phase(`generating previews for ${imported.length} files`); + await stageDerive(catalog, imported, { onProgress: span(40, 70) }); + + // -- cv ----------------------------------------------------------------- + phase("measuring quality"); + saveManifest({ stage: "cv" }); + const { hashes } = await stageCv(catalog, imported, { onProgress: span(70, 84) }); + + // -- group -------------------------------------------------------------- + phase("grouping bursts and pairs"); + saveManifest({ stage: "group" }); + const groupStats = stageGroup(catalog, imported, hashes); + span(84, 88)(1); + + // -- embed -------------------------------------------------------------- + phase("indexing for search"); + saveManifest({ stage: "embed" }); + const embedStats = await stageEmbed(catalog, imported, { onProgress: span(88, 99) }); + + saveManifest({ stage: "done", status: "done", finishedAt: new Date().toISOString() }); + progress(100); + console.log( + `DONE ${JSON.stringify({ + importId, + found: files.length, + copied: imported.length, + duplicates, + failed, + groups: groupStats.groups, + paired: groupStats.paired, + embedded: embedStats.embedded, + })}`, + ); + } catch (err) { + saveManifest({ status: "failed", error: String(err?.message ?? err) }); + throw err; + } finally { + await closeMetadataReader(); + catalog.close(); + } +} + +main().catch((err) => { + console.error(`ERROR ${err?.stack ?? err}`); + process.exit(1); +}); diff --git a/app/scripts/judge-llm.mjs b/app/scripts/judge-llm.mjs new file mode 100644 index 0000000..60563fd --- /dev/null +++ b/app/scripts/judge-llm.mjs @@ -0,0 +1,182 @@ +// The LLM judge: GPT-5.5 vision over downscaled thumbs, only where local CV +// couldn't decide — borderline verdicts and burst best-pick ties. Budgeted +// (--budget N, default from KEEPER_AI_BUDGET or 200 items), batched, and its +// output is repaired-then-validated before a single row changes. +// +// node app/scripts/judge-llm.mjs [--library ] [--budget 200] [--import ] +import fs from "node:fs"; +import { generateText } from "ai"; +import { openCatalog } from "./lib/catalog.mjs"; +import { ensureLayout, resolveHome, safeJoin } from "./lib/paths.mjs"; +import { readTaste, tasteForPrompt } from "./lib/taste.mjs"; +import { parseJudgeResponse } from "./lib/llm-util.mjs"; +import { isLlmConfigured, llmConfig, resolveModel, reasoningEffort } from "./llm.mjs"; + +const BATCH_SIZE = 8; +const MAX_OUTPUT_TOKENS = 4000; + +function arg(name) { + const i = process.argv.indexOf(`--${name}`); + return i >= 0 ? process.argv[i + 1] : undefined; +} + +const phase = (name) => console.log(`PHASE ${name}`); +const progress = (pct) => console.log(`PROGRESS ${Math.round(pct)}`); + +/** Borderline = the confidence band where a second opinion changes routing. */ +function isBorderline(asset) { + if (!asset.ai) return true; + if (asset.ai.source === "llm") return false; // already judged + return asset.ai.confidence >= 0.3 && asset.ai.confidence <= 0.8; +} + +function systemPrompt(taste) { + const tasteBlock = tasteForPrompt(taste); + return [ + "You are a photo culling assistant inside Keeper. You see downscaled previews of a photographer's shots that the local heuristics could not confidently sort.", + "For each image decide: keep (worth reviewing/editing), reject (junk: blurry, misfire, badly exposed, meaningless), or review (genuinely ambiguous).", + "IMPORTANT judgment calls:", + "- A technically imperfect photo of a meaningful moment (people, emotion, once-in-a-lifetime scenes) is a KEEP with reason important-moment. Never reject for sharpness alone when the content clearly matters.", + "- Duplicates/burst frames: prefer open eyes, natural expressions, better framing.", + "- Be honest about confidence: 0.9+ only when unmistakable.", + tasteBlock ? `\n${tasteBlock}` : "", + "\nRespond with ONLY a JSON object:", + '{"items":[{"assetId":"...","suggestion":"keep|reject|review","confidence":0.0,"reason":"blurry|eyes-closed|bad-framing|important-moment|llm-quality|other","detail":"short why","caption":"one-line description","tags":["lowercase","keywords"]}],"bestPicks":{"":""}}', + ].join("\n"); +} + +async function judgeBatch(model, taste, batch, groupsInBatch) { + const content = []; + for (const asset of batch) { + const label = [ + `assetId: ${asset.id}`, + asset.groupId ? `group: ${asset.groupId}` : null, + asset.ai ? `local verdict: ${asset.ai.suggestion} (${asset.ai.reasons.map((r) => r.code).join(", ")})` : null, + asset.mediaType === "video" ? `video poster frame, ${asset.durationSec ?? "?"}s` : null, + ] + .filter(Boolean) + .join(" | "); + content.push({ type: "text", text: label }); + content.push({ type: "image", image: asset.imageBytes }); + } + if (groupsInBatch.length > 0) { + content.push({ + type: "text", + text: `Burst groups present: ${groupsInBatch.join(", ")} — pick the best frame per group in bestPicks.`, + }); + } + + const { text } = await generateText({ + model, + system: systemPrompt(taste), + messages: [{ role: "user", content }], + maxOutputTokens: MAX_OUTPUT_TOKENS, + providerOptions: { openai: { reasoningEffort: reasoningEffort() } }, + }); + + const ids = batch.map((a) => a.id); + try { + return parseJudgeResponse(text, ids); + } catch (err) { + // One repair retry, carrying the validation error back to the model. + const { text: retryText } = await generateText({ + model, + system: systemPrompt(taste), + messages: [ + { role: "user", content }, + { role: "assistant", content: text }, + { + role: "user", + content: `Your response failed validation: ${err.message}. Reply again with ONLY the corrected JSON object.`, + }, + ], + maxOutputTokens: MAX_OUTPUT_TOKENS, + providerOptions: { openai: { reasoningEffort: reasoningEffort() } }, + }); + return parseJudgeResponse(retryText, ids); + } +} + +async function main() { + if (!isLlmConfigured()) { + throw new Error("no model configured — add an API key in Settings or app/.env.local"); + } + const home = ensureLayout(resolveHome(arg("library"))); + const catalog = openCatalog(home, { stampOnWrite: true }); + const budget = Math.max(1, Number(arg("budget")) || Number(process.env.KEEPER_AI_BUDGET) || 200); + const importId = arg("import"); + const taste = readTaste(home); + const { model: modelId } = llmConfig(); + + try { + phase("selecting borderline items"); + const candidates = catalog + .listAssets({ importId, flag: "unrated", limit: 20_000 }) + .filter((a) => isBorderline(a) && a.thumbRel) + .slice(0, budget); + + if (candidates.length === 0) { + console.log(`DONE ${JSON.stringify({ judged: 0, batches: 0 })}`); + return; + } + + phase(`judging ${candidates.length} items with ${modelId}`); + const model = resolveModel(); + let judged = 0; + let batches = 0; + + for (let i = 0; i < candidates.length; i += BATCH_SIZE) { + const slice = candidates.slice(i, i + BATCH_SIZE); + const batch = []; + for (const asset of slice) { + try { + batch.push({ ...asset, imageBytes: fs.readFileSync(safeJoin(home, asset.thumbRel)) }); + } catch { + // unreadable thumb — skip this asset + } + } + if (batch.length === 0) continue; + const groupsInBatch = [...new Set(batch.map((a) => a.groupId).filter(Boolean))]; + + try { + const result = await judgeBatch(model, taste, batch, groupsInBatch); + for (const item of result.items) { + catalog.setAiVerdict(item.assetId, { + suggestion: item.suggestion, + confidence: item.confidence, + reasons: [{ code: item.reason, detail: item.detail }], + source: "llm", + model: modelId, + at: new Date().toISOString(), + }); + const patch = {}; + if (item.caption) patch.caption = item.caption; + if (item.tags && item.tags.length > 0) { + const existing = catalog.getAsset(item.assetId)?.tags ?? []; + patch.tags = [...new Set([...existing, ...item.tags])]; + } + if (Object.keys(patch).length > 0) catalog.patchAsset(item.assetId, patch); + catalog.markStage(item.assetId, "judge"); + judged++; + } + for (const [groupId, assetId] of Object.entries(result.bestPicks)) { + if (catalog.getGroup(groupId)) catalog.setGroupPick(groupId, assetId, "llm"); + } + batches++; + } catch (err) { + console.error(`ERROR batch failed: ${err?.message ?? err}`); + } + progress(Math.min(99, ((i + BATCH_SIZE) / candidates.length) * 100)); + } + + progress(100); + console.log(`DONE ${JSON.stringify({ judged, batches, model: modelId })}`); + } finally { + catalog.close(); + } +} + +main().catch((err) => { + console.error(`ERROR ${err?.stack ?? err}`); + process.exit(1); +}); diff --git a/app/scripts/lib/catalog.mjs b/app/scripts/lib/catalog.mjs new file mode 100644 index 0000000..b9740f7 --- /dev/null +++ b/app/scripts/lib/catalog.mjs @@ -0,0 +1,538 @@ +// The catalog: one SQLite database per library (/.keeper/catalog.db) +// holding every asset record, burst group, and import manifest, plus the +// embedding vectors for semantic search. Uses node:sqlite (Node >= 22.13) so +// there is no native build step; WAL mode + busy_timeout make it safe for the +// app's catalog service and pipeline scripts to write concurrently. +// +// Rows are stored as hot columns (everything we filter/sort on) plus JSON +// side-columns for the rest. Every row is re-validated through +// AssetRecordSchema on read — catalog rows can be written by CLI tools and +// agents, so they are treated as untrusted input. +import { DatabaseSync } from "node:sqlite"; +import fs from "node:fs"; +import path from "node:path"; +import { + AssetRecordSchema, + GroupSchema, + ImportManifestSchema, +} from "@keeper/schema"; +import { catalogPath, keeperDir, touchStamp } from "./paths.mjs"; + +const SCHEMA_VERSION = 1; + +const CREATE_SQL = ` +CREATE TABLE IF NOT EXISTS assets ( + id TEXT PRIMARY KEY, + content_hash TEXT NOT NULL UNIQUE, + rel_path TEXT NOT NULL, + file_name TEXT NOT NULL, + media_type TEXT NOT NULL, + format TEXT NOT NULL, + byte_size INTEGER NOT NULL, + width INTEGER, + height INTEGER, + duration_sec REAL, + captured_at TEXT, + date_source TEXT NOT NULL DEFAULT 'unknown', + import_id TEXT, + imported_at TEXT, + group_id TEXT, + pair_role TEXT NOT NULL DEFAULT 'primary', + pair_primary_id TEXT, + offline INTEGER NOT NULL DEFAULT 0, + thumb_rel TEXT, + preview_rel TEXT, + poster_rel TEXT, + scrub_rel TEXT, + exif_json TEXT NOT NULL DEFAULT '{}', + quality_json TEXT NOT NULL DEFAULT '{}', + ai_json TEXT, + user_flag TEXT NOT NULL DEFAULT 'unrated', + user_rating INTEGER NOT NULL DEFAULT 0, + user_at TEXT, + tags_json TEXT NOT NULL DEFAULT '[]', + caption TEXT, + stages_json TEXT NOT NULL DEFAULT '{}', + embedding BLOB, + embed_model TEXT +); +CREATE INDEX IF NOT EXISTS idx_assets_captured ON assets(captured_at); +CREATE INDEX IF NOT EXISTS idx_assets_flag ON assets(user_flag); +CREATE INDEX IF NOT EXISTS idx_assets_group ON assets(group_id); +CREATE INDEX IF NOT EXISTS idx_assets_import ON assets(import_id); + +CREATE TABLE IF NOT EXISTS groups ( + id TEXT PRIMARY KEY, + kind TEXT NOT NULL, + best_pick_id TEXT, + pick_source TEXT +); + +CREATE TABLE IF NOT EXISTS imports ( + id TEXT PRIMARY KEY, + started_at TEXT NOT NULL, + manifest_json TEXT NOT NULL +); +`; + +function rowToAsset(row) { + const record = { + id: row.id, + relPath: row.rel_path, + fileName: row.file_name, + byteSize: row.byte_size, + contentHash: row.content_hash, + mediaType: row.media_type, + format: row.format, + width: row.width ?? undefined, + height: row.height ?? undefined, + durationSec: row.duration_sec ?? undefined, + capturedAt: row.captured_at ?? undefined, + dateSource: row.date_source, + exif: JSON.parse(row.exif_json || "{}"), + importId: row.import_id ?? undefined, + importedAt: row.imported_at ?? undefined, + groupId: row.group_id ?? undefined, + pairRole: row.pair_role, + pairPrimaryId: row.pair_primary_id ?? undefined, + offline: Boolean(row.offline), + thumbRel: row.thumb_rel ?? undefined, + previewRel: row.preview_rel ?? undefined, + posterRel: row.poster_rel ?? undefined, + scrubRel: row.scrub_rel ?? undefined, + quality: JSON.parse(row.quality_json || "{}"), + ai: row.ai_json ? JSON.parse(row.ai_json) : undefined, + user: { flag: row.user_flag, rating: row.user_rating, at: row.user_at ?? undefined }, + tags: JSON.parse(row.tags_json || "[]"), + caption: row.caption ?? undefined, + stages: JSON.parse(row.stages_json || "{}"), + }; + return AssetRecordSchema.parse(record); +} + +function assetToParams(asset) { + const a = AssetRecordSchema.parse(asset); + return { + id: a.id, + content_hash: a.contentHash, + rel_path: a.relPath, + file_name: a.fileName, + media_type: a.mediaType, + format: a.format, + byte_size: a.byteSize, + width: a.width ?? null, + height: a.height ?? null, + duration_sec: a.durationSec ?? null, + captured_at: a.capturedAt ?? null, + date_source: a.dateSource, + import_id: a.importId ?? null, + imported_at: a.importedAt ?? null, + group_id: a.groupId ?? null, + pair_role: a.pairRole, + pair_primary_id: a.pairPrimaryId ?? null, + offline: a.offline ? 1 : 0, + thumb_rel: a.thumbRel ?? null, + preview_rel: a.previewRel ?? null, + poster_rel: a.posterRel ?? null, + scrub_rel: a.scrubRel ?? null, + exif_json: JSON.stringify(a.exif), + quality_json: JSON.stringify(a.quality), + ai_json: a.ai ? JSON.stringify(a.ai) : null, + user_flag: a.user.flag, + user_rating: a.user.rating, + user_at: a.user.at ?? null, + tags_json: JSON.stringify(a.tags), + caption: a.caption ?? null, + stages_json: JSON.stringify(a.stages), + }; +} + +const UPSERT_COLS = [ + "id", "content_hash", "rel_path", "file_name", "media_type", "format", "byte_size", + "width", "height", "duration_sec", "captured_at", "date_source", "import_id", "imported_at", + "group_id", "pair_role", "pair_primary_id", "offline", "thumb_rel", "preview_rel", + "poster_rel", "scrub_rel", "exif_json", "quality_json", "ai_json", "user_flag", + "user_rating", "user_at", "tags_json", "caption", "stages_json", +]; + +/** + * Open (creating if needed) the catalog for a library home. + * Options: { stampOnWrite } — CLI/pipeline contexts touch the stamp file after + * writes so the running app knows to refresh; the app's own service must NOT + * stamp or it would refresh itself in a loop. + */ +export function openCatalog(home, { stampOnWrite = false } = {}) { + fs.mkdirSync(keeperDir(home), { recursive: true }); + const db = new DatabaseSync(catalogPath(home)); + db.exec("PRAGMA journal_mode = WAL"); + db.exec("PRAGMA busy_timeout = 5000"); + db.exec("PRAGMA synchronous = NORMAL"); + + const version = db.prepare("PRAGMA user_version").get().user_version; + if (version < SCHEMA_VERSION) { + db.exec(CREATE_SQL); + db.exec(`PRAGMA user_version = ${SCHEMA_VERSION}`); + } + + const stamp = () => { + if (stampOnWrite) touchStamp(home); + }; + + const upsertStmt = db.prepare(` + INSERT INTO assets (${UPSERT_COLS.join(", ")}) + VALUES (${UPSERT_COLS.map((c) => `:${c}`).join(", ")}) + ON CONFLICT(id) DO UPDATE SET ${UPSERT_COLS.filter((c) => c !== "id") + .map((c) => `${c} = :${c}`) + .join(", ")} + `); + + const getStmt = db.prepare("SELECT * FROM assets WHERE id = ?"); + + const api = { + db, + home, + + close() { + db.close(); + }, + + upsertAsset(asset) { + upsertStmt.run(assetToParams(asset)); + stamp(); + }, + + upsertAssets(assets) { + db.exec("BEGIN"); + try { + for (const asset of assets) upsertStmt.run(assetToParams(asset)); + db.exec("COMMIT"); + } catch (err) { + db.exec("ROLLBACK"); + throw err; + } + stamp(); + }, + + getAsset(id) { + const row = getStmt.get(id); + return row ? rowToAsset(row) : null; + }, + + getAssets(ids) { + const out = []; + for (const id of ids) { + const row = getStmt.get(id); + if (row) out.push(rowToAsset(row)); + } + return out; + }, + + /** Content-hash dedupe across the whole library. Returns existing id or null. */ + findByHash(contentHash) { + const row = db.prepare("SELECT id FROM assets WHERE content_hash = ?").get(contentHash); + return row ? row.id : null; + }, + + /** + * List assets with the filters the UI actually uses. Non-primary pair + * members (RAW siblings, Live Photo videos) are hidden unless includePaired. + */ + listAssets({ + flag, + mediaType, + minRating, + importId, + groupId, + pairPrimaryId, + day, + ids, + includePaired = false, + hasAi, + limit = 100_000, + offset = 0, + order = "captured_desc", + } = {}) { + const where = []; + const params = {}; + if (!includePaired) where.push("pair_role = 'primary'"); + if (flag) { + where.push("user_flag = :flag"); + params.flag = flag; + } + if (mediaType) { + where.push("media_type = :mediaType"); + params.mediaType = mediaType; + } + if (typeof minRating === "number" && minRating > 0) { + where.push("user_rating >= :minRating"); + params.minRating = minRating; + } + if (importId) { + where.push("import_id = :importId"); + params.importId = importId; + } + if (groupId) { + where.push("group_id = :groupId"); + params.groupId = groupId; + } + if (pairPrimaryId) { + where.push("pair_primary_id = :pairPrimaryId"); + params.pairPrimaryId = pairPrimaryId; + } + if (day) { + where.push("substr(coalesce(captured_at, imported_at, ''), 1, 10) = :day"); + params.day = day; + } + if (hasAi === true) where.push("ai_json IS NOT NULL"); + if (hasAi === false) where.push("ai_json IS NULL"); + if (ids && ids.length > 0) { + // ids come from our own search results; still cap and inline safely. + const list = ids.slice(0, 2000).map((_, i) => `:id${i}`); + ids.slice(0, 2000).forEach((id, i) => { + params[`id${i}`] = id; + }); + where.push(`id IN (${list.join(",")})`); + } + const orderSql = + order === "captured_asc" + ? "ORDER BY coalesce(captured_at, imported_at) ASC, file_name ASC" + : "ORDER BY coalesce(captured_at, imported_at) DESC, file_name DESC"; + const sql = `SELECT * FROM assets ${where.length ? `WHERE ${where.join(" AND ")}` : ""} ${orderSql} LIMIT :limit OFFSET :offset`; + params.limit = Math.min(limit, 100_000); + params.offset = offset; + return db.prepare(sql).all(params).map(rowToAsset); + }, + + /** Date buckets (YYYY-MM-DD) with counts, newest first — the grid skeleton. */ + listDays({ flag, mediaType } = {}) { + const where = ["pair_role = 'primary'"]; + const params = {}; + if (flag) { + where.push("user_flag = :flag"); + params.flag = flag; + } + if (mediaType) { + where.push("media_type = :mediaType"); + params.mediaType = mediaType; + } + const sql = ` + SELECT substr(coalesce(captured_at, imported_at, 'unknown'), 1, 10) AS day, COUNT(*) AS n + FROM assets WHERE ${where.join(" AND ")} + GROUP BY day ORDER BY day DESC + `; + return db.prepare(sql).all(params).map((r) => ({ day: r.day || "unknown", count: r.n })); + }, + + countsSummary() { + const row = db + .prepare( + `SELECT + COUNT(*) AS total, + SUM(CASE WHEN media_type = 'photo' THEN 1 ELSE 0 END) AS photos, + SUM(CASE WHEN media_type = 'video' THEN 1 ELSE 0 END) AS videos, + SUM(CASE WHEN user_flag = 'pick' THEN 1 ELSE 0 END) AS picks, + SUM(CASE WHEN user_flag = 'reject' THEN 1 ELSE 0 END) AS rejects, + SUM(CASE WHEN user_flag = 'unrated' THEN 1 ELSE 0 END) AS unrated, + SUM(CASE WHEN user_flag = 'unrated' AND ai_json IS NOT NULL THEN 1 ELSE 0 END) AS suggested, + SUM(byte_size) AS bytes + FROM assets WHERE pair_role = 'primary'`, + ) + .get(); + return { + total: row.total ?? 0, + photos: row.photos ?? 0, + videos: row.videos ?? 0, + picks: row.picks ?? 0, + rejects: row.rejects ?? 0, + unrated: row.unrated ?? 0, + suggested: row.suggested ?? 0, + bytes: row.bytes ?? 0, + }; + }, + + /** + * Set the user verdict on a set of assets. Returns the prior state of each + * affected asset so the caller can detect AI overrides (the taste signal) + * and build undo entries. + */ + setUserVerdict(ids, { flag, rating }) { + const before = api.getAssets(ids); + const at = new Date().toISOString(); + db.exec("BEGIN"); + try { + for (const id of ids) { + if (flag !== undefined && rating !== undefined) { + db.prepare("UPDATE assets SET user_flag = ?, user_rating = ?, user_at = ? WHERE id = ?").run( + flag, + rating, + at, + id, + ); + } else if (flag !== undefined) { + db.prepare("UPDATE assets SET user_flag = ?, user_at = ? WHERE id = ?").run(flag, at, id); + } else if (rating !== undefined) { + db.prepare("UPDATE assets SET user_rating = ?, user_at = ? WHERE id = ?").run(rating, at, id); + } + } + db.exec("COMMIT"); + } catch (err) { + db.exec("ROLLBACK"); + throw err; + } + stamp(); + return before; + }, + + setAiVerdict(id, ai) { + db.prepare("UPDATE assets SET ai_json = ? WHERE id = ?").run(JSON.stringify(ai), id); + stamp(); + }, + + /** Patch derived-artifact paths and media dimensions after a stage runs. */ + patchAsset(id, fields) { + const allowed = { + thumbRel: "thumb_rel", + previewRel: "preview_rel", + posterRel: "poster_rel", + scrubRel: "scrub_rel", + caption: "caption", + width: "width", + height: "height", + durationSec: "duration_sec", + groupId: "group_id", + offline: "offline", + }; + const sets = []; + const params = { id }; + for (const [key, col] of Object.entries(allowed)) { + if (fields[key] !== undefined) { + sets.push(`${col} = :${key}`); + params[key] = key === "offline" ? (fields[key] ? 1 : 0) : fields[key]; + } + } + if (fields.quality !== undefined) { + sets.push("quality_json = :quality"); + params.quality = JSON.stringify(fields.quality); + } + if (fields.tags !== undefined) { + sets.push("tags_json = :tags"); + params.tags = JSON.stringify(fields.tags.slice(0, 128)); + } + if (sets.length === 0) return; + db.prepare(`UPDATE assets SET ${sets.join(", ")} WHERE id = :id`).run(params); + stamp(); + }, + + setPairing(id, pairRole, pairPrimaryId) { + db.prepare("UPDATE assets SET pair_role = ?, pair_primary_id = ? WHERE id = ?").run( + pairRole, + pairPrimaryId ?? null, + id, + ); + stamp(); + }, + + markStage(id, stage) { + const row = getStmt.get(id); + if (!row) return; + const stages = JSON.parse(row.stages_json || "{}"); + stages[stage] = new Date().toISOString(); + db.prepare("UPDATE assets SET stages_json = ? WHERE id = ?").run(JSON.stringify(stages), id); + }, + + /** Assets that have not completed a pipeline stage (resume/reprocess). */ + assetsMissingStage(stage, limit = 10_000) { + const rows = db + .prepare( + `SELECT * FROM assets WHERE json_extract(stages_json, '$.' || ?) IS NULL + ORDER BY coalesce(captured_at, imported_at) DESC LIMIT ?`, + ) + .all(stage, limit); + return rows.map(rowToAsset); + }, + + setEmbedding(id, vector, model) { + const buf = Buffer.from(new Float32Array(vector).buffer); + db.prepare("UPDATE assets SET embedding = ?, embed_model = ? WHERE id = ?").run(buf, model, id); + }, + + /** All embeddings for brute-force search: [{ id, vec: Float32Array }]. */ + allEmbeddings() { + const rows = db + .prepare("SELECT id, embedding FROM assets WHERE embedding IS NOT NULL AND pair_role = 'primary'") + .all(); + return rows.map((r) => ({ + id: r.id, + vec: new Float32Array(r.embedding.buffer, r.embedding.byteOffset, r.embedding.byteLength / 4), + })); + }, + + embeddingCount() { + return db.prepare("SELECT COUNT(*) AS n FROM assets WHERE embedding IS NOT NULL").get().n; + }, + + upsertGroup(group) { + const g = GroupSchema.parse(group); + db.exec("BEGIN"); + try { + db.prepare( + `INSERT INTO groups (id, kind, best_pick_id, pick_source) VALUES (?, ?, ?, ?) + ON CONFLICT(id) DO UPDATE SET kind = excluded.kind, best_pick_id = excluded.best_pick_id, pick_source = excluded.pick_source`, + ).run(g.id, g.kind, g.bestPickId ?? null, g.pickSource ?? null); + for (const assetId of g.assetIds) { + db.prepare("UPDATE assets SET group_id = ? WHERE id = ?").run(g.id, assetId); + } + db.exec("COMMIT"); + } catch (err) { + db.exec("ROLLBACK"); + throw err; + } + stamp(); + }, + + getGroup(id) { + const row = db.prepare("SELECT * FROM groups WHERE id = ?").get(id); + if (!row) return null; + const members = db.prepare("SELECT id FROM assets WHERE group_id = ? ORDER BY file_name").all(id); + return GroupSchema.parse({ + id: row.id, + kind: row.kind, + assetIds: members.map((m) => m.id), + bestPickId: row.best_pick_id ?? undefined, + pickSource: row.pick_source ?? undefined, + }); + }, + + setGroupPick(groupId, bestPickId, source) { + db.prepare("UPDATE groups SET best_pick_id = ?, pick_source = ? WHERE id = ?").run( + bestPickId, + source, + groupId, + ); + stamp(); + }, + + saveImport(manifest) { + const m = ImportManifestSchema.parse(manifest); + db.prepare( + `INSERT INTO imports (id, started_at, manifest_json) VALUES (?, ?, ?) + ON CONFLICT(id) DO UPDATE SET manifest_json = excluded.manifest_json`, + ).run(m.id, m.startedAt, JSON.stringify(m)); + return m; + }, + + getImport(id) { + const row = db.prepare("SELECT manifest_json FROM imports WHERE id = ?").get(id); + return row ? ImportManifestSchema.parse(JSON.parse(row.manifest_json)) : null; + }, + + listImports(limit = 20) { + return db + .prepare("SELECT manifest_json FROM imports ORDER BY started_at DESC LIMIT ?") + .all(limit) + .map((r) => ImportManifestSchema.parse(JSON.parse(r.manifest_json))); + }, + }; + + return api; +} diff --git a/app/scripts/lib/catalog.test.mjs b/app/scripts/lib/catalog.test.mjs new file mode 100644 index 0000000..3869255 --- /dev/null +++ b/app/scripts/lib/catalog.test.mjs @@ -0,0 +1,132 @@ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { openCatalog } from "./catalog.mjs"; + +let home; +let catalog; + +const asset = (id, extra = {}) => ({ + id, + relPath: `library/2026/2026-07-04/${id}.jpg`, + fileName: `${id}.jpg`, + byteSize: 1000, + contentHash: `hash-${id}-0123456789abcdef`, + mediaType: "photo", + format: "jpeg", + capturedAt: "2026-07-04T12:00:00.000Z", + ...extra, +}); + +beforeEach(() => { + home = fs.mkdtempSync(path.join(os.tmpdir(), "keeper-test-")); + catalog = openCatalog(home); +}); + +afterEach(() => { + catalog.close(); + fs.rmSync(home, { recursive: true, force: true }); +}); + +describe("catalog", () => { + it("round-trips asset records through upsert/get", () => { + catalog.upsertAsset(asset("a1", { tags: ["ocean"], quality: { blurScore: 12 } })); + const got = catalog.getAsset("a1"); + expect(got?.fileName).toBe("a1.jpg"); + expect(got?.tags).toEqual(["ocean"]); + expect(got?.quality.blurScore).toBe(12); + expect(got?.user.flag).toBe("unrated"); + }); + + it("dedupes by content hash", () => { + catalog.upsertAsset(asset("a1")); + expect(catalog.findByHash("hash-a1-0123456789abcdef")).toBe("a1"); + expect(catalog.findByHash("hash-zz-0123456789abcdef")).toBeNull(); + }); + + it("rejects hostile rows on write (schema boundary)", () => { + expect(() => catalog.upsertAsset(asset("a1", { relPath: "../../etc/passwd" }))).toThrow(); + expect(() => catalog.upsertAsset(asset("a1", { byteSize: Infinity }))).toThrow(); + }); + + it("filters and hides paired siblings", () => { + catalog.upsertAssets([ + asset("jpg1"), + asset("raw1", { format: "raw", pairRole: "raw-sibling", pairPrimaryId: "jpg1" }), + asset("v1", { mediaType: "video", format: "mp4", durationSec: 5 }), + ]); + const all = catalog.listAssets({}); + expect(all.map((a) => a.id).sort()).toEqual(["jpg1", "v1"]); + const videos = catalog.listAssets({ mediaType: "video" }); + expect(videos.map((a) => a.id)).toEqual(["v1"]); + const withPaired = catalog.listAssets({ includePaired: true }); + expect(withPaired).toHaveLength(3); + }); + + it("buckets by day", () => { + catalog.upsertAssets([ + asset("a", { capturedAt: "2026-07-04T10:00:00.000Z" }), + asset("b", { capturedAt: "2026-07-04T11:00:00.000Z" }), + asset("c", { capturedAt: "2026-07-05T09:00:00.000Z" }), + ]); + const days = catalog.listDays(); + expect(days).toEqual([ + { day: "2026-07-05", count: 1 }, + { day: "2026-07-04", count: 2 }, + ]); + }); + + it("sets verdicts and returns prior state for undo", () => { + catalog.upsertAssets([asset("a"), asset("b")]); + const before = catalog.setUserVerdict(["a", "b"], { flag: "reject" }); + expect(before.map((b) => b.user.flag)).toEqual(["unrated", "unrated"]); + expect(catalog.getAsset("a")?.user.flag).toBe("reject"); + const summary = catalog.countsSummary(); + expect(summary.rejects).toBe(2); + }); + + it("stores groups and updates members", () => { + catalog.upsertAssets([asset("a"), asset("b")]); + catalog.upsertGroup({ id: "g1", kind: "burst", assetIds: ["a", "b"], bestPickId: "a", pickSource: "cv" }); + const group = catalog.getGroup("g1"); + expect(group?.assetIds.sort()).toEqual(["a", "b"]); + expect(catalog.getAsset("b")?.groupId).toBe("g1"); + catalog.setGroupPick("g1", "b", "user"); + expect(catalog.getGroup("g1")?.bestPickId).toBe("b"); + }); + + it("tracks pipeline stages for resume", () => { + catalog.upsertAssets([asset("a"), asset("b")]); + catalog.markStage("a", "derive"); + const missing = catalog.assetsMissingStage("derive"); + expect(missing.map((m) => m.id)).toEqual(["b"]); + }); + + it("stores and retrieves embeddings", () => { + catalog.upsertAsset(asset("a")); + const vec = new Array(8).fill(0).map((_, i) => i / 10); + catalog.setEmbedding("a", vec, "test-model"); + const all = catalog.allEmbeddings(); + expect(all).toHaveLength(1); + expect(all[0].id).toBe("a"); + expect(all[0].vec[3]).toBeCloseTo(0.3, 5); + expect(catalog.embeddingCount()).toBe(1); + }); + + it("persists import manifests", () => { + catalog.saveImport({ id: "imp1", startedAt: "2026-07-10T00:00:00Z", stage: "copy", counts: { found: 10 } }); + catalog.saveImport({ id: "imp1", startedAt: "2026-07-10T00:00:00Z", stage: "done", status: "done", counts: { found: 10, copied: 9 } }); + const imp = catalog.getImport("imp1"); + expect(imp?.stage).toBe("done"); + expect(imp?.counts.copied).toBe(9); + expect(catalog.listImports()).toHaveLength(1); + }); + + it("touches the stamp file only when asked", () => { + const stamped = openCatalog(home, { stampOnWrite: true }); + stamped.upsertAsset(asset("s1")); + expect(fs.existsSync(path.join(home, ".keeper", ".stamp"))).toBe(true); + stamped.close(); + }); +}); diff --git a/app/scripts/lib/embeddings.mjs b/app/scripts/lib/embeddings.mjs new file mode 100644 index 0000000..45fcc00 --- /dev/null +++ b/app/scripts/lib/embeddings.mjs @@ -0,0 +1,103 @@ +// Local CLIP embeddings via transformers.js (onnxruntime under the hood). +// One image vector per asset (thumbs are already normalized JPEGs), plus the +// matching text tower for natural-language search. The model downloads once +// into /.keeper/models and everything runs offline afterwards. +import fs from "node:fs"; +import path from "node:path"; +import { modelsDir } from "./paths.mjs"; + +export const EMBED_MODEL_ID = "Xenova/clip-vit-base-patch32"; + +let transformersPromise = null; +let visionPromise = null; +let textPromise = null; + +async function loadTransformers(home) { + if (!transformersPromise) { + transformersPromise = import("@huggingface/transformers").then((mod) => { + mod.env.cacheDir = modelsDir(home); + mod.env.allowLocalModels = true; + return mod; + }); + } + return transformersPromise; +} + +/** True once the model files exist locally (no network needed anymore). */ +export function modelCached(home) { + try { + // transformers.js lays the cache out as ///... + const dir = path.join(modelsDir(home), ...EMBED_MODEL_ID.split("/")); + return fs.existsSync(dir) && fs.readdirSync(dir).length > 0; + } catch { + return false; + } +} + +async function loadVision(home) { + if (!visionPromise) { + visionPromise = (async () => { + const { AutoProcessor, CLIPVisionModelWithProjection } = await loadTransformers(home); + const processor = await AutoProcessor.from_pretrained(EMBED_MODEL_ID); + const model = await CLIPVisionModelWithProjection.from_pretrained(EMBED_MODEL_ID, { + dtype: "q8", + }); + return { processor, model }; + })(); + } + return visionPromise; +} + +async function loadText(home) { + if (!textPromise) { + textPromise = (async () => { + const { AutoTokenizer, CLIPTextModelWithProjection } = await loadTransformers(home); + const tokenizer = await AutoTokenizer.from_pretrained(EMBED_MODEL_ID); + const model = await CLIPTextModelWithProjection.from_pretrained(EMBED_MODEL_ID, { + dtype: "q8", + }); + return { tokenizer, model }; + })(); + } + return textPromise; +} + +function l2Normalize(vec) { + let norm = 0; + for (const v of vec) norm += v * v; + norm = Math.sqrt(norm) || 1; + return Array.from(vec, (v) => v / norm); +} + +/** Embed one image file (thumb/poster JPEG). Returns a unit-norm number[]. */ +export async function embedImage(home, absImagePath) { + const { RawImage } = await loadTransformers(home); + const { processor, model } = await loadVision(home); + const image = await RawImage.read(absImagePath); + const inputs = await processor(image); + const { image_embeds } = await model(inputs); + return l2Normalize(image_embeds.data); +} + +/** Embed a search query with the text tower. Unit-norm number[]. */ +export async function embedText(home, query) { + const { tokenizer, model } = await loadText(home); + const inputs = tokenizer([query], { padding: true, truncation: true }); + const { text_embeds } = await model(inputs); + return l2Normalize(text_embeds.data); +} + +/** Cosine similarity of two unit-norm vectors = dot product. */ +export function cosine(a, b) { + let dot = 0; + const n = Math.min(a.length, b.length); + for (let i = 0; i < n; i++) dot += a[i] * b[i]; + return dot; +} + +/** Rank catalog embeddings against a query vector. Returns top-k [{id, score}]. */ +export function rankBySimilarity(queryVec, embeddings, k = 60) { + const scored = embeddings.map(({ id, vec }) => ({ id, score: cosine(queryVec, vec) })); + scored.sort((a, b) => b.score - a.score); + return scored.slice(0, k); +} diff --git a/app/scripts/lib/grouping.mjs b/app/scripts/lib/grouping.mjs new file mode 100644 index 0000000..62aebc6 --- /dev/null +++ b/app/scripts/lib/grouping.mjs @@ -0,0 +1,131 @@ +// Burst / near-duplicate clustering and file pairing (RAW+JPEG, Live Photos). +// Pure functions over asset descriptors so the logic is unit-testable. +import { hamming } from "./phash.mjs"; + +/** Capture-time gap (seconds) beyond which shots can't be the same burst. */ +const BURST_GAP_SEC = 3; +/** pHash Hamming distance at or below which frames count as near-duplicates. */ +const NEAR_DUP_DISTANCE = 12; + +function toTime(iso) { + const t = iso ? Date.parse(iso) : NaN; + return Number.isFinite(t) ? t / 1000 : null; +} + +/** + * Cluster photos into burst/similar groups. Input: [{ id, capturedAt, hash }] + * where hash is a pHash BigInt (or null when unavailable). Two passes: + * time-adjacency chains shots into candidate runs, then visual similarity + * confirms membership. Returns [{ kind, assetIds }] for groups of 2+. + */ +export function clusterBursts(items) { + const sorted = items + .filter((i) => toTime(i.capturedAt) !== null) + .sort((a, b) => toTime(a.capturedAt) - toTime(b.capturedAt)); + + const groups = []; + let current = []; + + const flush = () => { + if (current.length >= 2) { + groups.push({ kind: "burst", assetIds: current.map((i) => i.id) }); + } + current = []; + }; + + for (const item of sorted) { + if (current.length === 0) { + current.push(item); + continue; + } + const prev = current[current.length - 1]; + const gap = toTime(item.capturedAt) - toTime(prev.capturedAt); + const visuallyClose = + item.hash != null && prev.hash != null + ? hamming(item.hash, prev.hash) <= NEAR_DUP_DISTANCE + : gap <= 1; // no hashes: only sub-second chains count + if (gap <= BURST_GAP_SEC && visuallyClose) { + current.push(item); + } else { + flush(); + current.push(item); + } + } + flush(); + return groups; +} + +/** Strip the extension; uppercase for case-insensitive stem matching. */ +function stem(fileName) { + const dot = fileName.lastIndexOf("."); + return (dot > 0 ? fileName.slice(0, dot) : fileName).toUpperCase(); +} + +const RAW_FORMATS = new Set(["raw"]); +const JPEG_LIKE = new Set(["jpeg", "heic"]); + +/** + * Pair RAW+JPEG twins and Live Photos (HEIC still + MOV with the same stem or + * matching ContentIdentifier). Returns patches: the sibling hides behind the + * primary. Input: [{ id, fileName, format, mediaType, contentId, dirRel }]. + */ +export function pairSiblings(items) { + const patches = []; + const byStem = new Map(); + for (const item of items) { + const key = `${item.dirRel ?? ""}/${stem(item.fileName)}`; + if (!byStem.has(key)) byStem.set(key, []); + byStem.get(key).push(item); + } + const byContentId = new Map(); + for (const item of items) { + if (!item.contentId) continue; + if (!byContentId.has(item.contentId)) byContentId.set(item.contentId, []); + byContentId.get(item.contentId).push(item); + } + + const paired = new Set(); + + const pair = (primary, sibling, role) => { + if (paired.has(sibling.id) || paired.has(primary.id) || primary.id === sibling.id) return; + paired.add(sibling.id); + patches.push({ id: sibling.id, pairRole: role, pairPrimaryId: primary.id }); + }; + + // Live Photos first (ContentIdentifier is authoritative when present). + for (const members of byContentId.values()) { + const still = members.find((m) => m.mediaType === "photo"); + const video = members.find((m) => m.mediaType === "video"); + if (still && video) pair(still, video, "live-video"); + } + + for (const members of byStem.values()) { + if (members.length < 2) continue; + const raw = members.find((m) => RAW_FORMATS.has(m.format)); + const jpegLike = members.find((m) => JPEG_LIKE.has(m.format) && m.mediaType === "photo"); + if (raw && jpegLike) pair(jpegLike, raw, "raw-sibling"); + + const still = members.find((m) => m.mediaType === "photo" && m.format === "heic"); + const video = members.find((m) => m.mediaType === "video" && ["mov", "mp4"].includes(m.format)); + if (still && video) pair(still, video, "live-video"); + } + + return patches; +} + +/** Deterministic best-of-burst: the sharpest, tie-broken by exposure balance. */ +export function pickBest(assets) { + let best = null; + let bestScore = -Infinity; + for (const asset of assets) { + const q = asset.quality ?? {}; + const sharp = q.blurScore ?? 0; + const clipPenalty = ((q.clippedHighlights ?? 0) + (q.clippedShadows ?? 0)) * 200; + const score = sharp - clipPenalty; + if (score > bestScore) { + bestScore = score; + best = asset; + } + } + return best?.id; +} diff --git a/app/scripts/lib/grouping.test.mjs b/app/scripts/lib/grouping.test.mjs new file mode 100644 index 0000000..004c026 --- /dev/null +++ b/app/scripts/lib/grouping.test.mjs @@ -0,0 +1,97 @@ +import { describe, expect, it } from "vitest"; +import { clusterBursts, pairSiblings, pickBest } from "./grouping.mjs"; +import { phash } from "./phash.mjs"; + +function pat(fn) { + const g = new Uint8Array(32 * 32); + for (let y = 0; y < 32; y++) for (let x = 0; x < 32; x++) g[y * 32 + x] = fn(x, y); + return g; +} +const circleAt = (cx) => pat((x, y) => (Math.hypot(x - cx, y - 16) < 8 ? 220 : 30)); + +const t = (sec) => new Date(Date.UTC(2026, 6, 4, 12, 0, sec)).toISOString(); + +describe("clusterBursts", () => { + it("chains time-adjacent, visually similar shots", () => { + const h = phash(circleAt(16)); + const h2 = phash(circleAt(17)); + const items = [ + { id: "a", capturedAt: t(0), hash: h }, + { id: "b", capturedAt: t(1), hash: h2 }, + { id: "c", capturedAt: t(2), hash: h }, + { id: "far", capturedAt: t(60), hash: h }, + ]; + const groups = clusterBursts(items); + expect(groups).toHaveLength(1); + expect(groups[0].assetIds).toEqual(["a", "b", "c"]); + }); + + it("splits visually different shots even when close in time", () => { + const different = pat((x, y) => ((x * 13 + y * 31) % 7 < 3 ? 240 : 10)); + const items = [ + { id: "a", capturedAt: t(0), hash: phash(circleAt(16)) }, + { id: "b", capturedAt: t(1), hash: phash(different) }, + ]; + expect(clusterBursts(items)).toHaveLength(0); + }); + + it("ignores items without timestamps", () => { + expect(clusterBursts([{ id: "a", capturedAt: undefined, hash: null }])).toHaveLength(0); + }); +}); + +describe("pairSiblings", () => { + it("pairs RAW+JPEG twins by stem", () => { + const patches = pairSiblings([ + { id: "j", fileName: "IMG_0001.JPG", format: "jpeg", mediaType: "photo", dirRel: "2026/2026-07-04" }, + { id: "r", fileName: "IMG_0001.CR3", format: "raw", mediaType: "photo", dirRel: "2026/2026-07-04" }, + ]); + expect(patches).toEqual([{ id: "r", pairRole: "raw-sibling", pairPrimaryId: "j" }]); + }); + + it("pairs Live Photos by ContentIdentifier across names", () => { + const patches = pairSiblings([ + { id: "still", fileName: "IMG_2.HEIC", format: "heic", mediaType: "photo", contentId: "uuid-1", dirRel: "d" }, + { id: "vid", fileName: "IMG_2xx.MOV", format: "mov", mediaType: "video", contentId: "uuid-1", dirRel: "d" }, + ]); + expect(patches).toEqual([{ id: "vid", pairRole: "live-video", pairPrimaryId: "still" }]); + }); + + it("pairs Live Photos by stem when no ContentIdentifier", () => { + const patches = pairSiblings([ + { id: "still", fileName: "IMG_3.HEIC", format: "heic", mediaType: "photo", dirRel: "d" }, + { id: "vid", fileName: "IMG_3.MOV", format: "mov", mediaType: "video", dirRel: "d" }, + ]); + expect(patches).toEqual([{ id: "vid", pairRole: "live-video", pairPrimaryId: "still" }]); + }); + + it("does not pair unrelated files", () => { + expect( + pairSiblings([ + { id: "a", fileName: "IMG_1.JPG", format: "jpeg", mediaType: "photo", dirRel: "d" }, + { id: "b", fileName: "IMG_2.JPG", format: "jpeg", mediaType: "photo", dirRel: "d" }, + ]), + ).toEqual([]); + }); +}); + +describe("pickBest", () => { + it("prefers the sharpest frame", () => { + expect( + pickBest([ + { id: "a", quality: { blurScore: 20 } }, + { id: "b", quality: { blurScore: 200 } }, + { id: "c", quality: { blurScore: 90 } }, + ]), + ).toBe("b"); + }); + + it("penalizes clipped exposure", () => { + expect( + pickBest([ + { id: "sharp-blown", quality: { blurScore: 210, clippedHighlights: 0.6 } }, + { id: "slightly-soft", quality: { blurScore: 180, clippedHighlights: 0.01 } }, + ]), + ).toBe("slightly-soft"); + }); +}); diff --git a/app/scripts/lib/llm-util.mjs b/app/scripts/lib/llm-util.mjs new file mode 100644 index 0000000..7ba2e93 --- /dev/null +++ b/app/scripts/lib/llm-util.mjs @@ -0,0 +1,69 @@ +// Repair-not-trust helpers for LLM output (same discipline the video engine +// used): pull JSON out of a chatty response, clamp the fields we understand, +// and let zod do the final say. +import { parseJudgeBatch } from "@keeper/schema"; + +/** Pull the first {...} JSON object out of a model response (handles code fences). */ +export function extractJson(text) { + const fenced = text.match(/```(?:json)?\s*([\s\S]*?)```/i); + const body = fenced ? fenced[1] : text; + const start = body.indexOf("{"); + const end = body.lastIndexOf("}"); + if (start < 0 || end <= start) throw new Error("no JSON object in model output"); + return JSON.parse(body.slice(start, end + 1)); +} + +const SUGGESTIONS = new Set(["keep", "reject", "review"]); +const REASONS = new Set([ + "blurry", "soft-focus", "underexposed", "overexposed", "black-frame", "flat-frame", + "corrupt", "tiny-file", "accidental-clip", "screenshot", "screen-recording", + "duplicate-worse", "burst-not-pick", "eyes-closed", "bad-framing", + "sharpest-of-burst", "well-exposed", "important-moment", "llm-quality", "other", +]); + +/** Clamp common harmless deviations before strict schema validation. */ +export function sanitizeJudge(obj, validAssetIds) { + if (!obj || typeof obj !== "object") return obj; + const valid = new Set(validAssetIds); + const items = Array.isArray(obj.items) ? obj.items : []; + obj.items = items + .filter((item) => item && typeof item === "object" && valid.has(item.assetId)) + .map((item) => { + const out = { ...item }; + if (!SUGGESTIONS.has(out.suggestion)) out.suggestion = "review"; + if (typeof out.confidence !== "number" || !Number.isFinite(out.confidence)) out.confidence = 0.5; + out.confidence = Math.min(1, Math.max(0, out.confidence)); + if (out.reason !== undefined && !REASONS.has(out.reason)) out.reason = "llm-quality"; + if (typeof out.detail === "string") out.detail = out.detail.slice(0, 500); + if (typeof out.caption === "string") out.caption = out.caption.slice(0, 1000); + if (Array.isArray(out.tags)) { + out.tags = out.tags + .filter((t) => typeof t === "string") + .map((t) => t.toLowerCase().slice(0, 64)) + .slice(0, 32); + } else { + delete out.tags; + } + return out; + }) + .slice(0, 64); + + if (obj.bestPicks && typeof obj.bestPicks === "object") { + const picks = {}; + for (const [groupId, assetId] of Object.entries(obj.bestPicks)) { + if (typeof assetId === "string" && valid.has(assetId)) picks[groupId] = assetId; + } + obj.bestPicks = picks; + } else { + obj.bestPicks = {}; + } + return obj; +} + +/** extract -> sanitize -> validate. Throws with the validation errors joined. */ +export function parseJudgeResponse(text, validAssetIds) { + const sanitized = sanitizeJudge(extractJson(text), validAssetIds); + const result = parseJudgeBatch(sanitized); + if (!result.ok) throw new Error(`judge response invalid: ${result.errors?.join("; ")}`); + return result.value; +} diff --git a/app/scripts/lib/llm-util.test.mjs b/app/scripts/lib/llm-util.test.mjs new file mode 100644 index 0000000..972ba66 --- /dev/null +++ b/app/scripts/lib/llm-util.test.mjs @@ -0,0 +1,61 @@ +import { describe, expect, it } from "vitest"; +import { extractJson, parseJudgeResponse, sanitizeJudge } from "./llm-util.mjs"; + +describe("extractJson", () => { + it("parses bare JSON", () => { + expect(extractJson('{"a":1}')).toEqual({ a: 1 }); + }); + + it("parses fenced JSON with chatter", () => { + expect(extractJson('Sure!\n```json\n{"a":1}\n```\nDone.')).toEqual({ a: 1 }); + }); + + it("throws when no object exists", () => { + expect(() => extractJson("no json here")).toThrow(); + }); +}); + +describe("sanitizeJudge", () => { + it("drops items for unknown asset ids", () => { + const out = sanitizeJudge( + { items: [{ assetId: "known", suggestion: "keep", confidence: 0.9 }, { assetId: "hallucinated", suggestion: "keep", confidence: 0.9 }] }, + ["known"], + ); + expect(out.items).toHaveLength(1); + }); + + it("clamps bad enum values and confidences", () => { + const out = sanitizeJudge( + { items: [{ assetId: "a", suggestion: "delete", confidence: 42, reason: "made-up" }] }, + ["a"], + ); + expect(out.items[0].suggestion).toBe("review"); + expect(out.items[0].confidence).toBe(1); + expect(out.items[0].reason).toBe("llm-quality"); + }); + + it("filters bestPicks to known ids", () => { + const out = sanitizeJudge({ items: [], bestPicks: { g1: "a", g2: "nope" } }, ["a"]); + expect(out.bestPicks).toEqual({ g1: "a" }); + }); + + it("lowercases and caps tags", () => { + const out = sanitizeJudge( + { items: [{ assetId: "a", suggestion: "keep", confidence: 0.5, tags: ["OCEAN", 7, "Sunset"] }] }, + ["a"], + ); + expect(out.items[0].tags).toEqual(["ocean", "sunset"]); + }); +}); + +describe("parseJudgeResponse", () => { + it("end-to-end repairs a chatty response", () => { + const text = 'Here you go:\n```json\n{"items":[{"assetId":"a","suggestion":"keep","confidence":0.7,"caption":"beach at dusk"}],"bestPicks":{}}\n```'; + const result = parseJudgeResponse(text, ["a"]); + expect(result.items[0].caption).toBe("beach at dusk"); + }); + + it("throws on unrepairable output", () => { + expect(() => parseJudgeResponse("total nonsense", ["a"])).toThrow(); + }); +}); diff --git a/app/scripts/lib/media.mjs b/app/scripts/lib/media.mjs new file mode 100644 index 0000000..7f04f7c --- /dev/null +++ b/app/scripts/lib/media.mjs @@ -0,0 +1,316 @@ +// Media decoding + derivation helpers built on ffmpeg-static, exiftool, and +// (on macOS) sips. No native image libraries: pixel work happens by piping +// grayscale rawvideo out of ffmpeg into pure-JS math (quality.mjs, phash.mjs). +import { execFile } from "node:child_process"; +import { createRequire } from "node:module"; +import { promisify } from "node:util"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import ffmpegPath from "ffmpeg-static"; +import { exiftool } from "exiftool-vendored"; + +const require = createRequire(import.meta.url); + +const execFileP = promisify(execFile); +const MAX_BUFFER = 256 * 1024 * 1024; + +export const PHOTO_EXT = { + ".jpg": "jpeg", + ".jpeg": "jpeg", + ".png": "png", + ".heic": "heic", + ".heif": "heic", + ".webp": "webp", + ".gif": "gif", + ".tif": "tiff", + ".tiff": "tiff", +}; + +export const RAW_EXT = new Set([ + ".cr2", ".cr3", ".nef", ".nrw", ".arw", ".srf", ".dng", ".orf", ".rw2", ".raf", ".srw", ".pef", ".x3f", +]); + +export const VIDEO_EXT = { + ".mp4": "mp4", + ".m4v": "mp4", + ".mov": "mov", + ".avi": "avi", + ".mkv": "mkv", + ".webm": "webm", + ".mts": "other", + ".m2ts": "other", + ".3gp": "other", +}; + +/** Classify a file by extension. Returns { mediaType, format } or null. */ +export function classifyFile(fileName) { + const ext = path.extname(fileName).toLowerCase(); + if (RAW_EXT.has(ext)) return { mediaType: "photo", format: "raw" }; + if (ext in PHOTO_EXT) return { mediaType: "photo", format: PHOTO_EXT[ext] }; + if (ext in VIDEO_EXT) return { mediaType: "video", format: VIDEO_EXT[ext] }; + return null; +} + +export function isSupported(fileName) { + return classifyFile(fileName) !== null; +} + +async function runFfmpeg(args, { encoding } = {}) { + return execFileP(ffmpegPath, ["-hide_banner", "-loglevel", "error", ...args], { + maxBuffer: MAX_BUFFER, + encoding: encoding ?? "utf8", + }); +} + +const isDarwin = process.platform === "darwin"; + +/** + * Read the metadata Keeper cares about. exiftool handles photos, RAW, and + * video containers alike, and is batched behind a persistent process. + */ +export async function readMetadata(absPath) { + const tags = await exiftool.read(absPath); + const capturedAt = exifDate(tags.DateTimeOriginal) ?? exifDate(tags.CreateDate) ?? exifDate(tags.MediaCreateDate); + const gps = + typeof tags.GPSLatitude === "number" && typeof tags.GPSLongitude === "number" + ? { lat: tags.GPSLatitude, lon: tags.GPSLongitude } + : undefined; + return { + capturedAt, + width: numberOr(tags.ImageWidth, undefined), + height: numberOr(tags.ImageHeight, undefined), + durationSec: parseDuration(tags.Duration), + exif: { + make: strOr(tags.Make), + model: strOr(tags.Model), + lens: strOr(tags.LensModel ?? tags.LensID), + iso: numberOr(tags.ISO, undefined), + exposureSec: parseExposure(tags.ExposureTime), + fNumber: numberOr(tags.FNumber, undefined), + focalMm: parseFocal(tags.FocalLength), + gps, + contentId: strOr(tags.ContentIdentifier ?? tags.MediaGroupUUID), + }, + }; +} + +function exifDate(value) { + if (!value) return undefined; + if (typeof value === "string") { + // exiftool string form: "2026:07:04 18:23:11" + const m = value.match(/^(\d{4}):(\d{2}):(\d{2})[ T](.*)$/); + const iso = m ? `${m[1]}-${m[2]}-${m[3]}T${m[4]}` : value; + const t = Date.parse(iso); + return Number.isFinite(t) ? new Date(t).toISOString() : undefined; + } + if (typeof value === "object" && typeof value.toISOString === "function") { + try { + return new Date(value.toISOString()).toISOString(); + } catch { + return undefined; + } + } + return undefined; +} + +function strOr(v) { + return typeof v === "string" && v.length > 0 ? v.slice(0, 250) : undefined; +} + +function numberOr(v, fallback) { + return typeof v === "number" && Number.isFinite(v) ? v : fallback; +} + +function parseExposure(v) { + if (typeof v === "number" && Number.isFinite(v)) return v; + if (typeof v === "string") { + const m = v.match(/^1\/(\d+(?:\.\d+)?)$/); + if (m) return 1 / Number(m[1]); + const n = Number(v); + if (Number.isFinite(n)) return n; + } + return undefined; +} + +function parseFocal(v) { + if (typeof v === "number" && Number.isFinite(v)) return v; + if (typeof v === "string") { + const n = Number.parseFloat(v); + if (Number.isFinite(n)) return n; + } + return undefined; +} + +function parseDuration(v) { + if (typeof v === "number" && Number.isFinite(v)) return v; + if (typeof v === "string") { + // "0:00:12" or "12.4 s" + const hms = v.match(/^(\d+):(\d{2}):(\d{2}(?:\.\d+)?)$/); + if (hms) return Number(hms[1]) * 3600 + Number(hms[2]) * 60 + Number(hms[3]); + const secs = v.match(/^([\d.]+)\s*s$/); + if (secs) return Number(secs[1]); + } + return undefined; +} + +/** Shut down the persistent exiftool process (call at script exit). */ +export async function closeMetadataReader() { + try { + await exiftool.end(); + } catch { + // already gone + } +} + +/** + * Produce a JPEG version of any still, for thumbnailing and CV work: + * - jpeg/png/webp/tiff/gif: ffmpeg decodes them directly (returns input path). + * - RAW: extract the embedded preview via exiftool; fall back to sips (macOS). + * - HEIC: sips on macOS; try ffmpeg elsewhere (newer builds decode HEIC). + * Returns an absolute path to a decodable still, or null when hopeless. + */ +export async function ensureDecodableStill(absPath, format, tmpDir) { + if (format === "jpeg" || format === "png" || format === "webp" || format === "tiff" || format === "gif") { + return absPath; + } + const tmpOut = path.join(tmpDir, `${path.basename(absPath)}.decoded.jpg`); + + if (format === "raw") { + for (const tag of ["JpgFromRaw", "PreviewImage", "OtherImage", "ThumbnailImage"]) { + try { + const { stdout } = await execFileP( + exiftoolBin(), + ["-b", `-${tag}`, absPath], + { maxBuffer: MAX_BUFFER, encoding: "buffer" }, + ); + if (stdout && stdout.length > 4096) { + fs.writeFileSync(tmpOut, stdout); + return tmpOut; + } + } catch { + // tag missing — try the next + } + } + if (isDarwin) return sipsConvert(absPath, tmpOut); + return null; + } + + if (format === "heic") { + if (isDarwin) { + const out = await sipsConvert(absPath, tmpOut); + if (out) return out; + } + try { + await runFfmpeg(["-i", absPath, "-frames:v", "1", "-y", tmpOut]); + return fs.existsSync(tmpOut) ? tmpOut : null; + } catch { + return null; + } + } + + return null; +} + +let cachedExiftoolBin = null; +function exiftoolBin() { + if (!cachedExiftoolBin) { + // exiftool-vendored exposes the batched API; for one-shot binary output we + // call the vendored perl script directly. + cachedExiftoolBin = require.resolve("exiftool-vendored.pl/bin/exiftool"); + } + return cachedExiftoolBin; +} + +async function sipsConvert(absPath, outPath, maxDim) { + try { + const args = ["-s", "format", "jpeg", absPath, "--out", outPath]; + if (maxDim) args.splice(0, 0, "-Z", String(maxDim)); + await execFileP("/usr/bin/sips", args, { maxBuffer: MAX_BUFFER }); + return fs.existsSync(outPath) ? outPath : null; + } catch { + return null; + } +} + +/** Scale a decodable still down to a JPEG thumb (maxDim on the long edge). */ +export async function makeThumb(decodableStill, outPath, maxDim = 512) { + await runFfmpeg([ + "-i", decodableStill, + "-vf", `scale='min(${maxDim},iw)':'min(${maxDim},ih)':force_original_aspect_ratio=decrease`, + "-frames:v", "1", + "-q:v", "4", + "-y", outPath, + ]); + return outPath; +} + +/** Poster frame for a video (at ~1s in, clamped to duration). */ +export async function makePoster(absVideo, outPath, durationSec, maxDim = 768) { + const at = Math.min(1, Math.max(0, (durationSec ?? 2) / 3)).toFixed(2); + await runFfmpeg([ + "-ss", at, + "-i", absVideo, + "-frames:v", "1", + "-vf", `scale='min(${maxDim},iw)':'min(${maxDim},ih)':force_original_aspect_ratio=decrease`, + "-q:v", "4", + "-y", outPath, + ]); + return outPath; +} + +/** Horizontal strip of N frames for hover-scrubbing a video. */ +export async function makeScrubStrip(absVideo, outPath, durationSec, frames = 10) { + if (!durationSec || durationSec <= 0) return null; + const fps = Math.max(0.05, frames / durationSec); + try { + await runFfmpeg([ + "-i", absVideo, + "-vf", `fps=${fps.toFixed(4)},scale=180:-2,tile=${frames}x1`, + "-frames:v", "1", + "-q:v", "5", + "-y", outPath, + ]); + return fs.existsSync(outPath) ? outPath : null; + } catch { + return null; + } +} + +/** + * Extract a size x size grayscale buffer from any ffmpeg-decodable input + * (thumb JPEG or video frame). The fixed square keeps scores comparable. + */ +export async function extractGray(absPath, size = 128) { + const { stdout } = await execFileP( + ffmpegPath, + [ + "-hide_banner", "-loglevel", "error", + "-i", absPath, + "-vf", `scale=${size}:${size}`, + "-frames:v", "1", + "-f", "rawvideo", + "-pix_fmt", "gray", + "-", + ], + { maxBuffer: MAX_BUFFER, encoding: "buffer" }, + ); + if (stdout.length < size * size) throw new Error(`gray extraction returned ${stdout.length} bytes`); + return new Uint8Array(stdout.buffer, stdout.byteOffset, size * size); +} + +/** Fallback duration probe: parse ffmpeg's stderr banner. */ +export async function probeDurationSec(absPath) { + try { + await execFileP(ffmpegPath, ["-hide_banner", "-i", absPath], { maxBuffer: 4 * 1024 * 1024 }); + } catch (err) { + const stderr = String(err.stderr ?? ""); + const m = stderr.match(/Duration:\s*(\d+):(\d{2}):(\d{2}(?:\.\d+)?)/); + if (m) return Number(m[1]) * 3600 + Number(m[2]) * 60 + Number(m[3]); + } + return undefined; +} + +export function makeTmpDir(prefix = "keeper-") { + return fs.mkdtempSync(path.join(os.tmpdir(), prefix)); +} diff --git a/app/scripts/lib/paths.mjs b/app/scripts/lib/paths.mjs new file mode 100644 index 0000000..0d6cd07 --- /dev/null +++ b/app/scripts/lib/paths.mjs @@ -0,0 +1,82 @@ +// Library location resolution shared by every script, the catalog service, +// and the CLI tools. The Electron main process resolves the same way and +// re-exports KEEPER_LIBRARY_DIR into the environment of spawned scripts, so +// everyone agrees on one home. +import os from "node:os"; +import path from "node:path"; +import fs from "node:fs"; + +/** Resolve the library home: env override, else ~/Pictures/Keeper. */ +export function resolveHome(explicit) { + return ( + explicit || + process.env.KEEPER_LIBRARY_DIR || + process.env.KEEPER_HOME || + path.join(os.homedir(), "Pictures", "Keeper") + ); +} + +/** Where originals live: /library/YYYY/YYYY-MM-DD/. */ +export function libraryRoot(home) { + return path.join(home, "library"); +} + +/** Keeper's own data: catalog, thumbs, previews, models. */ +export function keeperDir(home) { + return path.join(home, ".keeper"); +} + +export function catalogPath(home) { + return path.join(keeperDir(home), "catalog.db"); +} + +export function tastePath(home) { + return path.join(home, "taste.json"); +} + +/** Touched after CLI/pipeline writes so the app knows to refresh. */ +export function stampPath(home) { + return path.join(keeperDir(home), ".stamp"); +} + +export function thumbsDir(home) { + return path.join(keeperDir(home), "thumbs"); +} + +export function previewsDir(home) { + return path.join(keeperDir(home), "previews"); +} + +export function modelsDir(home) { + return path.join(keeperDir(home), "models"); +} + +/** Create the whole on-disk layout (idempotent). */ +export function ensureLayout(home) { + for (const dir of [home, libraryRoot(home), keeperDir(home), thumbsDir(home), previewsDir(home), modelsDir(home)]) { + fs.mkdirSync(dir, { recursive: true }); + } + return home; +} + +/** + * Join + confine: the resolved path must stay inside root. Guards every path + * that came from a catalog row or an IPC/CLI argument. + */ +export function safeJoin(root, ...rel) { + const resolved = path.normalize(path.join(root, ...rel)); + const base = path.resolve(root); + if (resolved !== base && !resolved.startsWith(base + path.sep)) { + throw new Error(`path escapes ${base}: ${rel.join("/")}`); + } + return resolved; +} + +export function touchStamp(home) { + try { + fs.mkdirSync(keeperDir(home), { recursive: true }); + fs.writeFileSync(stampPath(home), String(Date.now())); + } catch { + // stamp is best-effort; a missed refresh is harmless + } +} diff --git a/app/scripts/lib/phash.mjs b/app/scripts/lib/phash.mjs new file mode 100644 index 0000000..d8fd16d --- /dev/null +++ b/app/scripts/lib/phash.mjs @@ -0,0 +1,105 @@ +// 64-bit perceptual hash (pHash): 32x32 grayscale -> 2D DCT -> sign of the +// top-left 8x8 AC coefficients vs their median. Near-duplicate frames land +// within a small Hamming distance of each other. Pure JS, no dependencies. + +const SIZE = 32; +const LOW = 8; + +/** Precomputed DCT-II cosine basis for a 32-point transform. */ +const COS = (() => { + const table = []; + for (let u = 0; u < LOW; u++) { + const row = new Float64Array(SIZE); + for (let x = 0; x < SIZE; x++) { + row[x] = Math.cos(((2 * x + 1) * u * Math.PI) / (2 * SIZE)); + } + table.push(row); + } + return table; +})(); + +/** + * Compute the pHash of a 32x32 grayscale buffer (1024 bytes). + * Returns a BigInt with 64 significant bits. + */ +export function phash(gray32) { + if (gray32.length !== SIZE * SIZE) throw new Error(`phash needs ${SIZE * SIZE} pixels, got ${gray32.length}`); + + // Row-column separable 2D DCT, keeping only the LOW x LOW corner. + const rows = []; + for (let y = 0; y < SIZE; y++) { + const row = new Float64Array(LOW); + for (let u = 0; u < LOW; u++) { + let acc = 0; + const basis = COS[u]; + for (let x = 0; x < SIZE; x++) acc += gray32[y * SIZE + x] * basis[x]; + row[u] = acc; + } + rows.push(row); + } + const coeffs = new Float64Array(LOW * LOW); + for (let v = 0; v < LOW; v++) { + const basis = COS[v]; + for (let u = 0; u < LOW; u++) { + let acc = 0; + for (let y = 0; y < SIZE; y++) acc += rows[y][u] * basis[y]; + coeffs[v * LOW + u] = acc; + } + } + + // Median of the AC coefficients (skip the DC term at [0]). + const ac = Array.from(coeffs.slice(1)).sort((a, b) => a - b); + const median = ac[Math.floor(ac.length / 2)]; + + let hash = 0n; + for (let i = 1; i < coeffs.length; i++) { + hash <<= 1n; + if (coeffs[i] > median) hash |= 1n; + } + return hash; +} + +/** Hamming distance between two pHashes (0 = identical structure). */ +export function hamming(a, b) { + let x = a ^ b; + let count = 0; + while (x > 0n) { + count += Number(x & 1n); + x >>= 1n; + } + return count; +} + +/** Box-filter a square grayscale buffer down to target x target. */ +export function downsampleGray(gray, size, target) { + if (size === target) return gray; + const out = new Uint8Array(target * target); + const ratio = size / target; + for (let y = 0; y < target; y++) { + const y0 = Math.floor(y * ratio); + const y1 = Math.min(size, Math.ceil((y + 1) * ratio)); + for (let x = 0; x < target; x++) { + const x0 = Math.floor(x * ratio); + const x1 = Math.min(size, Math.ceil((x + 1) * ratio)); + let sum = 0; + let n = 0; + for (let yy = y0; yy < y1; yy++) { + for (let xx = x0; xx < x1; xx++) { + sum += gray[yy * size + xx]; + n++; + } + } + out[y * target + x] = Math.round(sum / Math.max(1, n)); + } + } + return out; +} + +/** Serialize for storage. */ +export function phashToHex(hash) { + return hash.toString(16).padStart(16, "0"); +} + +export function phashFromHex(hex) { + return BigInt(`0x${hex}`); +} diff --git a/app/scripts/lib/phash.test.mjs b/app/scripts/lib/phash.test.mjs new file mode 100644 index 0000000..840220b --- /dev/null +++ b/app/scripts/lib/phash.test.mjs @@ -0,0 +1,50 @@ +import { describe, expect, it } from "vitest"; +import { downsampleGray, hamming, phash, phashFromHex, phashToHex } from "./phash.mjs"; + +function pattern(size, fn) { + const g = new Uint8Array(size * size); + for (let y = 0; y < size; y++) for (let x = 0; x < size; x++) g[y * size + x] = fn(x, y); + return g; +} + +const circle = (cx, cy, r) => (x, y) => (Math.hypot(x - cx, y - cy) < r ? 220 : 30); + +describe("phash", () => { + it("is identical for identical images", () => { + const a = phash(pattern(32, circle(16, 16, 8))); + const b = phash(pattern(32, circle(16, 16, 8))); + expect(hamming(a, b)).toBe(0); + }); + + it("is close for slightly shifted content", () => { + const a = phash(pattern(32, circle(16, 16, 8))); + const b = phash(pattern(32, circle(17, 16, 8))); + expect(hamming(a, b)).toBeLessThanOrEqual(10); + }); + + it("is far for structurally different content", () => { + const a = phash(pattern(32, circle(16, 16, 8))); + const b = phash(pattern(32, (x, y) => ((x * 13 + y * 31) % 7 < 3 ? 240 : 10))); + expect(hamming(a, b)).toBeGreaterThan(16); + }); + + it("round-trips through hex", () => { + const a = phash(pattern(32, circle(10, 20, 6))); + expect(phashFromHex(phashToHex(a))).toBe(a); + }); + + it("rejects wrong-size buffers", () => { + expect(() => phash(new Uint8Array(100))).toThrow(); + }); +}); + +describe("downsampleGray", () => { + it("preserves mean brightness", () => { + const src = pattern(128, (x) => (x < 64 ? 0 : 255)); + const out = downsampleGray(src, 128, 32); + expect(out.length).toBe(32 * 32); + const mean = out.reduce((s, v) => s + v, 0) / out.length; + expect(mean).toBeGreaterThan(115); + expect(mean).toBeLessThan(140); + }); +}); diff --git a/app/scripts/lib/pipeline.mjs b/app/scripts/lib/pipeline.mjs new file mode 100644 index 0000000..68ea548 --- /dev/null +++ b/app/scripts/lib/pipeline.mjs @@ -0,0 +1,249 @@ +// The derivation pipeline stages shared by import.mjs (fresh imports) and +// reprocess.mjs (resume / re-run). Each stage is checkpointed per asset in +// stages_json, so a crash or quit resumes where it left off. +import fs from "node:fs"; +import path from "node:path"; +import { libraryRoot, previewsDir, safeJoin, thumbsDir } from "./paths.mjs"; +import { + ensureDecodableStill, + extractGray, + makePoster, + makeScrubStrip, + makeThumb, + makeTmpDir, +} from "./media.mjs"; +import { cvVerdict, exposureStats, laplacianVariance } from "./quality.mjs"; +import { clusterBursts, pairSiblings, pickBest } from "./grouping.mjs"; +import { downsampleGray, phash } from "./phash.mjs"; +import { embedImage, EMBED_MODEL_ID } from "./embeddings.mjs"; +import { readTaste, tunedThresholds } from "./taste.mjs"; + +/** Run tasks over items with bounded concurrency; onDone fires per item. */ +export async function pool(items, worker, { concurrency = 4, onDone } = {}) { + const queue = [...items]; + let done = 0; + const errors = []; + const runners = Array.from({ length: Math.min(concurrency, queue.length) }, async () => { + while (queue.length > 0) { + const item = queue.shift(); + try { + await worker(item); + } catch (err) { + errors.push({ item, error: err }); + console.error(`ERROR ${err?.message ?? err}`); + } + done++; + onDone?.(done); + } + }); + await Promise.all(runners); + return errors; +} + +function absOriginal(home, asset) { + return safeJoin(home, asset.relPath); +} + +/** + * Stage: derive — thumbs for photos, poster + scrub strip for videos. + * Derivatives land in .keeper/{thumbs,previews}/.jpg. + */ +export async function stageDerive(catalog, assets, { onProgress } = {}) { + const home = catalog.home; + const tmpDir = makeTmpDir("keeper-derive-"); + fs.mkdirSync(thumbsDir(home), { recursive: true }); + fs.mkdirSync(previewsDir(home), { recursive: true }); + + await pool( + assets, + async (asset) => { + const src = absOriginal(home, asset); + if (!fs.existsSync(src)) { + catalog.patchAsset(asset.id, { offline: true }); + return; + } + if (asset.mediaType === "photo") { + const decodable = await ensureDecodableStill(src, asset.format, tmpDir); + if (!decodable) throw new Error(`cannot decode ${asset.fileName}`); + const thumbAbs = path.join(thumbsDir(home), `${asset.id}.jpg`); + await makeThumb(decodable, thumbAbs, 512); + const previewAbs = path.join(previewsDir(home), `${asset.id}.jpg`); + await makeThumb(decodable, previewAbs, 2048); + catalog.patchAsset(asset.id, { + thumbRel: path.relative(home, thumbAbs), + previewRel: path.relative(home, previewAbs), + }); + } else { + const posterAbs = path.join(thumbsDir(home), `${asset.id}.jpg`); + await makePoster(src, posterAbs, asset.durationSec); + const scrubAbs = path.join(previewsDir(home), `${asset.id}.scrub.jpg`); + const scrub = await makeScrubStrip(src, scrubAbs, asset.durationSec); + catalog.patchAsset(asset.id, { + thumbRel: path.relative(home, posterAbs), + posterRel: path.relative(home, posterAbs), + ...(scrub ? { scrubRel: path.relative(home, scrub) } : {}), + }); + } + catalog.markStage(asset.id, "derive"); + }, + { concurrency: 4, onDone: (n) => onProgress?.(n / assets.length) }, + ); + + fs.rmSync(tmpDir, { recursive: true, force: true }); +} + +/** + * Stage: cv — grayscale metrics + pHash from the thumb, then the + * deterministic verdict using taste-tuned thresholds. + * Returns { hashes: Map } for the grouping stage. + */ +export async function stageCv(catalog, assets, { onProgress } = {}) { + const home = catalog.home; + const thresholds = tunedThresholds(readTaste(home)); + const hashes = new Map(); + + await pool( + assets, + async (asset) => { + const fresh = catalog.getAsset(asset.id); + if (!fresh?.thumbRel) return; + const thumbAbs = safeJoin(home, fresh.thumbRel); + if (!fs.existsSync(thumbAbs)) return; + + const gray = await extractGray(thumbAbs, 128); + const quality = { + blurScore: Math.round(laplacianVariance(gray, 128, 128) * 100) / 100, + ...exposureStats(gray), + }; + hashes.set(asset.id, phash(downsampleGray(gray, 128, 32))); + + const ai = cvVerdict( + { + quality, + mediaType: fresh.mediaType, + durationSec: fresh.durationSec, + format: fresh.format, + exif: fresh.exif, + fileName: fresh.fileName, + byteSize: fresh.byteSize, + }, + thresholds, + ); + catalog.patchAsset(asset.id, { quality }); + catalog.setAiVerdict(asset.id, ai); + catalog.markStage(asset.id, "cv"); + }, + { concurrency: 4, onDone: (n) => onProgress?.(n / assets.length) }, + ); + + return { hashes }; +} + +/** + * Stage: group — pair RAW+JPEG / Live Photos, then cluster bursts and pick a + * best frame per group. Non-picks get a "duplicate-worse" review suggestion + * pointing at the pick (never a sure reject — bursts go to the needs-eye queue). + */ +export function stageGroup(catalog, assets, hashes) { + const fresh = catalog.getAssets(assets.map((a) => a.id)); + + const pairPatches = pairSiblings( + fresh.map((a) => ({ + id: a.id, + fileName: a.fileName, + format: a.format, + mediaType: a.mediaType, + contentId: a.exif.contentId, + dirRel: path.dirname(a.relPath), + })), + ); + for (const patch of pairPatches) { + catalog.setPairing(patch.id, patch.pairRole, patch.pairPrimaryId); + } + const hidden = new Set(pairPatches.map((p) => p.id)); + + const photos = fresh.filter((a) => a.mediaType === "photo" && !hidden.has(a.id)); + const clusters = clusterBursts( + photos.map((a) => ({ id: a.id, capturedAt: a.capturedAt, hash: hashes.get(a.id) ?? null })), + ); + + for (const cluster of clusters) { + const members = catalog.getAssets(cluster.assetIds); + const bestId = pickBest(members); + const groupId = `grp-${cluster.assetIds[0]}`; + catalog.upsertGroup({ + id: groupId, + kind: cluster.kind, + assetIds: cluster.assetIds, + bestPickId: bestId, + pickSource: "cv", + }); + for (const member of members) { + if (member.id === bestId) { + const ai = member.ai ?? { suggestion: "keep", confidence: 0.6, reasons: [], source: "cv" }; + catalog.setAiVerdict(member.id, { + ...ai, + suggestion: ai.suggestion === "reject" ? ai.suggestion : "keep", + reasons: [{ code: "sharpest-of-burst" }, ...ai.reasons.slice(0, 4)], + }); + } else if (member.ai?.suggestion !== "reject") { + catalog.setAiVerdict(member.id, { + suggestion: "reject", + confidence: 0.7, + reasons: [ + { code: "duplicate-worse", detail: "burst frame", refAssetId: bestId }, + ...(member.ai?.reasons.slice(0, 3) ?? []), + ], + source: "cv", + at: new Date().toISOString(), + }); + } + } + } + + for (const asset of fresh) catalog.markStage(asset.id, "group"); + return { groups: clusters.length, paired: pairPatches.length }; +} + +/** + * Stage: embed — CLIP vectors from thumbs/posters. Soft-fails when the model + * can't load (offline, first run without network): search then degrades to + * metadata matching until reprocess runs. + */ +export async function stageEmbed(catalog, assets, { onProgress } = {}) { + const home = catalog.home; + let failed = false; + + // Probe the model once; if it can't load, skip the stage quietly. + try { + const probe = assets.find((a) => catalog.getAsset(a.id)?.thumbRel); + if (!probe) return { embedded: 0, skipped: assets.length }; + const rec = catalog.getAsset(probe.id); + const vec = await embedImage(home, safeJoin(home, rec.thumbRel)); + catalog.setEmbedding(probe.id, vec, EMBED_MODEL_ID); + catalog.markStage(probe.id, "embed"); + } catch (err) { + console.error(`ERROR embed model unavailable: ${err?.message ?? err}`); + return { embedded: 0, skipped: assets.length }; + } + + let embedded = 1; + await pool( + assets, + async (asset) => { + const fresh = catalog.getAsset(asset.id); + if (!fresh?.thumbRel || fresh.stages.embed) return; + try { + const vec = await embedImage(home, safeJoin(home, fresh.thumbRel)); + catalog.setEmbedding(asset.id, vec, EMBED_MODEL_ID); + catalog.markStage(asset.id, "embed"); + embedded++; + } catch (err) { + failed = true; + throw err; + } + }, + { concurrency: 1, onDone: (n) => onProgress?.(n / assets.length) }, + ); + return { embedded, skipped: failed ? assets.length - embedded : 0 }; +} diff --git a/app/scripts/lib/quality.mjs b/app/scripts/lib/quality.mjs new file mode 100644 index 0000000..35e5984 --- /dev/null +++ b/app/scripts/lib/quality.mjs @@ -0,0 +1,159 @@ +// Technical-quality measurement and the deterministic CV verdict. All pure +// functions over grayscale pixel buffers (extracted via ffmpeg) so they are +// unit-testable without media fixtures. + +/** + * Variance of the 3x3 Laplacian over a grayscale image. The classic sharpness + * proxy: in-focus images have strong second-derivative energy, blurry ones + * don't. Computed on a normalized-size thumb so scores are comparable. + */ +export function laplacianVariance(gray, width, height) { + if (width < 3 || height < 3) return 0; + let sum = 0; + let sumSq = 0; + const n = (width - 2) * (height - 2); + for (let y = 1; y < height - 1; y++) { + const row = y * width; + for (let x = 1; x < width - 1; x++) { + const i = row + x; + const lap = + 4 * gray[i] - gray[i - 1] - gray[i + 1] - gray[i - width] - gray[i + width]; + sum += lap; + sumSq += lap * lap; + } + } + const mean = sum / n; + return sumSq / n - mean * mean; +} + +/** Mean/σ/clipping stats over a grayscale buffer. */ +export function exposureStats(gray) { + let sum = 0; + let sumSq = 0; + let high = 0; + let low = 0; + const n = gray.length; + for (let i = 0; i < n; i++) { + const v = gray[i]; + sum += v; + sumSq += v * v; + if (v >= 250) high++; + if (v <= 5) low++; + } + const mean = sum / n; + const variance = Math.max(0, sumSq / n - mean * mean); + return { + meanLuma: round2(mean), + lumaStdDev: round2(Math.sqrt(variance)), + clippedHighlights: round4(high / n), + clippedShadows: round4(low / n), + }; +} + +const round2 = (v) => Math.round(v * 100) / 100; +const round4 = (v) => Math.round(v * 10_000) / 10_000; + +/** Screenshots: PNGs at exact device resolutions with no camera EXIF. */ +export function isLikelyScreenshot({ format, exif, fileName }) { + if (exif?.make || exif?.model) return false; + if (/screenshot|screen shot|capture d.ecran/i.test(fileName)) return true; + return format === "png"; +} + +export function isLikelyScreenRecording({ mediaType, exif, fileName }) { + if (mediaType !== "video") return false; + if (/screen.?record|simulator/i.test(fileName)) return true; + return false; +} + +/** + * The deterministic CV verdict: quality measurements + taste thresholds in, + * an AI suggestion with reasons and calibrated confidence out. Conservative + * by design — only unambiguous junk gets a high-confidence reject; everything + * borderline routes to review for the human (or the LLM judge) to decide. + */ +export function cvVerdict({ quality, mediaType, durationSec, format, exif, fileName, byteSize }, thresholds) { + const reasons = []; + let suggestion = "keep"; + let confidence = 0.55; + + const q = quality ?? {}; + + if (byteSize !== undefined && byteSize < 1024) { + return verdict("reject", 0.97, [{ code: "corrupt", detail: "file is under 1KB" }]); + } + + // Black/flat frames: nearly no luma variation. + if (q.lumaStdDev !== undefined && q.lumaStdDev < 2.5) { + const dark = (q.meanLuma ?? 128) < 16; + return verdict("reject", 0.95, [ + { code: dark ? "black-frame" : "flat-frame", detail: `σ=${q.lumaStdDev}` }, + ]); + } + + // Accidental video: sub-second pocket clips. + if (mediaType === "video" && durationSec !== undefined && durationSec <= thresholds.accidentalClipSec) { + return verdict("reject", 0.9, [ + { code: "accidental-clip", detail: `${durationSec.toFixed(1)}s long` }, + ]); + } + + if (isLikelyScreenshot({ format, exif, fileName })) { + reasons.push({ code: "screenshot" }); + suggestion = "review"; + confidence = 0.6; + } + if (isLikelyScreenRecording({ mediaType, exif, fileName })) { + reasons.push({ code: "screen-recording" }); + suggestion = "review"; + confidence = 0.6; + } + + // Blur (photos only — video sharpness varies frame to frame). + if (mediaType === "photo" && q.blurScore !== undefined) { + if (q.blurScore < thresholds.blurReject) { + reasons.push({ code: "blurry", detail: `sharpness ${q.blurScore.toFixed(1)}` }); + return verdict("reject", 0.88, reasons); + } + if (q.blurScore < thresholds.blurReview) { + reasons.push({ code: "soft-focus", detail: `sharpness ${q.blurScore.toFixed(1)}` }); + suggestion = "review"; + confidence = 0.55; + } + } + + // Exposure. + if (q.clippedShadows !== undefined && q.clippedShadows > thresholds.clipReject && (q.meanLuma ?? 128) < 40) { + reasons.push({ code: "underexposed", detail: `${Math.round(q.clippedShadows * 100)}% crushed` }); + return verdict("reject", 0.85, reasons); + } + if (q.clippedHighlights !== undefined && q.clippedHighlights > thresholds.clipReject) { + reasons.push({ code: "overexposed", detail: `${Math.round(q.clippedHighlights * 100)}% blown` }); + return verdict("reject", 0.85, reasons); + } + if ( + (q.clippedShadows !== undefined && q.clippedShadows > thresholds.clipReview && (q.meanLuma ?? 128) < 60) || + (q.clippedHighlights !== undefined && q.clippedHighlights > thresholds.clipReview) + ) { + const under = (q.clippedShadows ?? 0) > (q.clippedHighlights ?? 0); + reasons.push({ code: under ? "underexposed" : "overexposed" }); + suggestion = "review"; + confidence = Math.min(confidence, 0.55); + } + + if (reasons.length === 0) { + reasons.push({ code: "well-exposed" }); + confidence = 0.7; + } + return verdict(suggestion, confidence, reasons); +} + +function verdict(suggestion, confidence, reasons) { + return { + suggestion, + confidence, + reasons, + source: "cv", + at: new Date().toISOString(), + }; +} diff --git a/app/scripts/lib/quality.test.mjs b/app/scripts/lib/quality.test.mjs new file mode 100644 index 0000000..4c07bd7 --- /dev/null +++ b/app/scripts/lib/quality.test.mjs @@ -0,0 +1,119 @@ +import { describe, expect, it } from "vitest"; +import { cvVerdict, exposureStats, laplacianVariance } from "./quality.mjs"; +import { parseTasteProfile } from "@keeper/schema"; + +const T = parseTasteProfile({}).thresholds; + +function flatGray(size, value) { + return new Uint8Array(size * size).fill(value); +} + +/** Checkerboard = maximal high-frequency energy = very sharp. */ +function checkerboard(size) { + const g = new Uint8Array(size * size); + for (let y = 0; y < size; y++) { + for (let x = 0; x < size; x++) { + g[y * size + x] = (x + y) % 2 === 0 ? 255 : 0; + } + } + return g; +} + +/** Smooth horizontal gradient = almost no second-derivative energy. */ +function gradient(size) { + const g = new Uint8Array(size * size); + for (let y = 0; y < size; y++) { + for (let x = 0; x < size; x++) { + g[y * size + x] = Math.round((x / (size - 1)) * 255); + } + } + return g; +} + +describe("laplacianVariance", () => { + it("scores sharp content far above smooth content", () => { + const sharp = laplacianVariance(checkerboard(64), 64, 64); + const smooth = laplacianVariance(gradient(64), 64, 64); + expect(sharp).toBeGreaterThan(10_000); + expect(smooth).toBeLessThan(50); + expect(sharp).toBeGreaterThan(smooth * 100); + }); + + it("is zero for flat frames", () => { + expect(laplacianVariance(flatGray(64, 128), 64, 64)).toBe(0); + }); +}); + +describe("exposureStats", () => { + it("measures clipping fractions", () => { + const g = new Uint8Array(100); + g.fill(255, 0, 30); // 30% blown + g.fill(0, 30, 40); // 10% crushed + g.fill(128, 40); + const stats = exposureStats(g); + expect(stats.clippedHighlights).toBeCloseTo(0.3, 5); + expect(stats.clippedShadows).toBeCloseTo(0.1, 5); + expect(stats.meanLuma).toBeGreaterThan(80); + }); + + it("flat frames have near-zero deviation", () => { + expect(exposureStats(flatGray(32, 7)).lumaStdDev).toBe(0); + }); +}); + +describe("cvVerdict", () => { + const base = { mediaType: "photo", format: "jpeg", exif: { make: "Canon" }, fileName: "IMG_1.jpg", byteSize: 5_000_000 }; + + it("hard-rejects black frames with high confidence", () => { + const v = cvVerdict({ ...base, quality: { meanLuma: 2, lumaStdDev: 0.5 } }, T); + expect(v.suggestion).toBe("reject"); + expect(v.confidence).toBeGreaterThanOrEqual(0.9); + expect(v.reasons[0].code).toBe("black-frame"); + }); + + it("rejects blurry photos below the taste threshold", () => { + const v = cvVerdict({ ...base, quality: { blurScore: T.blurReject - 1, lumaStdDev: 40, meanLuma: 120 } }, T); + expect(v.suggestion).toBe("reject"); + expect(v.reasons.map((r) => r.code)).toContain("blurry"); + }); + + it("routes soft-but-not-terrible photos to review", () => { + const v = cvVerdict( + { ...base, quality: { blurScore: (T.blurReject + T.blurReview) / 2, lumaStdDev: 40, meanLuma: 120 } }, + T, + ); + expect(v.suggestion).toBe("review"); + }); + + it("keeps sharp, well-exposed photos", () => { + const v = cvVerdict({ ...base, quality: { blurScore: 500, lumaStdDev: 50, meanLuma: 120, clippedHighlights: 0.01, clippedShadows: 0.01 } }, T); + expect(v.suggestion).toBe("keep"); + expect(v.reasons[0].code).toBe("well-exposed"); + }); + + it("suggests rejecting sub-second accidental videos", () => { + const v = cvVerdict( + { mediaType: "video", format: "mp4", exif: {}, fileName: "clip.mp4", byteSize: 1e6, durationSec: 0.6, quality: { lumaStdDev: 40, meanLuma: 100 } }, + T, + ); + expect(v.suggestion).toBe("reject"); + expect(v.reasons[0].code).toBe("accidental-clip"); + }); + + it("never blur-rejects videos", () => { + const v = cvVerdict( + { mediaType: "video", format: "mp4", exif: {}, fileName: "clip.mp4", byteSize: 1e8, durationSec: 12, quality: { blurScore: 1, lumaStdDev: 40, meanLuma: 100 } }, + T, + ); + expect(v.suggestion).toBe("keep"); + }); + + it("flags screenshots for review, not rejection", () => { + const v = cvVerdict( + { mediaType: "photo", format: "png", exif: {}, fileName: "Screenshot 2026-07-01.png", byteSize: 1e6, quality: { blurScore: 900, lumaStdDev: 60, meanLuma: 128 } }, + T, + ); + expect(v.suggestion).toBe("review"); + expect(v.reasons.map((r) => r.code)).toContain("screenshot"); + }); +}); diff --git a/app/scripts/lib/taste.mjs b/app/scripts/lib/taste.mjs new file mode 100644 index 0000000..1e2c195 --- /dev/null +++ b/app/scripts/lib/taste.mjs @@ -0,0 +1,117 @@ +// taste.json — the learned culling profile. Pure read/merge/update helpers so +// the feedback logic is unit-testable; disk I/O stays thin. +import fs from "node:fs"; +import { parseTasteProfile } from "@keeper/schema"; +import { tastePath } from "./paths.mjs"; + +export function readTaste(home) { + try { + return parseTasteProfile(JSON.parse(fs.readFileSync(tastePath(home), "utf8"))); + } catch { + return parseTasteProfile({}); + } +} + +export function writeTaste(home, taste) { + const validated = parseTasteProfile(taste); + validated.updatedAt = new Date().toISOString(); + fs.writeFileSync(tastePath(home), `${JSON.stringify(validated, null, 2)}\n`); + return validated; +} + +const MAX_EXEMPLARS = 200; + +/** + * Fold one user verdict into the profile. Overrides (user contradicts the AI) + * become few-shot exemplars + stats; confirmations only bump stats. Returns + * the updated profile (callers persist it). + */ +export function recordVerdict(taste, { asset, userFlag }) { + const ai = asset.ai; + if (!ai || userFlag === "unrated") return taste; + const reason = ai.reasons[0]?.code ?? "other"; + const stats = { ...taste.stats }; + const entry = { kept: 0, confirmed: 0, ...(stats[reason] ?? {}) }; + + const contradicts = + (ai.suggestion === "reject" && userFlag === "pick") || + (ai.suggestion === "keep" && userFlag === "reject"); + + if (contradicts) { + entry.kept += 1; + stats[reason] = entry; + const exemplar = { + assetId: asset.id, + thumbRel: asset.thumbRel, + aiSuggestion: ai.suggestion, + aiReason: reason, + userFlag, + at: new Date().toISOString(), + }; + return { + ...taste, + stats, + exemplars: [...taste.exemplars.slice(-(MAX_EXEMPLARS - 1)), exemplar], + }; + } + + const confirms = + (ai.suggestion === "reject" && userFlag === "reject") || + (ai.suggestion === "keep" && userFlag === "pick"); + if (confirms) { + entry.confirmed += 1; + stats[reason] = entry; + return { ...taste, stats }; + } + return taste; +} + +/** + * Re-derive thresholds from override stats: repeated "you rejected this but I + * kept it" signals loosen the matching threshold, and vice versa. Moves are + * deliberately small and bounded so a handful of clicks can't swing the + * pipeline wildly. + */ +export function tunedThresholds(taste) { + const t = { ...taste.thresholds }; + const blur = taste.stats["blurry"] ?? { kept: 0, confirmed: 0 }; + // Every 5 net overrides of blur rejects lowers the reject bar ~20%, floor 4. + const netKept = Math.max(0, blur.kept - blur.confirmed / 4); + const steps = Math.min(5, Math.floor(netKept / 5)); + t.blurReject = Math.max(4, t.blurReject * 0.8 ** steps); + + const exposure = combineStats(taste.stats, ["underexposed", "overexposed"]); + const expSteps = Math.min(4, Math.floor(Math.max(0, exposure.kept - exposure.confirmed / 4) / 5)); + t.clipReject = Math.min(0.95, t.clipReject + expSteps * 0.08); + + return t; +} + +function combineStats(stats, codes) { + return codes.reduce( + (acc, code) => { + const s = stats[code] ?? { kept: 0, confirmed: 0 }; + return { kept: acc.kept + s.kept, confirmed: acc.confirmed + s.confirmed }; + }, + { kept: 0, confirmed: 0 }, + ); +} + +/** Render the standing rules + recent overrides for LLM judge prompts. */ +export function tasteForPrompt(taste, { maxExemplars = 12 } = {}) { + const lines = []; + if (taste.rules.length > 0) { + lines.push("The user's standing rules (follow them strictly):"); + for (const rule of taste.rules) lines.push(`- ${rule.text}`); + } + const recent = taste.exemplars.slice(-maxExemplars); + if (recent.length > 0) { + lines.push("Recent corrections (the user disagreed with the AI; learn from these):"); + for (const ex of recent) { + lines.push( + `- AI said ${ex.aiSuggestion}${ex.aiReason ? ` (${ex.aiReason})` : ""}, user chose ${ex.userFlag}${ex.note ? ` — ${ex.note}` : ""}`, + ); + } + } + return lines.join("\n"); +} diff --git a/app/scripts/lib/taste.test.mjs b/app/scripts/lib/taste.test.mjs new file mode 100644 index 0000000..dba1a21 --- /dev/null +++ b/app/scripts/lib/taste.test.mjs @@ -0,0 +1,87 @@ +import { describe, expect, it } from "vitest"; +import { parseTasteProfile } from "@keeper/schema"; +import { recordVerdict, tasteForPrompt, tunedThresholds } from "./taste.mjs"; + +const blank = () => parseTasteProfile({}); + +const rejectedAsset = (id = "a1") => ({ + id, + thumbRel: `.keeper/thumbs/${id}.jpg`, + ai: { + suggestion: "reject", + confidence: 0.88, + reasons: [{ code: "blurry" }], + source: "cv", + }, +}); + +describe("recordVerdict", () => { + it("turns contradictions into exemplars + kept stats", () => { + const taste = recordVerdict(blank(), { asset: rejectedAsset(), userFlag: "pick" }); + expect(taste.exemplars).toHaveLength(1); + expect(taste.exemplars[0].aiReason).toBe("blurry"); + expect(taste.exemplars[0].userFlag).toBe("pick"); + expect(taste.stats.blurry.kept).toBe(1); + }); + + it("counts confirmations without exemplars", () => { + const taste = recordVerdict(blank(), { asset: rejectedAsset(), userFlag: "reject" }); + expect(taste.exemplars).toHaveLength(0); + expect(taste.stats.blurry.confirmed).toBe(1); + }); + + it("ignores verdicts on assets without AI suggestions", () => { + const taste = recordVerdict(blank(), { asset: { id: "x", ai: undefined }, userFlag: "pick" }); + expect(taste.exemplars).toHaveLength(0); + expect(Object.keys(taste.stats)).toHaveLength(0); + }); + + it("caps the exemplar list", () => { + let taste = blank(); + for (let i = 0; i < 250; i++) { + taste = recordVerdict(taste, { asset: rejectedAsset(`a${i}`), userFlag: "pick" }); + } + expect(taste.exemplars.length).toBeLessThanOrEqual(200); + expect(parseTasteProfile(taste).exemplars.length).toBeLessThanOrEqual(200); + }); +}); + +describe("tunedThresholds", () => { + it("keeps defaults with no overrides", () => { + const t = tunedThresholds(blank()); + expect(t.blurReject).toBe(blank().thresholds.blurReject); + }); + + it("loosens the blur bar after repeated overrides", () => { + let taste = blank(); + for (let i = 0; i < 12; i++) { + taste = recordVerdict(taste, { asset: rejectedAsset(`a${i}`), userFlag: "pick" }); + } + const t = tunedThresholds(taste); + expect(t.blurReject).toBeLessThan(blank().thresholds.blurReject); + expect(t.blurReject).toBeGreaterThanOrEqual(4); // bounded, never zero + }); + + it("confirmations offset overrides", () => { + let taste = blank(); + for (let i = 0; i < 6; i++) taste = recordVerdict(taste, { asset: rejectedAsset(`k${i}`), userFlag: "pick" }); + for (let i = 0; i < 24; i++) taste = recordVerdict(taste, { asset: rejectedAsset(`c${i}`), userFlag: "reject" }); + const t = tunedThresholds(taste); + expect(t.blurReject).toBe(blank().thresholds.blurReject); + }); +}); + +describe("tasteForPrompt", () => { + it("renders rules and recent corrections", () => { + let taste = blank(); + taste = { ...taste, rules: [{ id: "r1", text: "Never reject photos of my kids" }] }; + taste = recordVerdict(taste, { asset: rejectedAsset(), userFlag: "pick" }); + const prompt = tasteForPrompt(taste); + expect(prompt).toContain("Never reject photos of my kids"); + expect(prompt).toContain("AI said reject (blurry), user chose pick"); + }); + + it("is empty for a blank profile", () => { + expect(tasteForPrompt(blank())).toBe(""); + }); +}); diff --git a/app/scripts/lib/xmp.mjs b/app/scripts/lib/xmp.mjs new file mode 100644 index 0000000..7391780 --- /dev/null +++ b/app/scripts/lib/xmp.mjs @@ -0,0 +1,61 @@ +// Minimal XMP sidecar generation — the interop lingua franca. Lightroom, +// Capture One, and Bridge read xmp:Rating (-1 = rejected), xmp:Label, and +// dc:subject keywords from a sidecar named .xmp next to the original. + +function escapeXml(s) { + return s + .replace(/&/g, "&") + .replace(//g, ">") + .replace(/"/g, """); +} + +/** + * Build an XMP packet for an asset. Rating mapping: + * - user rejected -> xmp:Rating = -1 (Lightroom's "rejected" flag) + * - rated 1..5 -> xmp:Rating = n + * - picked, unrated-> xmp:Rating = 0 with a "keeper-pick" keyword + */ +export function buildXmp({ rating = 0, flag = "unrated", tags = [], caption }) { + const xmpRating = flag === "reject" ? -1 : rating; + const keywords = [...tags]; + if (flag === "pick") keywords.push("keeper-pick"); + + const subjectBlock = + keywords.length > 0 + ? ` + + +${keywords.map((k) => ` ${escapeXml(k)}`).join("\n")} + + ` + : ""; + + const descriptionBlock = caption + ? ` + + + ${escapeXml(caption)} + + ` + : ""; + + return ` + + + ${subjectBlock}${descriptionBlock} + + + + +`; +} + +/** Sidecar path convention: IMG_0001.jpg -> IMG_0001.xmp (Lightroom style). */ +export function sidecarPath(mediaPath) { + const dot = mediaPath.lastIndexOf("."); + return dot > 0 ? `${mediaPath.slice(0, dot)}.xmp` : `${mediaPath}.xmp`; +} diff --git a/app/scripts/lib/xmp.test.mjs b/app/scripts/lib/xmp.test.mjs new file mode 100644 index 0000000..d4e8c46 --- /dev/null +++ b/app/scripts/lib/xmp.test.mjs @@ -0,0 +1,39 @@ +import { describe, expect, it } from "vitest"; +import { buildXmp, sidecarPath } from "./xmp.mjs"; + +describe("buildXmp", () => { + it("maps rejection to Rating -1 (Lightroom convention)", () => { + const xmp = buildXmp({ flag: "reject", rating: 3 }); + expect(xmp).toContain('xmp:Rating="-1"'); + }); + + it("keeps star ratings for non-rejects", () => { + expect(buildXmp({ flag: "pick", rating: 4 })).toContain('xmp:Rating="4"'); + }); + + it("adds keeper-pick keyword and tags as dc:subject", () => { + const xmp = buildXmp({ flag: "pick", rating: 0, tags: ["ocean", "sunset"] }); + expect(xmp).toContain("ocean"); + expect(xmp).toContain("keeper-pick"); + }); + + it("escapes XML in captions", () => { + const xmp = buildXmp({ flag: "unrated", rating: 0, caption: 'kids & "playing"' }); + expect(xmp).toContain("kids & <dogs> "playing""); + expect(xmp).not.toContain(""); + }); + + it("omits subject/description blocks when empty", () => { + const xmp = buildXmp({ flag: "unrated", rating: 0 }); + expect(xmp).not.toContain("dc:subject"); + expect(xmp).not.toContain("dc:description"); + }); +}); + +describe("sidecarPath", () => { + it("swaps the extension", () => { + expect(sidecarPath("/x/IMG_1.CR3")).toBe("/x/IMG_1.xmp"); + expect(sidecarPath("/x/clip.mp4")).toBe("/x/clip.xmp"); + expect(sidecarPath("/x/noext")).toBe("/x/noext.xmp"); + }); +}); diff --git a/app/scripts/llm.mjs b/app/scripts/llm.mjs index 5835dcb..fe0d2ba 100644 --- a/app/scripts/llm.mjs +++ b/app/scripts/llm.mjs @@ -3,36 +3,36 @@ // (self-hosted Gemma via Ollama/vLLM, an enterprise gateway, Azure, etc.) // purely through environment variables — no code change to rotate models. // -// APERTURE_LLM_PROVIDER openai | anthropic | openai-compatible (default: openai) -// APERTURE_LLM_MODEL model id (default: gpt-5.5) -// APERTURE_LLM_BASE_URL override base URL (gateways / local) (optional) -// APERTURE_LLM_API_KEY generic key; falls back to OPENAI_API_KEY / ANTHROPIC_API_KEY +// KEEPER_LLM_PROVIDER openai | anthropic | openai-compatible (default: openai) +// KEEPER_LLM_MODEL model id (default: gpt-5.5) +// KEEPER_LLM_BASE_URL override base URL (gateways / local) (optional) +// KEEPER_LLM_API_KEY generic key; falls back to OPENAI_API_KEY / ANTHROPIC_API_KEY import { createOpenAI } from "@ai-sdk/openai"; import { createAnthropic } from "@ai-sdk/anthropic"; import { createOpenAICompatible } from "@ai-sdk/openai-compatible"; export function llmConfig() { - const provider = (process.env.APERTURE_LLM_PROVIDER || "openai").toLowerCase(); - const model = process.env.APERTURE_LLM_MODEL || "gpt-5.5"; - const baseURL = process.env.APERTURE_LLM_BASE_URL || undefined; + const provider = (process.env.KEEPER_LLM_PROVIDER || "openai").toLowerCase(); + const model = process.env.KEEPER_LLM_MODEL || "gpt-5.5"; + const baseURL = process.env.KEEPER_LLM_BASE_URL || undefined; const apiKey = - process.env.APERTURE_LLM_API_KEY || + process.env.KEEPER_LLM_API_KEY || (provider === "anthropic" ? process.env.ANTHROPIC_API_KEY : process.env.OPENAI_API_KEY) || process.env.OPENAI_API_KEY || process.env.ANTHROPIC_API_KEY; return { provider, model, baseURL, apiKey }; } -/** True when generation can run. Local/compatible endpoints may need only a baseURL. */ +/** True when the LLM tier can run. Local/compatible endpoints may need only a baseURL. */ export function isLlmConfigured() { const { provider, apiKey, baseURL } = llmConfig(); if (provider === "openai-compatible") return Boolean(baseURL || apiKey); return Boolean(apiKey); } -/** Reasoning effort for OpenAI reasoning models (Settings → Agent Preferences). */ +/** Reasoning effort for OpenAI reasoning models (Settings → AI). */ export function reasoningEffort() { - const v = (process.env.APERTURE_REASONING_EFFORT || "low").toLowerCase(); + const v = (process.env.KEEPER_REASONING_EFFORT || "low").toLowerCase(); return ["low", "medium", "high"].includes(v) ? v : "low"; } @@ -42,7 +42,7 @@ export function resolveModel() { case "anthropic": return createAnthropic({ apiKey, baseURL })(model); case "openai-compatible": - return createOpenAICompatible({ name: "aperture-llm", apiKey: apiKey ?? "", baseURL })(model); + return createOpenAICompatible({ name: "keeper-llm", apiKey: apiKey ?? "", baseURL })(model); case "openai": default: return createOpenAI({ apiKey, baseURL })(model); diff --git a/app/scripts/query.mjs b/app/scripts/query.mjs new file mode 100644 index 0000000..a105789 --- /dev/null +++ b/app/scripts/query.mjs @@ -0,0 +1,118 @@ +// Agent/CLI window into the catalog. Prints JSON to stdout. +// +// node app/scripts/query.mjs --counts +// node app/scripts/query.mjs --search "ocean at sunset" [--limit 20] +// node app/scripts/query.mjs --list unrated|pick|reject [--limit 50] +// node app/scripts/query.mjs --review (the three confidence queues) +// node app/scripts/query.mjs --asset (full record incl. thumb path) +// node app/scripts/query.mjs --imports +// All accept --library . +import { reviewQueue } from "@keeper/schema"; +import { openCatalog } from "./lib/catalog.mjs"; +import { embedText, modelCached, rankBySimilarity } from "./lib/embeddings.mjs"; +import { ensureLayout, resolveHome, safeJoin } from "./lib/paths.mjs"; +import { readTaste } from "./lib/taste.mjs"; + +function arg(name) { + const i = process.argv.indexOf(`--${name}`); + return i >= 0 ? process.argv[i + 1] : undefined; +} +const has = (name) => process.argv.includes(`--${name}`); + +function brief(home, asset) { + return { + id: asset.id, + fileName: asset.fileName, + mediaType: asset.mediaType, + capturedAt: asset.capturedAt, + flag: asset.user.flag, + rating: asset.user.rating, + ai: asset.ai ? { suggestion: asset.ai.suggestion, confidence: asset.ai.confidence, reasons: asset.ai.reasons.map((r) => r.code) } : null, + groupId: asset.groupId, + tags: asset.tags, + caption: asset.caption, + thumbPath: asset.thumbRel ? safeJoin(home, asset.thumbRel) : null, + originalPath: safeJoin(home, asset.relPath), + }; +} + +async function main() { + const home = ensureLayout(resolveHome(arg("library"))); + const catalog = openCatalog(home, { stampOnWrite: false }); + const limit = Math.min(500, Number(arg("limit")) || 50); + + try { + if (has("counts")) { + console.log(JSON.stringify({ counts: catalog.countsSummary(), days: catalog.listDays().length }, null, 2)); + return; + } + + if (has("imports")) { + console.log(JSON.stringify(catalog.listImports(20), null, 2)); + return; + } + + if (arg("asset")) { + const asset = catalog.getAsset(arg("asset")); + console.log(JSON.stringify(asset ? brief(home, asset) : null, null, 2)); + return; + } + + if (arg("list")) { + const flag = arg("list"); + const assets = catalog.listAssets({ flag, limit }); + console.log(JSON.stringify(assets.map((a) => brief(home, a)), null, 2)); + return; + } + + if (has("review")) { + const sure = readTaste(home).thresholds.sureConfidence; + const unrated = catalog.listAssets({ flag: "unrated", hasAi: true, limit: 50_000 }); + const queues = { "sure-reject": [], "sure-keep": [], "needs-eye": [] }; + for (const asset of unrated) { + const q = reviewQueue(asset, sure); + if (q in queues && queues[q].length < limit) queues[q].push(brief(home, asset)); + } + console.log(JSON.stringify(queues, null, 2)); + return; + } + + if (arg("search")) { + const query = arg("search"); + const matrix = catalog.allEmbeddings(); + if (matrix.length > 0 && modelCached(home)) { + const vec = await embedText(home, query); + const ranked = rankBySimilarity(vec, matrix, limit); + const records = catalog.getAssets(ranked.map((r) => r.id)); + const byId = new Map(records.map((r) => [r.id, r])); + console.log( + JSON.stringify( + ranked + .filter((r) => byId.has(r.id)) + .map((r) => ({ score: Math.round(r.score * 1000) / 1000, ...brief(home, byId.get(r.id)) })), + null, + 2, + ), + ); + } else { + console.log(JSON.stringify({ error: "no embeddings yet — run reprocess.mjs --stage embed" }, null, 2)); + } + return; + } + + console.log( + JSON.stringify( + { usage: "--counts | --imports | --asset | --list | --review | --search " }, + null, + 2, + ), + ); + } finally { + catalog.close(); + } +} + +main().catch((err) => { + console.error(`ERROR ${err?.stack ?? err}`); + process.exit(1); +}); diff --git a/app/scripts/render.mjs b/app/scripts/render.mjs deleted file mode 100644 index ca587e6..0000000 --- a/app/scripts/render.mjs +++ /dev/null @@ -1,79 +0,0 @@ -// Standalone Remotion render. Run in a child Node process (spawned by the -// Electron main process, or directly: `node app/scripts/render.mjs --slug `). -// Emits line-oriented progress the main process parses: PHASE/PROGRESS/DONE/ERROR. -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { bundle } from "@remotion/bundler"; -import { ensureBrowser, renderMedia, selectComposition } from "@remotion/renderer"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const edl = JSON.parse(fs.readFileSync(path.join(projectDir, "edl.json"), "utf8")); - - // Export overrides from Settings. Frame rate re-times nothing (EDL timing is - // in seconds); resolution renders the same layout at a scale factor. - const fpsOverride = Number(arg("fps")) || null; - if (fpsOverride) edl.format = { ...edl.format, fps: fpsOverride }; - const resolution = Number(arg("resolution")) || null; - const shortEdge = Math.min(edl.format.width ?? 1080, edl.format.height ?? 1920); - const scale = resolution ? Math.min(1, resolution / shortEdge) : 1; - const compression = arg("compression") || "social"; - - const entryPoint = path.join(repoRoot, "app", "src", "renderer", "src", "motion", "index.ts"); - const rendersDir = path.join(projectDir, "renders"); - fs.mkdirSync(rendersDir, { recursive: true }); - const output = path.join(rendersDir, `${slug}-${Date.now()}.mp4`); - - // No assetBaseUrl: the composition uses staticFile(), served from publicDir. - const inputProps = { edl }; - - console.log("PHASE preparing"); - await ensureBrowser(); - - console.log("PHASE bundling"); - const serveUrl = await bundle({ entryPoint, publicDir: projectDir }); - - console.log("PHASE composition"); - const composition = await selectComposition({ serveUrl, id: "SocialVideo", inputProps }); - - console.log("PHASE rendering"); - // Hardware-accelerated encode when the user enabled it (VideoToolbox on macOS, - // etc.); "if-possible" falls back to software when unavailable. - const hwaccel = process.argv.includes("--hwaccel"); - // Compression presets: software encodes steer by CRF (lower = better quality); - // hardware encoders ignore CRF, so steer those by target bitrate instead. - const CRF = { high: 18, social: 23, max: 28 }; - const BITRATE = { high: "14M", social: "8M", max: "4M" }; - await renderMedia({ - composition, - serveUrl, - codec: "h264", - outputLocation: output, - inputProps, - scale, - hardwareAcceleration: hwaccel ? "if-possible" : "disable", - ...(hwaccel - ? { videoBitrate: BITRATE[compression] ?? BITRATE.social } - : { crf: CRF[compression] ?? CRF.social }), - onProgress: ({ progress }) => console.log(`PROGRESS ${Math.round(progress * 100)}`), - }); - - console.log(`DONE ${output}`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/reprocess.mjs b/app/scripts/reprocess.mjs new file mode 100644 index 0000000..6b3eeaf --- /dev/null +++ b/app/scripts/reprocess.mjs @@ -0,0 +1,70 @@ +// Resume/re-run pipeline stages for assets that missed them (interrupted +// import, model that wasn't downloaded yet, threshold changes). +// +// node app/scripts/reprocess.mjs [--library ] [--stage derive|cv|group|embed|all] +import { openCatalog } from "./lib/catalog.mjs"; +import { closeMetadataReader } from "./lib/media.mjs"; +import { ensureLayout, resolveHome } from "./lib/paths.mjs"; +import { stageCv, stageDerive, stageEmbed, stageGroup } from "./lib/pipeline.mjs"; + +function arg(name) { + const i = process.argv.indexOf(`--${name}`); + return i >= 0 ? process.argv[i + 1] : undefined; +} + +const phase = (name) => console.log(`PHASE ${name}`); +const progress = (pct) => console.log(`PROGRESS ${Math.round(pct)}`); +const span = (from, to) => (frac) => progress(from + (to - from) * Math.min(1, Math.max(0, frac))); + +async function main() { + const home = ensureLayout(resolveHome(arg("library"))); + const which = arg("stage") ?? "all"; + const catalog = openCatalog(home, { stampOnWrite: true }); + const summary = {}; + + try { + if (which === "derive" || which === "all") { + const missing = catalog.assetsMissingStage("derive"); + phase(`deriving ${missing.length}`); + await stageDerive(catalog, missing, { onProgress: span(0, 30) }); + summary.derived = missing.length; + } + + let hashes = new Map(); + if (which === "cv" || which === "group" || which === "all") { + const missing = catalog.assetsMissingStage("cv"); + phase(`measuring ${missing.length}`); + ({ hashes } = await stageCv(catalog, missing, { onProgress: span(30, 60) })); + summary.measured = missing.length; + } + + if (which === "group" || which === "all") { + const missing = catalog.assetsMissingStage("group"); + phase(`grouping ${missing.length}`); + if (missing.length > 0) { + const stats = stageGroup(catalog, missing, hashes); + summary.groups = stats.groups; + } + span(60, 70)(1); + } + + if (which === "embed" || which === "all") { + const missing = catalog.assetsMissingStage("embed"); + phase(`indexing ${missing.length}`); + const stats = await stageEmbed(catalog, missing, { onProgress: span(70, 100) }); + summary.embedded = stats.embedded; + summary.embedSkipped = stats.skipped; + } + + progress(100); + console.log(`DONE ${JSON.stringify(summary)}`); + } finally { + await closeMetadataReader(); + catalog.close(); + } +} + +main().catch((err) => { + console.error(`ERROR ${err?.stack ?? err}`); + process.exit(1); +}); diff --git a/app/scripts/transcribe.mjs b/app/scripts/transcribe.mjs deleted file mode 100644 index 08cdfe3..0000000 --- a/app/scripts/transcribe.mjs +++ /dev/null @@ -1,92 +0,0 @@ -// Whisper auto-captions. Extracts audio with ffmpeg, runs whisper.cpp locally, -// and writes word-level caption timings into the project's edl.json caption track. -// Run: `node app/scripts/transcribe.mjs --slug ` -// -// NOTE: needs real speech audio to produce meaningful output (the bundled test -// clips are tone-only). Whisper binary + model download on first run. -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { execFileSync } from "node:child_process"; -import ffmpegPath from "ffmpeg-static"; -import { - downloadWhisperModel, - installWhisperCpp, - toCaptions, - transcribe as whisperTranscribe, -} from "@remotion/install-whisper-cpp"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); -const WHISPER_DIR = path.join(repoRoot, ".whisper"); -const MODEL = "base.en"; - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} -const round = (n) => Math.round(n * 100) / 100; - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const edlPath = path.join(projectDir, "edl.json"); - const edl = JSON.parse(fs.readFileSync(edlPath, "utf8")); - - // Prefer the voiceover clip's asset (that's the speech we want captioned), - // then any audio asset, then fall back to the video's own audio. - const voClip = edl.tracks - .flatMap((t) => (t.type === "audio" ? t.clips : [])) - .find((c) => c.role === "voiceover"); - const voAsset = voClip ? edl.assets.find((a) => a.id === voClip.assetId) : undefined; - const asset = - voAsset ?? - edl.assets.find((a) => a.kind === "audio") ?? - edl.assets.find((a) => a.kind === "video"); - if (!asset) throw new Error("no audio/video asset to transcribe"); - - // edl.json is untrusted (shareable project file): the transcode input must - // stay inside the project folder, mirroring the app's safeProjectPath guard. - const input = path.resolve(projectDir, asset.src); - if (!input.startsWith(path.resolve(projectDir) + path.sep)) { - throw new Error(`asset src escapes the project folder: ${asset.src}`); - } - const wavDir = path.join(projectDir, "transcripts"); - fs.mkdirSync(wavDir, { recursive: true }); - const wav = path.join(wavDir, `${slug}-16k.wav`); - - console.log("PHASE extracting-audio"); - execFileSync(ffmpegPath, ["-y", "-i", input, "-ar", "16000", "-ac", "1", "-c:a", "pcm_s16le", wav], { - stdio: "ignore", - }); - - console.log("PHASE installing-whisper"); - await installWhisperCpp({ to: WHISPER_DIR, version: "1.5.5" }); - await downloadWhisperModel({ model: MODEL, folder: WHISPER_DIR }); - - console.log("PHASE transcribing"); - const whisperCppOutput = await whisperTranscribe({ - inputPath: wav, - whisperPath: WHISPER_DIR, - model: MODEL, - tokenLevelTimestamps: true, - }); - const { captions } = toCaptions({ whisperCppOutput }); - const words = captions - .map((c) => ({ text: String(c.text).trim(), start: round(c.startMs / 1000), end: round(c.endMs / 1000) })) - .filter((w) => w.text); - - const capTrack = edl.tracks.find((t) => t.type === "caption"); - if (capTrack) capTrack.words = words; - else edl.tracks.push({ id: "cap", type: "caption", style: "karaoke", words }); - - fs.writeFileSync(edlPath, `${JSON.stringify(edl, null, 2)}\n`); - console.log(`DONE ${words.length} caption words`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/tts-util.mjs b/app/scripts/tts-util.mjs deleted file mode 100644 index 3589a50..0000000 --- a/app/scripts/tts-util.mjs +++ /dev/null @@ -1,81 +0,0 @@ -// Pure helpers for the ElevenLabs TTS pipeline (ported from Claudia's -// narration preprocessing, general subset). Kept dependency-free and pure so -// they unit-test cleanly. - -const round = (n) => Math.round(n * 100) / 100; - -/** - * Prepare narration text for TTS: - * - symbols ElevenLabs reads unpredictably: `§` -> "Section", ` & ` -> " and " - * - a 0.25s break after each sentence inside a paragraph (natural pacing) - * - a 0.5s break between paragraphs (a breath between beats) - * On-screen text is untouched — this transform applies to the spoken text only. - */ -export function preprocessForTts(text) { - const paragraphs = String(text) - .replace(/§/g, "Section ") - .replace(/\s&\s/g, " and ") - .split(/\n\s*\n/) - .map((p) => p.trim()) - .filter(Boolean) - .map((p) => - // Insert a short break after sentence-ending punctuation followed by a - // space (keeps decimals like "2.5" and trailing sentence-enders intact). - p.replace(/([.!?])\s+(?=[A-Z0-9"'])/g, '$1 '), - ); - return paragraphs.join('\n\n\n\n'); -} - -/** - * Convert ElevenLabs character alignment (from /with-timestamps) into - * word-level timings compatible with the EDL caption track. Break tags may or - * may not appear in the alignment depending on API behavior — tokens that - * look like markup are dropped either way. - */ -export function alignmentToWords(alignment) { - const chars = alignment?.characters ?? []; - const starts = alignment?.character_start_times_seconds ?? []; - const ends = alignment?.character_end_times_seconds ?? []; - const words = []; - let text = ""; - let start = null; - let end = null; - let inTag = false; - - const flush = () => { - if (text) words.push({ text, start: round(start ?? 0), end: round(end ?? start ?? 0) }); - text = ""; - start = null; - end = null; - }; - - for (let i = 0; i < chars.length; i++) { - const ch = chars[i]; - // Markup spans (e.g. ) are never spoken words; skip - // them wholesale, including the spaces inside them. - if (inTag) { - if (ch === ">") inTag = false; - continue; - } - if (ch === "<") { - flush(); - inTag = true; - continue; - } - if (/\s/.test(ch)) { - flush(); - continue; - } - if (text === "") start = starts[i]; - text += ch; - end = ends[i]; - } - flush(); - return words; -} - -/** Stable cache key for a synthesis request (voice + model + processed text). */ -export async function synthesisHash(voiceId, modelId, processedText) { - const { createHash } = await import("node:crypto"); - return createHash("sha256").update(`${voiceId}\n${modelId}\n${processedText}`).digest("hex").slice(0, 12); -} diff --git a/app/scripts/tts-util.test.mjs b/app/scripts/tts-util.test.mjs deleted file mode 100644 index 17846e3..0000000 --- a/app/scripts/tts-util.test.mjs +++ /dev/null @@ -1,65 +0,0 @@ -import { describe, expect, it } from "vitest"; -import { alignmentToWords, preprocessForTts, synthesisHash } from "./tts-util.mjs"; - -describe("preprocessForTts", () => { - it("inserts paragraph breaks between beats", () => { - const out = preprocessForTts("First beat.\n\nSecond beat."); - expect(out).toBe('First beat.\n\n\n\nSecond beat.'); - }); - - it("inserts sentence breaks within a paragraph", () => { - const out = preprocessForTts("One sentence. Another one! Third?"); - expect(out).toContain('One sentence. Another one!'); - expect(out).toContain('Another one! Third?'); - }); - - it("substitutes symbols ElevenLabs mispronounces and keeps decimals intact", () => { - const out = preprocessForTts("Section §230 applies to Smith & Co at 2.5 percent."); - expect(out).toContain("Section 230"); - expect(out).toContain("Smith and Co"); - expect(out).toContain("2.5 percent"); - expect(out).not.toContain("§"); - }); - - it("drops empty paragraphs from stray blank lines", () => { - const out = preprocessForTts("A.\n\n\n\nB."); - expect(out.match(//g)).toHaveLength(1); - }); -}); - -describe("alignmentToWords", () => { - const align = (text, offset = 0) => ({ - characters: [...text], - character_start_times_seconds: [...text].map((_, i) => offset + i * 0.1), - character_end_times_seconds: [...text].map((_, i) => offset + (i + 1) * 0.1), - }); - - it("groups characters into words with start/end times", () => { - const words = alignmentToWords(align("hi there")); - expect(words).toEqual([ - { text: "hi", start: 0, end: 0.2 }, - { text: "there", start: 0.3, end: 0.8 }, - ]); - }); - - it("drops break-tag tokens if the API includes them in the alignment", () => { - const words = alignmentToWords(align('go now')); - expect(words.map((w) => w.text)).toEqual(["go", "now"]); - }); - - it("handles an empty alignment", () => { - expect(alignmentToWords(undefined)).toEqual([]); - expect(alignmentToWords({})).toEqual([]); - }); -}); - -describe("synthesisHash", () => { - it("is stable for identical inputs and distinct across voices", async () => { - const a = await synthesisHash("voiceA", "model", "text"); - const b = await synthesisHash("voiceA", "model", "text"); - const c = await synthesisHash("voiceB", "model", "text"); - expect(a).toBe(b); - expect(a).not.toBe(c); - expect(a).toMatch(/^[0-9a-f]{12}$/); - }); -}); diff --git a/app/scripts/tts.mjs b/app/scripts/tts.mjs deleted file mode 100644 index 650c8b5..0000000 --- a/app/scripts/tts.mjs +++ /dev/null @@ -1,181 +0,0 @@ -// Synthesize the project's narration (narration.md) with an ElevenLabs voice -// and land it on the timeline: audio on the "vo" track, word-level captions on -// the caption track, music ducked. One API call via /with-timestamps returns -// both audio and word timings (no STT pass needed — unlike Claudia, which -// derives timings via scribe_v1 STT). -// -// Pipeline: preprocess (breaks + substitutions) -> TTS -> two-pass ffmpeg -// loudnorm to -14 LUFS / -1.5 dBTP (raw ElevenLabs lands ~-36 LUFS) -> write -// assets/voiceover-.mp3 -> update edl.json. Synthesis is cached by -// hash(voice+model+text) so an unchanged script never re-burns credits. -// -// Run: ELEVENLABS_API_KEY=... node app/scripts/tts.mjs --slug --voice -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { spawnSync } from "node:child_process"; -import ffmpegPath from "ffmpeg-static"; -import { alignmentToWords, preprocessForTts, synthesisHash } from "./tts-util.mjs"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); -const MODEL_ID = "eleven_multilingual_v2"; -const TARGET_LUFS = -14; -const TARGET_TP = -1.5; - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} - -// Two-pass loudnorm (ported from Claudia's normalize-audio.mjs): pass 1 -// measures, pass 2 applies with linear=true so it's gain-only and word -// timings don't shift. -function normalizeLoudness(file) { - const measure = spawnSync( - ffmpegPath, - ["-hide_banner", "-i", file, "-af", `loudnorm=I=${TARGET_LUFS}:TP=${TARGET_TP}:LRA=11:print_format=json`, "-f", "null", "-"], - { encoding: "utf8" }, - ); - const m = measure.stderr.match(/\{[\s\S]*\}/); - if (!m) return false; - let stats; - try { - stats = JSON.parse(m[0]); - } catch { - return false; - } - if (Math.abs(Number(stats.input_i) - TARGET_LUFS) <= 1) return true; // already at target - const out = `${file}.norm.mp3`; - const apply = spawnSync(ffmpegPath, [ - "-y", "-hide_banner", "-i", file, - "-af", - `loudnorm=I=${TARGET_LUFS}:TP=${TARGET_TP}:LRA=11:linear=true:measured_I=${stats.input_i}:measured_TP=${stats.input_tp}:measured_LRA=${stats.input_lra}:measured_thresh=${stats.input_thresh}`, - "-b:a", "192k", - out, - ]); - if (apply.status !== 0 || !fs.existsSync(out)) return false; - fs.renameSync(out, file); - return true; -} - -async function main() { - const slug = arg("slug"); - const voiceId = arg("voice"); - if (!slug) throw new Error("missing --slug"); - if (!voiceId) throw new Error("missing --voice"); - const apiKey = process.env.ELEVENLABS_API_KEY; - if (!apiKey) { - console.error("ERROR no ElevenLabs API key configured (Settings -> Voices)."); - process.exit(3); - } - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const narration = fs.readFileSync(path.join(projectDir, "narration.md"), "utf8").trim(); - if (!narration) { - console.error("ERROR narration.md is empty — write or draft a script first."); - process.exit(2); - } - - console.log("PHASE preprocessing"); - const processed = preprocessForTts(narration); - const hash = await synthesisHash(voiceId, MODEL_ID, processed); - const rel = `assets/voiceover-${hash}.mp3`; - const audioFile = path.join(projectDir, rel); - const timingsFile = path.join(projectDir, "transcripts", `voiceover-${hash}.words.json`); - fs.mkdirSync(path.dirname(audioFile), { recursive: true }); - fs.mkdirSync(path.dirname(timingsFile), { recursive: true }); - - let words; - if (fs.existsSync(audioFile) && fs.existsSync(timingsFile)) { - console.log("PHASE reusing cached synthesis"); - words = JSON.parse(fs.readFileSync(timingsFile, "utf8")); - } else { - console.log("PHASE synthesizing"); - const res = await fetch( - `https://api.elevenlabs.io/v1/text-to-speech/${voiceId}/with-timestamps?output_format=mp3_44100_128`, - { - method: "POST", - headers: { "xi-api-key": apiKey, "Content-Type": "application/json" }, - body: JSON.stringify({ - text: processed, - model_id: MODEL_ID, - voice_settings: { stability: 0.5 }, - }), - }, - ); - if (!res.ok) { - let detail = `HTTP ${res.status}`; - try { - const body = await res.json(); - detail = typeof body.detail === "string" ? body.detail : (body.detail?.message ?? detail); - } catch { - // keep the status text - } - console.error(`ERROR ElevenLabs synthesis failed: ${detail}`); - process.exit(2); - } - console.log("PROGRESS 60"); - const data = await res.json(); - fs.writeFileSync(audioFile, Buffer.from(data.audio_base64, "base64")); - words = alignmentToWords(data.alignment); - fs.writeFileSync(timingsFile, `${JSON.stringify(words)}\n`); - - console.log("PHASE normalizing loudness"); - if (!normalizeLoudness(audioFile)) { - console.error("WARN loudness normalization skipped (measurement failed)"); - } - } - console.log("PROGRESS 85"); - - console.log("PHASE updating timeline"); - const edlPath = path.join(projectDir, "edl.json"); - const edl = JSON.parse(fs.readFileSync(edlPath, "utf8")); - const durationSec = words.length > 0 ? Math.max(words[words.length - 1].end, 1) : 1; - const assetId = `vo-${hash}`; - - // Asset entry (replace any prior generated-VO assets no longer referenced). - edl.assets = (edl.assets ?? []).filter((a) => !(a.id.startsWith("vo-") && a.id !== assetId)); - if (!edl.assets.some((a) => a.id === assetId)) { - edl.assets.push({ id: assetId, kind: "audio", src: rel, durationSec }); - } - - // Voiceover clip: replace previous *generated* VOs, keep mic recordings. - let vo = edl.tracks.find((t) => t.type === "audio" && t.id === "vo"); - if (!vo) { - vo = { id: "vo", type: "audio", name: "Voiceover", clips: [] }; - edl.tracks.push(vo); - } - vo.clips = (vo.clips ?? []).filter((c) => !String(c.assetId).startsWith("vo-")); - vo.clips.push({ - id: `a-${assetId}`, - assetId, - start: 0, - in: 0, - out: durationSec, - gain: 0, - duckUnderVoice: false, - role: "voiceover", - }); - - // Duck every music bed under the new narration. - for (const track of edl.tracks) { - if (track.type !== "audio") continue; - for (const clip of track.clips ?? []) { - if (clip.role === "music") clip.duckUnderVoice = true; - } - } - - // Captions from the synthesis alignment (no whisper pass needed). - const cap = edl.tracks.find((t) => t.type === "caption"); - if (cap) cap.words = words; - else edl.tracks.push({ id: "cap", type: "caption", style: "karaoke", words }); - - fs.writeFileSync(edlPath, `${JSON.stringify(edl, null, 2)}\n`); - console.log(`DONE voiceover ${rel} (${words.length} caption words, ${durationSec.toFixed(1)}s)`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/scripts/verdict.mjs b/app/scripts/verdict.mjs new file mode 100644 index 0000000..5daba91 --- /dev/null +++ b/app/scripts/verdict.mjs @@ -0,0 +1,58 @@ +// Agent/CLI verdict setter — the write half of query.mjs. Records taste +// signals exactly like the app does (contradictions become exemplars). +// +// node app/scripts/verdict.mjs --ids a,b,c --flag pick|reject|unrated [--rating 0..5] +// node app/scripts/verdict.mjs --group --best +// All accept --library . +import { openCatalog } from "./lib/catalog.mjs"; +import { ensureLayout, resolveHome } from "./lib/paths.mjs"; +import { readTaste, recordVerdict, writeTaste } from "./lib/taste.mjs"; + +function arg(name) { + const i = process.argv.indexOf(`--${name}`); + return i >= 0 ? process.argv[i + 1] : undefined; +} + +async function main() { + const home = ensureLayout(resolveHome(arg("library"))); + const catalog = openCatalog(home, { stampOnWrite: true }); + + try { + if (arg("group") && arg("best")) { + catalog.setGroupPick(arg("group"), arg("best"), "user"); + console.log(JSON.stringify({ ok: true, group: arg("group"), best: arg("best") })); + return; + } + + const ids = (arg("ids") ?? "").split(",").filter(Boolean); + if (ids.length === 0) throw new Error("missing --ids or --group/--best"); + const flag = arg("flag"); + const rating = arg("rating") !== undefined ? Number(arg("rating")) : undefined; + if (!flag && rating === undefined) throw new Error("missing --flag or --rating"); + if (flag && !["pick", "reject", "unrated"].includes(flag)) throw new Error(`bad flag: ${flag}`); + if (rating !== undefined && !(Number.isInteger(rating) && rating >= 0 && rating <= 5)) { + throw new Error(`bad rating: ${arg("rating")}`); + } + + const before = catalog.setUserVerdict(ids, { flag, rating }); + if (flag) { + let taste = readTaste(home); + let changed = false; + for (const prior of before) { + if (prior.ai && prior.user.flag === "unrated") { + taste = recordVerdict(taste, { asset: prior, userFlag: flag }); + changed = true; + } + } + if (changed) writeTaste(home, taste); + } + console.log(JSON.stringify({ ok: true, updated: ids.length })); + } finally { + catalog.close(); + } +} + +main().catch((err) => { + console.error(`ERROR ${err?.stack ?? err}`); + process.exit(1); +}); diff --git a/app/scripts/write-narration.mjs b/app/scripts/write-narration.mjs deleted file mode 100644 index 3c7df24..0000000 --- a/app/scripts/write-narration.mjs +++ /dev/null @@ -1,97 +0,0 @@ -// Draft voiceover narration for a project's current cut (Claudia-style: the -// LLM writes, the human reviews in the VO dialog before any TTS credits burn). -// Reads prompt.md + edl.json (+ style.json when present), sizes the script to -// the cut (~2.5 spoken words/sec), and writes projects//narration.md. -// -// Run: OPENAI_API_KEY=... node app/scripts/write-narration.mjs --slug -import { fileURLToPath } from "node:url"; -import path from "node:path"; -import fs from "node:fs"; -import { generateText } from "ai"; -import { isLlmConfigured, llmConfig, resolveModel, reasoningEffort } from "./llm.mjs"; - -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const repoRoot = path.resolve(__dirname, "..", ".."); -const WORDS_PER_SEC = 2.5; - -function arg(name) { - const i = process.argv.indexOf(`--${name}`); - return i >= 0 ? process.argv[i + 1] : undefined; -} -function readMaybe(file) { - try { - return fs.readFileSync(file, "utf8"); - } catch { - return ""; - } -} - -function videoLengthSec(edl) { - return (edl.tracks ?? []) - .filter((t) => t.type === "video") - .flatMap((t) => t.clips ?? []) - .reduce((m, c) => Math.max(m, (c.start ?? 0) + (c.out ?? 0) - (c.in ?? 0)), 0); -} - -async function main() { - const slug = arg("slug"); - if (!slug) throw new Error("missing --slug"); - if (!isLlmConfigured()) { - console.error("ERROR no LLM credentials configured (set OPENAI_API_KEY in app/.env.local)."); - process.exit(3); - } - - const projectDir = path.join(process.env.APERTURE_PROJECTS_DIR || path.join(repoRoot, "projects"), slug); - const edl = JSON.parse(readMaybe(path.join(projectDir, "edl.json")) || "{}"); - const promptMd = readMaybe(path.join(projectDir, "prompt.md")); - const style = readMaybe(path.join(projectDir, "style.json")); - - const lenSec = Math.max(videoLengthSec(edl), 6); - const targetWords = Math.round(lenSec * WORDS_PER_SEC); - const overlays = (edl.tracks ?? []) - .filter((t) => t.type === "text") - .flatMap((t) => t.clips ?? []) - .map((c) => `${c.start}s: "${c.text}"`) - .join("\n"); - - const { provider, model } = llmConfig(); - console.log(`PHASE drafting narration with ${provider}/${model}`); - - const { text } = await generateText({ - model: resolveModel(), - maxOutputTokens: 2000, - providerOptions: { openai: { reasoningEffort: reasoningEffort() } }, - prompt: [ - "You write spoken voiceover narration for a short-form vertical social video.", - `The cut is ${lenSec.toFixed(1)} seconds long. Write about ${targetWords} words (spoken pace ~2.5 words/sec) — never more than ${Math.round(targetWords * 1.15)}.`, - "Rules:", - "- Return ONLY the narration text. No headings, no quotes, no stage directions, no markdown.", - "- Short, spoken-language sentences. Contractions are good. No em-dash pivots, no 'stop X, start Y' patterns.", - "- Separate beats with a blank line (paragraph = beat).", - "- Open with a hook line that lands in the first two seconds.", - "- Don't read the on-screen text verbatim; complement it.", - "", - "=== CREATOR PROMPT ===", - promptMd || "(none)", - "", - "=== ON-SCREEN TEXT OVERLAYS ===", - overlays || "(none)", - "", - "=== STYLE PROFILE (may be empty) ===", - style || "(none)", - ].join("\n"), - }); - - const narration = text.trim(); - if (!narration) { - console.error("ERROR model returned an empty narration"); - process.exit(2); - } - fs.writeFileSync(path.join(projectDir, "narration.md"), `${narration}\n`); - console.log(`DONE ${narration.split(/\s+/).length} words for ${lenSec.toFixed(1)}s`); -} - -main().catch((err) => { - console.error(`ERROR ${err?.stack || err}`); - process.exit(1); -}); diff --git a/app/src/main/audio-sources.ts b/app/src/main/audio-sources.ts deleted file mode 100644 index a971b00..0000000 --- a/app/src/main/audio-sources.ts +++ /dev/null @@ -1,54 +0,0 @@ -/** - * Provider registry for "add audio from URL". Everything downloads through - * yt-dlp (which handles extraction for thousands of sites plus plain media - * URLs), so supporting a new platform is usually just a new matcher here — - * the registry exists to keep an explicit allowlist instead of accepting any - * link the extractor happens to know. - * - * Deliberately NOT supportable: DRM-protected sources (Spotify, Apple Music). - */ - -export interface AudioSource { - id: string; - label: string; - matches: (url: URL) => boolean; -} - -const SOURCES: AudioSource[] = [ - { - id: "soundcloud", - label: "SoundCloud", - matches: (u) => /(^|\.)soundcloud\.com$|(^|\.)snd\.sc$/.test(u.hostname), - }, - { - id: "direct", - label: "Direct audio URL", - matches: (u) => /\.(mp3|wav|m4a|aac|ogg|flac)$/i.test(u.pathname), - }, - // Future: { id: "youtube", matches: hostname youtube.com/youtu.be } — the - // download path already handles it; add the matcher when we decide to. -]; - -export type ResolvedAudioSource = - | { ok: true; id: string; label: string; url: string } - | { ok: false; error: string }; - -export function resolveAudioSource(raw: string): ResolvedAudioSource { - let url: URL; - try { - url = new URL(raw.trim()); - } catch { - return { ok: false, error: "That doesn't look like a URL." }; - } - if (url.protocol !== "https:" && url.protocol !== "http:") { - return { ok: false, error: "Only http(s) links are supported." }; - } - const source = SOURCES.find((s) => s.matches(url)); - if (!source) { - return { - ok: false, - error: "Unsupported source — paste a SoundCloud link or a direct audio file URL.", - }; - } - return { ok: true, id: source.id, label: source.label, url: url.toString() }; -} diff --git a/app/src/main/index.ts b/app/src/main/index.ts index ae65ade..3088e9b 100644 --- a/app/src/main/index.ts +++ b/app/src/main/index.ts @@ -1,106 +1,37 @@ -import { spawn } from "node:child_process"; +import { type ChildProcessWithoutNullStreams, spawn } from "node:child_process"; import { - copyFileSync, createReadStream, existsSync, type FSWatcher, mkdirSync, - readdirSync, readFileSync, rmSync, statSync, watch, writeFileSync, } from "node:fs"; -import { mkdtempSync } from "node:fs"; -import { tmpdir } from "node:os"; -import { basename, dirname, extname, join, normalize, sep } from "node:path"; +import { homedir } from "node:os"; +import { join, normalize, sep } from "node:path"; +import { createInterface } from "node:readline"; import { Readable } from "node:stream"; -import { resolveAudioSource } from "./audio-sources"; import { app, BrowserWindow, dialog, ipcMain, type IpcMainInvokeEvent, nativeImage, protocol, shell } from "electron"; -import ffmpegPath from "ffmpeg-static"; -import { - type Benchmarks, - durationSeconds, - type Edl, - parseBenchmarks, - parseEdl, - parseEdlOrThrow, - parseMeta, - parseStyleProfile, - type StyleProfile, -} from "@reel/edl"; // In dev (electron-vite) __dirname is /app/out/main, so the repo root is // three levels up. Allow an override for packaged/other layouts. -const REPO_ROOT = process.env["REEL_ROOT"] ?? join(__dirname, "..", "..", ".."); +const REPO_ROOT = process.env["KEEPER_ROOT"] ?? join(__dirname, "..", "..", ".."); const ICON_PATH = join(REPO_ROOT, "app", "resources", "icon.png"); const SCRIPTS_DIR = join(REPO_ROOT, "app", "scripts"); -const RENDER_SCRIPT = join(SCRIPTS_DIR, "render.mjs"); -const ANALYZE_SCRIPT = join(SCRIPTS_DIR, "analyze.mjs"); -const GENERATE_LLM_SCRIPT = join(SCRIPTS_DIR, "generate-llm.mjs"); -const CRITIQUE_LLM_SCRIPT = join(SCRIPTS_DIR, "critique-llm.mjs"); -const AUTOTUNE_LLM_SCRIPT = join(SCRIPTS_DIR, "autotune-llm.mjs"); -const TRANSCRIBE_SCRIPT = join(SCRIPTS_DIR, "transcribe.mjs"); - -// Provider-agnostic LLM config (mirrors app/scripts/llm.mjs) so the UI can show -// the active model and Generate can route to the LLM path vs the offline baseline. -function llmInfo(): { provider: string; model: string; configured: boolean } { - const provider = (process.env["APERTURE_LLM_PROVIDER"] || "openai").toLowerCase(); - const model = process.env["APERTURE_LLM_MODEL"] || "gpt-5.5"; - const baseURL = process.env["APERTURE_LLM_BASE_URL"]; - const apiKey = - process.env["APERTURE_LLM_API_KEY"] || - (provider === "anthropic" ? process.env["ANTHROPIC_API_KEY"] : process.env["OPENAI_API_KEY"]) || - process.env["OPENAI_API_KEY"] || - process.env["ANTHROPIC_API_KEY"]; - const configured = provider === "openai-compatible" ? Boolean(baseURL || apiKey) : Boolean(apiKey); - return { provider, model, configured }; -} -const EXTRACT_FRAMES_SCRIPT = join(SCRIPTS_DIR, "extract-frames.mjs"); -const WRITE_NARRATION_SCRIPT = join(SCRIPTS_DIR, "write-narration.mjs"); -const TTS_SCRIPT = join(SCRIPTS_DIR, "tts.mjs"); -const ANALYZE_STYLE_SCRIPT = join(SCRIPTS_DIR, "analyze-style.mjs"); -const ANALYZE_COLLECTION_SCRIPT = join(SCRIPTS_DIR, "analyze-collection.mjs"); -const ANALYZE_BENCHMARKS_SCRIPT = join(SCRIPTS_DIR, "analyze-benchmarks.mjs"); -const AUTOTUNE_SCRIPT = join(SCRIPTS_DIR, "autotune.mjs"); -const BUNDLED_MUSIC_DIR = join(REPO_ROOT, "app", "resources", "music"); - -const VIDEO_EXT = new Set([".mp4", ".mov", ".webm", ".m4v"]); -const AUDIO_EXT = new Set([".mp3", ".wav", ".m4a", ".aac", ".ogg"]); -const IMAGE_EXT = new Set([".png", ".jpg", ".jpeg", ".webp", ".gif"]); - -function assetKindFor(file: string): "video" | "audio" | "image" | null { - const ext = extname(file).toLowerCase(); - if (VIDEO_EXT.has(ext)) return "video"; - if (AUDIO_EXT.has(ext)) return "audio"; - if (IMAGE_EXT.has(ext)) return "image"; - return null; -} - -/** Resolve + guard a path so it can never escape the given root. */ -function safePath(root: string, rel: string[]): string { - const base = normalize(root); - const file = normalize(join(base, ...rel)); - // Compare against root + separator so a sibling like "-evil" can't pass. - if (file !== base && !file.startsWith(base + sep)) throw new Error("path escapes storage dir"); - return file; -} - -function safeProjectPath(slug: string, ...rel: string[]): string { - return safePath(PROJECTS_DIR, [slug, ...rel]); -} - -function safeStylePath(id: string, ...rel: string[]): string { - return safePath(STYLES_DIR, [id, ...rel]); -} +const IMPORT_SCRIPT = join(SCRIPTS_DIR, "import.mjs"); +const REPROCESS_SCRIPT = join(SCRIPTS_DIR, "reprocess.mjs"); +const JUDGE_SCRIPT = join(SCRIPTS_DIR, "judge-llm.mjs"); +const EXPORT_SCRIPT = join(SCRIPTS_DIR, "export.mjs"); +const CATALOG_SERVICE_SCRIPT = join(SCRIPTS_DIR, "catalog-service.mjs"); // Captured so dialogs can parent to the window. let mainWindow: BrowserWindow | null = null; // The macOS menu bar / dock tooltip / About panel use app.name, which defaults -// to "Electron" in dev. Set it before the default menu is built. (Packaging -// later should also set productName in the builder config.) +// to "Electron" in dev. Set it before the default menu is built. app.setName("Keeper"); // Load a local, gitignored env file (KEY=VALUE) so secrets like OPENAI_API_KEY @@ -127,27 +58,21 @@ function loadLocalEnv(): void { } loadLocalEnv(); -// ---- Persistent app settings (hardware acceleration, storage location, etc.) ---- -export type ExportFps = "project" | "24" | "30" | "60"; -export type ExportResolution = "project" | "1080" | "720"; -export type ExportCompression = "social" | "high" | "max"; +// ---- Persistent app settings ------------------------------------------------ export type ReasoningEffort = "low" | "medium" | "high"; interface AppSettings { + /** Hardware-accelerated video decode (Chromium switch; needs restart). */ hwDecode: boolean; - hwEncode: boolean; - /** User-chosen root folder for projects + styles (default ~/Documents/Keeper). */ - homeDir?: string; - /** Export defaults applied by the render pipeline. */ - exportFps: ExportFps; - exportResolution: ExportResolution; - exportCompression: ExportCompression; - /** Agent preferences (env vars from .env.local always win). */ + /** User-chosen library root (default ~/Pictures/Keeper). Applies on restart. */ + libraryDir?: string; + /** AI preferences (env vars from .env.local always win). */ agentModel: string; agentApiKey?: string; reasoningEffort: ReasoningEffort; - /** ElevenLabs voiceover (env var wins, same as the LLM key). */ - elevenLabsApiKey?: string; - defaultVoiceId?: string; + /** Max items the LLM judge may analyze per run. */ + aiBudget: number; + /** Run the LLM judge automatically after each import. */ + autoJudge: boolean; } const SETTINGS_PATH = join(app.getPath("userData"), "settings.json"); function readSettings(): AppSettings { @@ -161,16 +86,15 @@ function readSettings(): AppSettings { typeof v === "string" && (options as readonly string[]).includes(v) ? (v as T) : dflt; return { hwDecode: Boolean(s.hwDecode), - hwEncode: Boolean(s.hwEncode), - homeDir: typeof s.homeDir === "string" && s.homeDir ? s.homeDir : undefined, - exportFps: oneOf(s.exportFps, ["project", "24", "30", "60"] as const, "project"), - exportResolution: oneOf(s.exportResolution, ["project", "1080", "720"] as const, "project"), - exportCompression: oneOf(s.exportCompression, ["social", "high", "max"] as const, "social"), + libraryDir: typeof s.libraryDir === "string" && s.libraryDir ? s.libraryDir : undefined, agentModel: typeof s.agentModel === "string" && s.agentModel ? s.agentModel : "gpt-5.5", agentApiKey: typeof s.agentApiKey === "string" && s.agentApiKey ? s.agentApiKey : undefined, reasoningEffort: oneOf(s.reasoningEffort, ["low", "medium", "high"] as const, "low"), - elevenLabsApiKey: typeof s.elevenLabsApiKey === "string" && s.elevenLabsApiKey ? s.elevenLabsApiKey : undefined, - defaultVoiceId: typeof s.defaultVoiceId === "string" && s.defaultVoiceId ? s.defaultVoiceId : undefined, + aiBudget: + typeof s.aiBudget === "number" && Number.isFinite(s.aiBudget) && s.aiBudget > 0 + ? Math.min(5000, Math.round(s.aiBudget)) + : 200, + autoJudge: Boolean(s.autoJudge), }; } function writeSettings(patch: Partial): AppSettings { @@ -184,54 +108,71 @@ function writeSettings(patch: Partial): AppSettings { return next; } -// Agent preferences flow to the LLM layer via the same env vars .env.local uses. +// AI preferences flow to the LLM layer via the same env vars .env.local uses. // Anything explicitly set in the environment (shell or .env.local) stays // authoritative; settings only fill the gaps. `envLocked` is captured once at // startup, before the first injection. const envLocked = { - provider: "APERTURE_LLM_PROVIDER" in process.env, - model: "APERTURE_LLM_MODEL" in process.env, + provider: "KEEPER_LLM_PROVIDER" in process.env, + model: "KEEPER_LLM_MODEL" in process.env, apiKey: Boolean( - process.env["APERTURE_LLM_API_KEY"] || process.env["OPENAI_API_KEY"] || process.env["ANTHROPIC_API_KEY"], + process.env["KEEPER_LLM_API_KEY"] || process.env["OPENAI_API_KEY"] || process.env["ANTHROPIC_API_KEY"], ), - effort: "APERTURE_REASONING_EFFORT" in process.env, - elevenLabsKey: Boolean(process.env["ELEVENLABS_API_KEY"]), + effort: "KEEPER_REASONING_EFFORT" in process.env, }; function applyAgentEnv(s: AppSettings): void { if (!envLocked.model) { - process.env["APERTURE_LLM_MODEL"] = s.agentModel; + process.env["KEEPER_LLM_MODEL"] = s.agentModel; if (!envLocked.provider) { - process.env["APERTURE_LLM_PROVIDER"] = s.agentModel.startsWith("claude") ? "anthropic" : "openai"; + process.env["KEEPER_LLM_PROVIDER"] = s.agentModel.startsWith("claude") ? "anthropic" : "openai"; } } if (!envLocked.apiKey) { - if (s.agentApiKey) process.env["APERTURE_LLM_API_KEY"] = s.agentApiKey; - else delete process.env["APERTURE_LLM_API_KEY"]; - } - if (!envLocked.effort) process.env["APERTURE_REASONING_EFFORT"] = s.reasoningEffort; - if (!envLocked.elevenLabsKey) { - if (s.elevenLabsApiKey) process.env["ELEVENLABS_API_KEY"] = s.elevenLabsApiKey; - else delete process.env["ELEVENLABS_API_KEY"]; + if (s.agentApiKey) process.env["KEEPER_LLM_API_KEY"] = s.agentApiKey; + else delete process.env["KEEPER_LLM_API_KEY"]; } + if (!envLocked.effort) process.env["KEEPER_REASONING_EFFORT"] = s.reasoningEffort; + process.env["KEEPER_AI_BUDGET"] = String(s.aiBudget); } applyAgentEnv(readSettings()); -// User-owned storage (Screen Studio style): projects + styles live under the -// user's home folder, not the repo/app bundle. Resolution: env override (dev) -> -// user-picked folder (settings) -> ~/Documents/Keeper. Resolved once at startup. -const APP_HOME = - process.env["APERTURE_HOME"] ?? readSettings().homeDir ?? join(app.getPath("documents"), "Keeper"); -const PROJECTS_DIR = process.env["REEL_PROJECTS_DIR"] ?? join(APP_HOME, "projects"); -const STYLES_DIR = process.env["APERTURE_STYLES_DIR"] ?? join(APP_HOME, "styles"); +// Provider-agnostic LLM config (mirrors app/scripts/llm.mjs) so the UI can show +// the active model and the judge can refuse cleanly when unconfigured. +function llmInfo(): { provider: string; model: string; configured: boolean } { + const provider = (process.env["KEEPER_LLM_PROVIDER"] || "openai").toLowerCase(); + const model = process.env["KEEPER_LLM_MODEL"] || "gpt-5.5"; + const baseURL = process.env["KEEPER_LLM_BASE_URL"]; + const apiKey = + process.env["KEEPER_LLM_API_KEY"] || + (provider === "anthropic" ? process.env["ANTHROPIC_API_KEY"] : process.env["OPENAI_API_KEY"]) || + process.env["OPENAI_API_KEY"] || + process.env["ANTHROPIC_API_KEY"]; + const configured = provider === "openai-compatible" ? Boolean(baseURL || apiKey) : Boolean(apiKey); + return { provider, model, configured }; +} + +// User-owned storage: the library lives under the user's home folder, never the +// repo/app bundle. Resolution: env override (dev) -> user-picked folder +// (settings) -> ~/Pictures/Keeper. Resolved once at startup. +const LIBRARY_HOME = + process.env["KEEPER_LIBRARY_DIR"] ?? readSettings().libraryDir ?? join(homedir(), "Pictures", "Keeper"); try { - mkdirSync(PROJECTS_DIR, { recursive: true }); - mkdirSync(STYLES_DIR, { recursive: true }); + mkdirSync(join(LIBRARY_HOME, "library"), { recursive: true }); + mkdirSync(join(LIBRARY_HOME, ".keeper"), { recursive: true }); } catch { // directories are best-effort at startup } -// Spawned scripts (render/analyze/generate/...) inherit these to find the same dirs. -process.env["APERTURE_PROJECTS_DIR"] = PROJECTS_DIR; -process.env["APERTURE_STYLES_DIR"] = STYLES_DIR; +// Spawned scripts (import/judge/export/...) inherit this to find the same library. +process.env["KEEPER_LIBRARY_DIR"] = LIBRARY_HOME; + +/** Resolve + guard a library-relative path so it can never escape the home. */ +function safeHomePath(...rel: string[]): string { + const base = normalize(LIBRARY_HOME); + const file = normalize(join(base, ...rel)); + // Compare against root + separator so a sibling like "-evil" can't pass. + if (file !== base && !file.startsWith(base + sep)) throw new Error("path escapes library"); + return file; +} // Hardware-accelerated video decode (playback) is a Chromium switch that must be // set before the app is ready, so toggling it needs a restart. @@ -240,11 +181,11 @@ if (readSettings().hwDecode) { app.commandLine.appendSwitch("enable-features", "PlatformHEVCDecoderSupport"); } -// Serve project media to the sandboxed renderer (the Remotion Player can't read -// file:// from an http origin). reel-asset:/// -> projects// +// Serve library media to the sandboxed renderer (it can't read file:// from an +// http origin). keeper-asset://home/ -> / protocol.registerSchemesAsPrivileged([ { - scheme: "reel-asset", + scheme: "keeper-asset", privileges: { standard: true, secure: true, supportFetchAPI: true, stream: true, bypassCSP: true }, }, ]); @@ -259,16 +200,8 @@ function mimeFor(file: string): string { return "video/quicktime"; case ".webm": return "video/webm"; - case ".mp3": - return "audio/mpeg"; - case ".wav": - return "audio/wav"; - case ".m4a": - return "audio/mp4"; - case ".aac": - return "audio/aac"; - case ".ogg": - return "audio/ogg"; + case ".mkv": + return "video/x-matroska"; case ".png": return "image/png"; case ".jpg": @@ -278,627 +211,96 @@ function mimeFor(file: string): string { return "image/webp"; case ".gif": return "image/gif"; + case ".heic": + case ".heif": + return "image/heic"; + case ".tif": + case ".tiff": + return "image/tiff"; default: return "application/octet-stream"; } } -function createWindow(): void { - const win = new BrowserWindow({ - width: 1440, - height: 900, - minWidth: 1100, - minHeight: 700, - show: false, - autoHideMenuBar: true, - backgroundColor: "#16140f", - title: "Keeper", - icon: ICON_PATH, - webPreferences: { - preload: join(__dirname, "../preload/index.js"), - sandbox: false, - }, - }); - - mainWindow = win; - win.on("ready-to-show", () => win.show()); - win.webContents.setWindowOpenHandler((details) => { - void shell.openExternal(details.url); - return { action: "deny" }; - }); - - if (process.env["ELECTRON_RENDERER_URL"]) { - void win.loadURL(process.env["ELECTRON_RENDERER_URL"]); - } else { - void win.loadFile(join(__dirname, "../renderer/index.html")); - } -} - -function readJson(file: string): unknown { - return JSON.parse(readFileSync(file, "utf8")); -} - -function readMeta(slug: string) { - try { - return parseMeta(readJson(safeProjectPath(slug, "meta.json"))); - } catch { - return parseMeta({}); - } -} - -function loadProject(slug: string) { - try { - const dir = safeProjectPath(slug); - const raw = JSON.parse(readFileSync(join(dir, "edl.json"), "utf8")); - const result = parseEdl(raw); - let promptText = ""; - try { - promptText = readFileSync(join(dir, "prompt.md"), "utf8"); - } catch { - // prompt.md is optional - } - return { ...result, slug, dir, promptText, meta: readMeta(slug) }; - } catch (err) { - return { ok: false, errors: [String(err)], slug }; - } -} - -// We write edl.json from two places: the editor (autosave) and the agent/scripts. -// Track our own writes so the file watcher doesn't echo a reload back to the UI -// that just saved (which would clobber in-flight edits / loop). -const lastSelfWrite = new Map(); -let activeWatcher: { slug: string; watcher: FSWatcher } | null = null; - -function writeEdl(slug: string, edl: Edl): { ok: boolean; error?: string } { - try { - const validated = parseEdlOrThrow(edl); - const file = safeProjectPath(slug, "edl.json"); - lastSelfWrite.set(slug, Date.now()); - writeFileSync(file, `${JSON.stringify(validated, null, 2)}\n`); - return { ok: true }; - } catch (err) { - return { ok: false, error: String(err) }; - } +// ---- Catalog service (child Node process owning the SQLite catalog) ---------- +// Keeps the native-free main process out of DB concerns and heavy queries off +// the UI event loop. Line-delimited JSON-RPC over stdio; lazily (re)spawned. +interface ServiceRequest { + resolve: (value: unknown) => void; + reject: (err: Error) => void; + timer: NodeJS.Timeout; } -function touchMeta(slug: string, patch: Partial>): void { - try { - const meta = { ...readMeta(slug), ...patch, updatedAt: new Date().toISOString() }; - writeFileSync(safeProjectPath(slug, "meta.json"), `${JSON.stringify(meta, null, 2)}\n`); - } catch { - // meta is best-effort - } -} +let service: ChildProcessWithoutNullStreams | null = null; +let serviceSeq = 0; +const servicePending = new Map(); +let quitting = false; -function watchProject(slug: string, event: IpcMainInvokeEvent): void { - activeWatcher?.watcher.close(); - activeWatcher = null; - const file = safeProjectPath(slug, "edl.json"); - if (!existsSync(file)) return; - let timer: NodeJS.Timeout | null = null; - const watcher = watch(file, () => { - // Ignore the echo from our own autosave. - if (Date.now() - (lastSelfWrite.get(slug) ?? 0) < 1200) return; - if (timer) clearTimeout(timer); - timer = setTimeout(() => { - if (!event.sender.isDestroyed()) event.sender.send("project:changed", slug); - }, 200); +function startService(): ChildProcessWithoutNullStreams { + const child = spawn("node", [CATALOG_SERVICE_SCRIPT, "--library", LIBRARY_HOME], { + cwd: REPO_ROOT, + env: process.env, + stdio: ["pipe", "pipe", "pipe"], }); - activeWatcher = { slug, watcher }; -} - -export interface ProjectSummary { - slug: string; - title: string; - platform: string; - status: string; - durationSec: number; - assetCount: number; - updatedAt?: string; -} - -function slugify(name: string): string { - return ( - name - .toLowerCase() - .trim() - .replace(/[^a-z0-9]+/g, "-") - .replace(/^-+|-+$/g, "") - .slice(0, 60) || "project" - ); -} - -function listProjects(): ProjectSummary[] { - let entries: string[]; - try { - entries = readdirSync(PROJECTS_DIR); - } catch { - return []; - } - const summaries: ProjectSummary[] = []; - for (const slug of entries) { - const edlFile = join(PROJECTS_DIR, slug, "edl.json"); - if (!existsSync(edlFile)) continue; - let durationSec = 0; - let assetCount = 0; + const rl = createInterface({ input: child.stdout }); + rl.on("line", (line) => { + let msg: { id?: number; ok?: boolean; result?: unknown; error?: string }; try { - const parsed = parseEdl(readJson(edlFile)); - if (parsed.ok && parsed.edl) { - durationSec = durationSeconds(parsed.edl); - assetCount = parsed.edl.assets.length; - } + msg = JSON.parse(line); } catch { - // skip unreadable edl, still list the project - } - const meta = readMeta(slug); - let updatedAt = meta.updatedAt; - try { - if (!updatedAt) updatedAt = statSync(edlFile).mtime.toISOString(); - } catch { - // ignore - } - summaries.push({ - slug, - title: meta.title || slug, - platform: meta.platform, - status: meta.status, - durationSec, - assetCount, - updatedAt, - }); - } - return summaries.sort((a, b) => (b.updatedAt ?? "").localeCompare(a.updatedAt ?? "")); -} - -function createProject(input: { - title: string; - prompt?: string; - platform?: string; - styleProfileId?: string; -}): { ok: boolean; slug?: string; error?: string } { - try { - const base = slugify(input.title); - let slug = base; - let n = 2; - while (existsSync(join(PROJECTS_DIR, slug))) slug = `${base}-${n++}`; - const dir = join(PROJECTS_DIR, slug); - for (const sub of ["assets", "references", "benchmarks", "transcripts", "renders"]) { - mkdirSync(join(dir, sub), { recursive: true }); - } - const now = new Date().toISOString(); - const meta = parseMeta({ - title: input.title.trim() || slug, - createdAt: now, - updatedAt: now, - platform: input.platform ?? "reels", - status: "draft", - styleProfileId: input.styleProfileId, - }); - writeFileSync(join(dir, "meta.json"), `${JSON.stringify(meta, null, 2)}\n`); - writeFileSync(join(dir, "prompt.md"), input.prompt?.trim() ? `${input.prompt.trim()}\n` : `# ${meta.title}\n`); - const emptyEdl = parseEdlOrThrow({ tracks: [{ id: "v", type: "video", clips: [] }] }); - writeFileSync(join(dir, "edl.json"), `${JSON.stringify(emptyEdl, null, 2)}\n`); - return { ok: true, slug }; - } catch (err) { - return { ok: false, error: String(err) }; - } -} - -function deleteProject(slug: string): { ok: boolean; error?: string } { - try { - if (!slug || slug.includes("/") || slug.includes("\\")) return { ok: false, error: "invalid slug" }; - const dir = safeProjectPath(slug); - if (normalize(dir) === normalize(PROJECTS_DIR)) return { ok: false, error: "invalid slug" }; - if (activeWatcher?.slug === slug) { - activeWatcher.watcher.close(); - activeWatcher = null; - } - rmSync(dir, { recursive: true, force: true }); - return { ok: true }; - } catch (err) { - return { ok: false, error: String(err) }; - } -} - -function findFirstVideoSrc(slug: string): string | null { - // Prefer an asset declared in the EDL; fall back to scanning assets/. - try { - const parsed = parseEdl(readJson(safeProjectPath(slug, "edl.json"))); - const asset = parsed.ok ? parsed.edl?.assets.find((a) => a.kind === "video") : undefined; - if (asset) return asset.src; - } catch { - // fall through - } - try { - const file = readdirSync(safeProjectPath(slug, "assets")).find((f) => assetKindFor(f) === "video"); - if (file) return `assets/${file}`; - } catch { - // no assets dir - } - return null; -} - -// Generate (and cache) a poster frame for the project's first video clip. -async function ensureThumbnail(slug: string): Promise { - const thumb = safeProjectPath(slug, ".thumb.jpg"); - const edlFile = safeProjectPath(slug, "edl.json"); - try { - if (existsSync(thumb) && statSync(thumb).mtimeMs >= statSync(edlFile).mtimeMs) { - return `reel-asset://${slug}/.thumb.jpg`; + return; } - } catch { - // regenerate - } - const src = findFirstVideoSrc(slug); - if (!src || !ffmpegPath) return null; - const input = safeProjectPath(slug, src); - const ok = await new Promise((resolve) => { - const child = spawn(ffmpegPath as string, [ - "-y", - "-ss", - "0.8", - "-i", - input, - "-frames:v", - "1", - "-vf", - "scale=360:-1", - thumb, - ]); - child.on("close", (code) => resolve(code === 0)); - child.on("error", () => resolve(false)); + if (typeof msg.id !== "number") return; // readiness banner or notification + const pending = servicePending.get(msg.id); + if (!pending) return; + servicePending.delete(msg.id); + clearTimeout(pending.timer); + if (msg.ok) pending.resolve(msg.result); + else pending.reject(new Error(msg.error ?? "catalog service error")); }); - return ok ? `reel-asset://${slug}/.thumb.jpg` : null; -} - -export interface ImportedAsset { - id: string; - kind: "video" | "audio" | "image"; - src: string; - durationSec?: number; -} - -// Parse a media file's duration out of ffmpeg's stderr banner. ffmpeg-static is -// already a dependency (used by transcribe.mjs); ffprobe isn't bundled. -function probeDurationSec(file: string): Promise { - return new Promise((resolve) => { - if (!ffmpegPath) return resolve(undefined); - const child = spawn(ffmpegPath as string, ["-i", file]); - let err = ""; - child.stderr.on("data", (c: Buffer) => (err += c.toString())); - child.on("close", () => { - const m = err.match(/Duration:\s*(\d+):(\d+):(\d+(?:\.\d+)?)/); - resolve(m ? Number(m[1]) * 3600 + Number(m[2]) * 60 + Number(m[3]) : undefined); - }); - child.on("error", () => resolve(undefined)); + child.stderr.on("data", (chunk: Buffer) => { + console.error(`[catalog-service] ${chunk.toString().trim()}`); }); -} - -function uniqueDest(dir: string, name: string): { name: string; dest: string } { - const ext = extname(name); - const stem = basename(name, ext); - let candidate = name; - let dest = join(dir, candidate); - let i = 2; - while (existsSync(dest)) { - candidate = `${stem}-${i++}${ext}`; - dest = join(dir, candidate); - } - return { name: candidate, dest }; -} - -async function describeAsset(dir: string, name: string): Promise { - const kind = assetKindFor(name); - if (!kind) return null; - const durationSec = kind === "image" ? undefined : await probeDurationSec(join(dir, name)); - const id = basename(name, extname(name)).replace(/[^a-zA-Z0-9_-]+/g, "-"); - return { id, kind, src: `assets/${name}`, durationSec }; -} - -// Background H.264 proxy so the editor scrubs smoothly even for HEVC/.MOV; when -// done, patch the project's edl.json so the preview switches to the proxy. Export -// still uses the original for full quality. -function proxyRel(id: string): string { - return `assets/.proxies/${id}.mp4`; -} -async function generateProxy(slug: string, assetSrc: string, id: string): Promise { - if (!ffmpegPath) return; - const input = safeProjectPath(slug, assetSrc); - const outDir = safeProjectPath(slug, "assets", ".proxies"); - mkdirSync(outDir, { recursive: true }); - const out = join(outDir, `${id}.mp4`); - const ok = await new Promise((resolve) => { - const child = spawn(ffmpegPath as string, [ - "-y", "-i", input, "-an", - "-vf", "scale='min(720,iw)':-2", - "-c:v", "libx264", "-preset", "veryfast", "-pix_fmt", "yuv420p", "-movflags", "+faststart", - out, - ]); - child.on("close", (code) => resolve(code === 0)); - child.on("error", () => resolve(false)); - }); - if (ok) patchAssetProxy(slug, id, proxyRel(id)); -} -// Retry-patch: the renderer's autosave may not have written the new asset yet. -function patchAssetProxy(slug: string, id: string, proxySrc: string, attempt = 0): void { - try { - const file = safeProjectPath(slug, "edl.json"); - const edl = JSON.parse(readFileSync(file, "utf8")); - const asset = (edl.assets ?? []).find((a: { id: string }) => a.id === id); - if (asset && existsSync(safeProjectPath(slug, proxySrc))) { - asset.proxySrc = proxySrc; - // Intentionally do NOT mark lastSelfWrite: we want the watcher to reload - // the editor so the preview picks up the proxy. - writeFileSync(file, `${JSON.stringify(edl, null, 2)}\n`); - return; + child.on("exit", () => { + for (const [, pending] of servicePending) { + clearTimeout(pending.timer); + pending.reject(new Error("catalog service exited")); } - } catch { - // fall through to retry - } - if (attempt < 12) setTimeout(() => patchAssetProxy(slug, id, proxySrc, attempt + 1), 800); -} - -async function importAssets(slug: string, paths: string[]): Promise<{ ok: boolean; assets: ImportedAsset[] }> { - const dir = safeProjectPath(slug, "assets"); - mkdirSync(dir, { recursive: true }); - const added: ImportedAsset[] = []; - for (const p of paths) { - if (!assetKindFor(p)) continue; - const { name, dest } = uniqueDest(dir, basename(p)); - copyFileSync(p, dest); - const desc = await describeAsset(dir, name); - if (desc) { - added.push(desc); - if (desc.kind === "video") void generateProxy(slug, desc.src, desc.id); - } - } - return { ok: true, assets: added }; -} - -async function importAssetBuffer( - slug: string, - filename: string, - data: Uint8Array, -): Promise<{ ok: boolean; assets: ImportedAsset[] }> { - const dir = safeProjectPath(slug, "assets"); - mkdirSync(dir, { recursive: true }); - const { name, dest } = uniqueDest(dir, filename); - writeFileSync(dest, Buffer.from(data)); - const desc = await describeAsset(dir, name); - return { ok: true, assets: desc ? [desc] : [] }; -} - -function listBundledMusic(): string[] { - try { - return readdirSync(BUNDLED_MUSIC_DIR).filter((f) => AUDIO_EXT.has(extname(f).toLowerCase())); - } catch { - return []; - } -} - -async function importBundledMusic( - slug: string, - name: string, -): Promise<{ ok: boolean; assets: ImportedAsset[]; error?: string }> { - const source = join(BUNDLED_MUSIC_DIR, basename(name)); - if (!source.startsWith(normalize(BUNDLED_MUSIC_DIR)) || !existsSync(source)) { - return { ok: false, assets: [], error: "unknown track" }; - } - return importAssets(slug, [source]); -} - -// ---- ElevenLabs voices (list / clone / delete) ---- -export interface VoiceSummary { - id: string; - name: string; - category: string; -} - -const EL_API = "https://api.elevenlabs.io/v1"; - -function elevenLabsKey(): string | undefined { - return process.env["ELEVENLABS_API_KEY"] || undefined; -} - -async function elError(res: Response): Promise { - try { - const body = (await res.json()) as { detail?: { message?: string } | string }; - const detail = typeof body.detail === "string" ? body.detail : body.detail?.message; - return detail || `ElevenLabs HTTP ${res.status}`; - } catch { - return `ElevenLabs HTTP ${res.status}`; - } -} - -async function listVoices(): Promise<{ ok: boolean; voices: VoiceSummary[]; error?: string }> { - const key = elevenLabsKey(); - if (!key) return { ok: false, voices: [], error: "No ElevenLabs API key configured." }; - try { - const res = await fetch(`${EL_API}/voices`, { headers: { "xi-api-key": key } }); - if (!res.ok) return { ok: false, voices: [], error: await elError(res) }; - const data = (await res.json()) as { voices?: { voice_id: string; name: string; category?: string }[] }; - return { - ok: true, - voices: (data.voices ?? []).map((v) => ({ id: v.voice_id, name: v.name, category: v.category ?? "premade" })), - }; - } catch (err) { - return { ok: false, voices: [], error: String(err) }; - } -} - -// Voice samples arrive as file paths and/or an in-memory mic recording (webm). -// ElevenLabs is picky about container formats, so everything is transcoded to -// mp3 with the bundled ffmpeg before upload. -async function transcodeSampleToMp3(input: string, outDir: string, stem: string): Promise { - if (!ffmpegPath) return null; - const out = join(outDir, `${stem}.mp3`); - const ok = await new Promise((resolve) => { - const child = spawn(ffmpegPath as string, ["-y", "-i", input, "-ac", "1", "-b:a", "128k", out]); - child.on("close", (code) => resolve(code === 0)); - child.on("error", () => resolve(false)); + servicePending.clear(); + service = null; + }); + return child; +} + +function callService(method: string, params: Record = {}): Promise { + if (quitting) return Promise.reject(new Error("shutting down")); + if (!service) service = startService(); + const child = service; + const id = ++serviceSeq; + return new Promise((resolve, reject) => { + const timer = setTimeout(() => { + servicePending.delete(id); + reject(new Error(`catalog service timeout (${method})`)); + }, 60_000); + servicePending.set(id, { resolve: resolve as (v: unknown) => void, reject, timer }); + child.stdin.write(`${JSON.stringify({ id, method, params })}\n`); }); - return ok ? out : null; -} - -async function cloneVoice(input: { - name: string; - paths: string[]; - recording?: { name: string; data: Uint8Array }; - consent: boolean; -}): Promise<{ ok: boolean; voiceId?: string; error?: string }> { - const key = elevenLabsKey(); - if (!key) return { ok: false, error: "No ElevenLabs API key configured." }; - if (!input.consent) return { ok: false, error: "Consent is required to clone a voice." }; - const name = input.name.trim(); - if (!name) return { ok: false, error: "Give the voice a name." }; - - const tmp = mkdtempSync(join(tmpdir(), "aperture-voice-")); - try { - const staged: string[] = []; - for (const p of input.paths) { - if (assetKindFor(p) === "audio" || extname(p).toLowerCase() === ".webm") staged.push(p); - } - if (input.recording) { - const raw = join(tmp, input.recording.name); - writeFileSync(raw, Buffer.from(input.recording.data)); - staged.push(raw); - } - if (staged.length === 0) return { ok: false, error: "Add at least one audio sample." }; - - const form = new FormData(); - form.append("name", name); - for (let i = 0; i < staged.length; i++) { - const mp3 = await transcodeSampleToMp3(staged[i], tmp, `sample-${i + 1}`); - if (!mp3) return { ok: false, error: `Could not read sample: ${basename(staged[i])}` }; - form.append("files", new Blob([readFileSync(mp3)], { type: "audio/mpeg" }), basename(mp3)); - } - - const res = await fetch(`${EL_API}/voices/add`, { - method: "POST", - headers: { "xi-api-key": key }, - body: form, - }); - if (!res.ok) return { ok: false, error: await elError(res) }; - const data = (await res.json()) as { voice_id?: string }; - if (!data.voice_id) return { ok: false, error: "ElevenLabs did not return a voice id." }; - return { ok: true, voiceId: data.voice_id }; - } catch (err) { - return { ok: false, error: String(err) }; - } finally { - rmSync(tmp, { recursive: true, force: true }); - } -} - -async function deleteVoice(id: string): Promise<{ ok: boolean; error?: string }> { - const key = elevenLabsKey(); - if (!key) return { ok: false, error: "No ElevenLabs API key configured." }; - if (!/^[a-zA-Z0-9]{8,64}$/.test(id)) return { ok: false, error: "invalid voice id" }; - try { - const res = await fetch(`${EL_API}/voices/${id}`, { method: "DELETE", headers: { "xi-api-key": key } }); - if (!res.ok) return { ok: false, error: await elError(res) }; - return { ok: true }; - } catch (err) { - return { ok: false, error: String(err) }; - } -} - -// ---- Audio from URL (SoundCloud etc.) ---- -// yt-dlp does the extraction; like whisper.cpp it is fetched on first use -// (~35MB universal macOS binary) and cached in userData/bin. -const YTDLP_RELEASE_URL = "https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp_macos"; -const YTDLP_BIN = join(app.getPath("userData"), "bin", "yt-dlp"); - -async function ensureYtDlp(): Promise { - if (existsSync(YTDLP_BIN)) return YTDLP_BIN; - mkdirSync(dirname(YTDLP_BIN), { recursive: true }); - const res = await fetch(YTDLP_RELEASE_URL); - if (!res.ok) throw new Error(`could not fetch the audio downloader (HTTP ${res.status})`); - writeFileSync(YTDLP_BIN, Buffer.from(await res.arrayBuffer()), { mode: 0o755 }); - return YTDLP_BIN; } -async function importAudioFromUrl( - slug: string, - rawUrl: string, - event: IpcMainInvokeEvent, -): Promise<{ ok: boolean; assets: ImportedAsset[]; error?: string }> { - const source = resolveAudioSource(rawUrl); - if (!source.ok) return { ok: false, assets: [], error: source.error }; - const send = (channel: string, value: unknown) => { - if (!event.sender.isDestroyed()) event.sender.send(channel, value); - }; - let tmp: string | null = null; +/** Wrap a service call in the { ok, error? } IPC result shape. */ +async function serviceResult(method: string, params: Record = {}): Promise< + { ok: true; result: T } | { ok: false; error: string } +> { try { - if (!existsSync(YTDLP_BIN)) send("audiourl:phase", "getting the downloader (first run)"); - const bin = await ensureYtDlp(); - tmp = mkdtempSync(join(tmpdir(), "aperture-audio-")); - send("audiourl:phase", `fetching from ${source.label}`); - - const args = [ - "--no-playlist", - "--no-warnings", - "--newline", - "--max-filesize", "200m", - "-f", "bestaudio/best", - "-x", - "--audio-format", "m4a", - "--ffmpeg-location", ffmpegPath as string, - "-o", join(tmp, "%(title).120B.%(ext)s"), - source.url, - ]; - await new Promise((resolve, reject) => { - const child = spawn(bin, args); - let stderr = ""; - child.stdout.on("data", (chunk: Buffer) => { - const m = chunk.toString().match(/\[download\]\s+([\d.]+)%/); - if (m) send("audiourl:progress", Math.round(Number(m[1]))); - }); - child.stderr.on("data", (chunk: Buffer) => (stderr += chunk.toString())); - const timer = setTimeout(() => { - child.kill("SIGKILL"); - reject(new Error("timed out after 5 minutes")); - }, 300_000); - child.on("close", (code) => { - clearTimeout(timer); - if (code === 0) resolve(); - else reject(new Error(stderr.trim().split("\n").pop() || `downloader exited with code ${code}`)); - }); - child.on("error", (err) => { - clearTimeout(timer); - reject(err); - }); - }); - - const produced = readdirSync(tmp).find((f) => assetKindFor(f) === "audio"); - if (!produced) return { ok: false, assets: [], error: "The link didn't yield an audio file." }; - send("audiourl:phase", "importing"); - return await importAssets(slug, [join(tmp, produced)]); + return { ok: true, result: await callService(method, params) }; } catch (err) { - return { ok: false, assets: [], error: err instanceof Error ? err.message : String(err) }; - } finally { - if (tmp) rmSync(tmp, { recursive: true, force: true }); + return { ok: false, error: String(err instanceof Error ? err.message : err) }; } } -function importInto( - slug: string, - sub: string, - paths: string[], -): { ok: boolean; files: string[] } { - const dir = safeProjectPath(slug, sub); - mkdirSync(dir, { recursive: true }); - const files: string[] = []; - for (const p of paths) { - if (assetKindFor(p) !== "video") continue; - const { name, dest } = uniqueDest(dir, basename(p)); - copyFileSync(p, dest); - files.push(name); - } - return { ok: true, files }; -} - -// Spawn a Node script with arbitrary args, streaming its PHASE/PROGRESS/DONE -// protocol back to the renderer on `${channelPrefix}:*` channels. +// ---- Engine script spawning --------------------------------------------------- +// Spawn a Node script, streaming its PHASE/PROGRESS/DONE protocol back to the +// renderer on `${channelPrefix}:*` channels. function runScriptArgs( scriptPath: string, args: string[], @@ -910,12 +312,13 @@ function runScriptArgs( let output = ""; let stderr = ""; child.stdout.on("data", (chunk: Buffer) => { + if (event.sender.isDestroyed()) return; for (const line of chunk.toString().split("\n")) { - const progress = line.match(/PROGRESS (\d+)/); + const progress = line.match(/^PROGRESS (\d+)/); if (progress) event.sender.send(`${channelPrefix}:progress`, Number(progress[1])); - const phase = line.match(/PHASE (.+)/); + const phase = line.match(/^PHASE (.+)/); if (phase) event.sender.send(`${channelPrefix}:phase`, phase[1].trim()); - const done = line.match(/DONE (.+)/); + const done = line.match(/^DONE (.+)/); if (done) output = done[1].trim(); } }); @@ -928,181 +331,90 @@ function runScriptArgs( }); } -function runScript( - scriptPath: string, - slug: string, - event: IpcMainInvokeEvent, - channelPrefix: string, - extraArgs: string[] = [], -): Promise<{ ok: boolean; output?: string; error?: string }> { - // Engine scripts join the slug onto the projects dir themselves, so enforce - // slug shape at this IPC boundary (same rule slugify produces). - if (!/^[a-z0-9][a-z0-9_-]{0,63}$/i.test(slug)) { - return Promise.resolve({ ok: false, error: "invalid project id" }); - } - return runScriptArgs(scriptPath, ["--slug", slug, ...extraArgs], event, channelPrefix); -} - -// ---- Global style library (styles// : sources/, .frames/, profile.json) ---- -export interface StyleSummary { - id: string; - name: string; - clips: number; - analyzed: boolean; - updatedAt?: string; -} - -function listStyles(): StyleSummary[] { - let ids: string[]; +// ---- Stamp watcher: CLI/agent writes -> renderer refresh ---------------------- +// Pipeline scripts and agent CLIs touch /.keeper/.stamp after writes; the +// app's own service never does (it would loop). Debounced push to the renderer. +let stampWatcher: FSWatcher | null = null; +function watchStamp(): void { + const stampFile = join(LIBRARY_HOME, ".keeper", ".stamp"); try { - ids = readdirSync(STYLES_DIR); + if (!existsSync(stampFile)) writeFileSync(stampFile, "0"); } catch { - return []; - } - const out: StyleSummary[] = []; - for (const id of ids) { - const dir = join(STYLES_DIR, id); - try { - if (!statSync(dir).isDirectory()) continue; - } catch { - continue; - } - let name = id; - let analyzed = false; - let updatedAt: string | undefined; - try { - const p = JSON.parse(readFileSync(join(dir, "profile.json"), "utf8")); - name = p.name || id; - analyzed = Boolean(p.styleGuide || (p.exemplars?.length ?? 0) > 0 || p.palette?.length); - updatedAt = p.source?.generatedAt; - } catch { - // no profile yet - } - let clips = 0; - try { - clips = readdirSync(join(dir, "sources")).filter((f) => assetKindFor(f) === "video").length; - } catch { - // no sources yet - } - out.push({ id, name, clips, analyzed, updatedAt }); + return; } - return out.sort((a, b) => (b.updatedAt ?? "").localeCompare(a.updatedAt ?? "")); -} - -function createStyle(name: string): { ok: boolean; id?: string; error?: string } { - try { - const base = slugify(name); - let id = base; - let n = 2; - while (existsSync(join(STYLES_DIR, id))) id = `${base}-${n++}`; - mkdirSync(join(STYLES_DIR, id, "sources"), { recursive: true }); - writeFileSync( - join(STYLES_DIR, id, "profile.json"), - `${JSON.stringify({ id, name: name.trim() || id, palette: [], exemplars: [], do: [], avoid: [] }, null, 2)}\n`, - ); - return { ok: true, id }; - } catch (err) { - return { ok: false, error: String(err) }; - } -} - -function importStyleSources(id: string, paths: string[]): { ok: boolean; files: string[] } { - const dir = safeStylePath(id, "sources"); - mkdirSync(dir, { recursive: true }); - const files: string[] = []; - for (const p of paths) { - if (assetKindFor(p) !== "video") continue; - const { name, dest } = uniqueDest(dir, basename(p)); - copyFileSync(p, dest); - files.push(name); - } - return { ok: true, files }; -} - -// Open a native picker (files or a whole folder) and import the chosen videos. -async function addStyleSourcesFromDialog( - id: string, - mode: "files" | "folder", -): Promise<{ ok: boolean; files: string[]; error?: string }> { + let timer: NodeJS.Timeout | null = null; try { - const win = mainWindow ?? BrowserWindow.getFocusedWindow(); - const opts: Electron.OpenDialogOptions = { - title: mode === "folder" ? "Choose a folder of reference videos" : "Choose reference videos", - properties: mode === "folder" ? ["openDirectory"] : ["openFile", "multiSelections"], - filters: mode === "files" ? [{ name: "Video", extensions: ["mp4", "mov", "webm", "m4v"] }] : undefined, - }; - const result = win ? await dialog.showOpenDialog(win, opts) : await dialog.showOpenDialog(opts); - if (result.canceled || result.filePaths.length === 0) return { ok: true, files: [] }; - let paths = result.filePaths; - if (mode === "folder") { - const folder = result.filePaths[0]; - paths = readdirSync(folder) - .filter((f) => assetKindFor(f) === "video") - .map((f) => join(folder, f)); - } - return importStyleSources(id, paths); - } catch (err) { - return { ok: false, files: [], error: String(err) }; + stampWatcher = watch(stampFile, () => { + if (timer) clearTimeout(timer); + timer = setTimeout(() => { + void callService("refreshEmbeddings").catch(() => undefined); + if (mainWindow && !mainWindow.isDestroyed()) { + mainWindow.webContents.send("library:changed"); + } + }, 500); + }); + } catch { + // stamp watching is best-effort } } -// Open the native picker and create a new style in one step, named after the -// chosen folder (no name prompt needed — window.prompt isn't supported in Electron). -async function createStyleFromDialog( - mode: "files" | "folder", -): Promise<{ ok: boolean; id?: string; name?: string; files?: string[]; canceled?: boolean; error?: string }> { - try { - const win = mainWindow ?? BrowserWindow.getFocusedWindow(); - const opts: Electron.OpenDialogOptions = { - title: mode === "folder" ? "Choose a folder of reference videos" : "Choose reference videos", - properties: mode === "folder" ? ["openDirectory"] : ["openFile", "multiSelections"], - filters: mode === "files" ? [{ name: "Video", extensions: ["mp4", "mov", "webm", "m4v"] }] : undefined, - }; - const result = win ? await dialog.showOpenDialog(win, opts) : await dialog.showOpenDialog(opts); - if (result.canceled || result.filePaths.length === 0) return { ok: true, canceled: true }; +function createWindow(): void { + const win = new BrowserWindow({ + width: 1440, + height: 900, + minWidth: 1100, + minHeight: 700, + show: false, + autoHideMenuBar: true, + backgroundColor: "#111013", + title: "Keeper", + icon: ICON_PATH, + webPreferences: { + preload: join(__dirname, "../preload/index.js"), + sandbox: false, + }, + }); - let paths: string[]; - let name: string; - if (mode === "folder") { - const folder = result.filePaths[0]; - name = basename(folder) || "My Style"; - paths = readdirSync(folder) - .filter((f) => assetKindFor(f) === "video") - .map((f) => join(folder, f)); - } else { - paths = result.filePaths; - name = basename(dirname(paths[0])) || "My Style"; - } - if (paths.length === 0) return { ok: false, error: "No videos found in the selection." }; + mainWindow = win; + win.on("ready-to-show", () => win.show()); + win.webContents.setWindowOpenHandler((details) => { + void shell.openExternal(details.url); + return { action: "deny" }; + }); - const created = createStyle(name); - if (!created.ok || !created.id) return created; - const imp = importStyleSources(created.id, paths); - return { ok: true, id: created.id, name, files: imp.files }; - } catch (err) { - return { ok: false, error: String(err) }; + if (process.env["ELECTRON_RENDERER_URL"]) { + void win.loadURL(process.env["ELECTRON_RENDERER_URL"]); + } else { + void win.loadFile(join(__dirname, "../renderer/index.html")); } } -function getStyle(id: string): StyleProfile | null { - try { - return parseStyleProfile(JSON.parse(readFileSync(safeStylePath(id, "profile.json"), "utf8"))); - } catch { - return null; - } +// ---- Dialog helpers ------------------------------------------------------------ +async function pickPaths(mode: "files" | "folder"): Promise { + const win = mainWindow ?? BrowserWindow.getFocusedWindow(); + const opts: Electron.OpenDialogOptions = { + title: mode === "folder" ? "Choose a folder to import" : "Choose photos and videos", + properties: + mode === "folder" ? ["openDirectory"] : ["openFile", "multiSelections", "treatPackageAsDirectory"], + filters: + mode === "files" + ? [ + { + name: "Media", + extensions: [ + "jpg", "jpeg", "png", "heic", "heif", "webp", "gif", "tif", "tiff", + "cr2", "cr3", "nef", "nrw", "arw", "dng", "orf", "rw2", "raf", "srw", "pef", + "mp4", "m4v", "mov", "avi", "mkv", "webm", + ], + }, + ] + : undefined, + }; + const result = win ? await dialog.showOpenDialog(win, opts) : await dialog.showOpenDialog(opts); + return result.canceled ? [] : result.filePaths; } -function deleteStyle(id: string): { ok: boolean; error?: string } { - try { - if (!id || id.includes("/") || id.includes("\\")) return { ok: false, error: "invalid id" }; - const dir = safeStylePath(id); - if (normalize(dir) === normalize(STYLES_DIR)) return { ok: false, error: "invalid id" }; - rmSync(dir, { recursive: true, force: true }); - return { ok: true }; - } catch (err) { - return { ok: false, error: String(err) }; - } -} +const FLAGS = new Set(["pick", "reject", "unrated"]); app.whenReady().then(() => { // macOS ignores the BrowserWindow `icon`; set the dock icon so the app shows @@ -1112,13 +424,13 @@ app.whenReady().then(() => { if (!dockIcon.isEmpty()) app.dock.setIcon(dockIcon); } - protocol.handle("reel-asset", (request) => { + protocol.handle("keeper-asset", (request) => { const url = new URL(request.url); - const slug = url.hostname; + if (url.hostname !== "home") return new Response("Forbidden", { status: 403 }); const rel = decodeURIComponent(url.pathname).replace(/^\/+/, ""); let file: string; try { - file = safeProjectPath(slug, rel); + file = safeHomePath(rel); } catch { return new Response("Forbidden", { status: 403 }); } @@ -1133,8 +445,8 @@ app.whenReady().then(() => { const mime = mimeFor(file); const range = request.headers.get("Range"); - // Stream from disk with byte-range support so