diff --git a/README.md b/README.md index 2816c66..d57bd5a 100644 --- a/README.md +++ b/README.md @@ -94,7 +94,7 @@ folder, so the set stays small as coverage grows. | Skill | Category | Install name | Works on | |---|---|---|---| | Bookmarks | browser | `organizing-bookmarks` | Every major browser. The same page saved four times under slightly different links, bookmarks that no longer go anywhere, folders holding one thing. | -| Folders | files | `organizing-folders` | Downloads, your Desktop, or any folder you name. The same file saved twice, things you have not opened in months, and the huge items you forgot were there. | +| Folders | files | `organizing-folders` | Downloads, your Desktop, a cloud drive, or any folder you name. The same file saved twice, things you have not opened in months, the huge items you forgot were there, and the conflicting copies your devices left behind. | | Notes | notes | `organizing-notes` | An Obsidian vault or any folder of notes. Notes you started and never finished, two versions of the same list, tags that all mean the same thing. | ### Which one should I use? diff --git a/references/app-data-locations.md b/references/app-data-locations.md index bb7189c..56e7357 100644 --- a/references/app-data-locations.md +++ b/references/app-data-locations.md @@ -94,6 +94,29 @@ Bookmarks live in `places.sqlite`, table `moz_bookmarks` joined to `moz_places`. | Downloads | `~/Downloads` | `%USERPROFILE%\Downloads` | | Documents | `~/Documents` | `%USERPROFILE%\Documents` | +### Cloud drives + +Sync folders are ordinary directories, so the same skill covers them. `scan_tree.py` +reports which provider a path belongs to. + +| Provider | Where it usually lives | +|---|---| +| iCloud Drive | `~/Library/Mobile Documents/`, and `~/iCloud Drive` | +| Dropbox | `~/Dropbox` | +| OneDrive | `~/OneDrive`, `%USERPROFILE%\OneDrive` | +| Google Drive | `~/Google Drive`, `~/Library/CloudStorage/GoogleDrive-*` | +| Box | `~/Box` | +| Nextcloud | `~/Nextcloud` | + +Two hazards, both handled by `scan_tree.py`: + +- **Placeholders.** Evicted files are left as stubs. Acting on one destroys its + content. Reported as `placeholders_not_downloaded`. +- **Conflicted copies.** Every client names them differently and none clean up after + themselves. Reported as `conflicts`, each paired with the original it competes with + and flagged `same_content` so identical copies can be separated from ones holding + work that exists nowhere else. + - **Permission:** Full Disk Access on macOS for all three. Nothing on Linux or Windows. - **Parser:** `scripts/scan_tree.py` - **Deletion:** never `rm`. Use `python3 scripts/platforms.py trash `, which diff --git a/scripts/scan_tree.py b/scripts/scan_tree.py index a08772b..5187a95 100755 --- a/scripts/scan_tree.py +++ b/scripts/scan_tree.py @@ -11,7 +11,7 @@ Refuses to touch anything on the denylist. Never follows symlinks out of root. Stdlib only. """ -import argparse, hashlib, json, os, time +import argparse, hashlib, json, os, re, time from collections import defaultdict DENY_NAMES = {".ssh", ".gnupg", ".aws", ".config", ".kube", "Library"} @@ -30,6 +30,75 @@ EXT_TO_CAT = {e: c for c, exts in CATEGORIES.items() for e in exts} +# Every sync client marks a conflict differently, and none of them clean up after +# themselves. These are the patterns they actually write. +CONFLICT_PATTERNS = [ + (re.compile(r"\(([^()]*?)'s conflicted copy \d{4}-\d{2}-\d{2}\)", re.I), "Dropbox"), + (re.compile(r"\bconflicted copy\b", re.I), "Dropbox"), + (re.compile(r"-[A-Za-z0-9]+'s conflicted copy", re.I), "Dropbox"), + (re.compile(r"\((?:[^()]*\s)?conflicted copy(?:\s[^()]*)?\)", re.I), "Dropbox"), + (re.compile(r"-\s?[A-Za-z0-9._-]+'s\s+conflict", re.I), "Dropbox"), + (re.compile(r"\(\d+\)_conflict", re.I), "Nextcloud"), + (re.compile(r"_conflict-\d{8}-\d{6}", re.I), "Nextcloud"), + (re.compile(r"\bconflicted version\b", re.I), "Box"), + (re.compile(r"-\s?[A-Za-z0-9._-]+\s?\(\d+\)(?=\.\w+$)"), None), + (re.compile(r"\(Case Conflict\)", re.I), "Dropbox"), + (re.compile(r"\bconflicted\b", re.I), None), + # OneDrive appends the machine name: "Budget-DESKTOP-A1B2C3.xlsx" + (re.compile(r"-(?:DESKTOP|LAPTOP|MacBook|MBP|PC)-[A-Z0-9]{4,}(?=\.[^.]*$)", re.I), "OneDrive"), + # Google Drive: "Budget (1).xlsx" alongside "Budget.xlsx" is handled by pairing, + # not by name alone, because (1) is also just a second download. +] + + +def conflict_of(name): + """Return the sync client that produced this name, or None. + + Named conflicts only. A trailing "(1)" is ambiguous on its own and is left to + duplicate detection, which compares content rather than guessing from a name. + """ + for pattern, client in CONFLICT_PATTERNS: + if pattern.search(name): + return client or "sync client" + return None + + +def base_name_of(name): + """Strip a conflict marker to recover the name the file is competing with. + + Patterns are applied to the whole filename, so any that anchor on the + extension still match, and the leftover separator before the dot is cleaned + up afterwards. + """ + stripped = name + for pattern, _ in CONFLICT_PATTERNS: + stripped = pattern.sub("", stripped) + stem, ext = os.path.splitext(stripped) + stem = re.sub(r"[\s._-]+$", "", stem) + return stem + ext + + +CLOUD_ROOTS = { + "iCloud Drive": ("Mobile Documents", "iCloud"), + "Dropbox": ("Dropbox",), + "OneDrive": ("OneDrive",), + "Google Drive": ("Google Drive", "GoogleDrive", "CloudStorage"), + "Box": ("Box",), + "Nextcloud": ("Nextcloud",), + "Sync": ("Sync",), +} + + +def cloud_provider(root): + """Which sync service, if any, this path lives inside.""" + real = os.path.realpath(os.path.expanduser(root)) + for name, markers in CLOUD_ROOTS.items(): + for marker in markers: + if os.sep + marker in real + os.sep or real.endswith(os.sep + marker): + return name + return None + + def denied(path, root): """True if path escapes root or touches a denylisted location.""" real_root = os.path.realpath(root) @@ -82,7 +151,9 @@ def scan(root, max_depth, min_size_mb): skipped += 1 continue ext = os.path.splitext(name)[1].lower() + conflict = conflict_of(name) files.append({ + "conflict": conflict, "path": os.path.relpath(full, root), "name": name, "ext": ext, @@ -90,6 +161,10 @@ def scan(root, max_depth, min_size_mb): "size_bytes": st.st_size, "size_mb": round(st.st_size / 1e6, 2), "age_days": int((now - st.st_mtime) / 86400), + # A zero-byte file that is not meant to be empty, or an explicit + # placeholder extension, means the content is not on this machine. + "placeholder": ext in (".icloud", ".cloud") + or (st.st_size == 0 and ext not in ("", ".gitkeep", ".keep")), }) by_cat, by_age = defaultdict(lambda: {"count": 0, "size_mb": 0.0}), defaultdict(int) @@ -114,8 +189,34 @@ def scan(root, max_depth, min_size_mb): pass dupes = {k: v for k, v in groups.items() if len(v) > 1} + # Pair each conflicted copy with the original it is competing against. + by_name = {} + for f in files: + by_name.setdefault(os.path.join(os.path.dirname(f["path"]), f["name"]), f) + conflicts = [] + for f in files: + if not f["conflict"]: + continue + original = os.path.join(os.path.dirname(f["path"]), base_name_of(f["name"])) + match = by_name.get(original) + conflicts.append({ + "copy": f["path"], + "client": f["conflict"], + "size_mb": f["size_mb"], + "age_days": f["age_days"], + "original": match["path"] if match else None, + "same_content": bool(match) and f["size_bytes"] == match["size_bytes"], + }) + + placeholders = [f["path"] for f in files if f["placeholder"]] + return { "root": root, + "cloud_provider": cloud_provider(root), + "conflicted_copies": len(conflicts), + "conflicts": conflicts[:30], + "placeholders_not_downloaded": len(placeholders), + "placeholder_examples": placeholders[:10], "total_files": len(files), "total_size_mb": round(sum(f["size_mb"] for f in files), 2), "skipped_denied_or_symlink": skipped, diff --git a/site/index.html b/site/index.html index a1b617f..b793edc 100644 --- a/site/index.html +++ b/site/index.html @@ -18,10 +18,12 @@ --warn-text:#9a3412; --warn-bg:rgb(234 88 12 / .14); --term-bg:#0e1116; --term-fg:#d6dae2; --term-dim:#767e8d; --term-accent:#7fa0ff; --radius:10px; - --glass:rgb(255 255 255 / .46); - --glass-brd:rgb(255 255 255 / .85); + --glass:rgb(255 255 255 / .55); + /* A white edge on a near-white page is no edge at all. The border is a cool + hairline; the white inner highlight above it still reads as lit glass. */ + --glass-brd:rgb(20 32 74 / .12); --glass-hi:rgb(255 255 255 / .95); - --glass-sh:0 18px 50px -28px rgb(20 32 74 / .34); + --glass-sh:0 16px 40px -22px rgb(20 32 74 / .22); --amb-1:rgb(47 91 255 / .30); --amb-2:rgb(0 176 208 / .24); --amb-3:rgb(255 138 76 / .12); --sans:ui-sans-serif,system-ui,-apple-system,"SF Pro Text","Segoe UI",Roboto,Helvetica,Arial,sans-serif; --mono:ui-monospace,"SF Mono","JetBrains Mono","Menlo","Consolas",monospace; @@ -397,15 +399,28 @@ .js .rise{opacity:0; transform:translateY(16px); transition:opacity .6s cubic-bezier(.16,1,.3,1), transform .6s cubic-bezier(.16,1,.3,1)} .js .rise.in{opacity:1; transform:none} @media (prefers-reduced-transparency:reduce){ + /* Every glass surface needs a solid stand-in here. A translucent white panel + with the blur stripped is invisible on a light page, which leaves the + navigation, the buttons and the cards floating with no edges at all. */ html{background:var(--bg)} body{background:var(--bg)} - .nav, .cell, .chip, .term, .codeblock, .glass{ - backdrop-filter:none; -webkit-backdrop-filter:none; box-shadow:none; + .nav, .navshell, .brandpill, .ann, .tab, .btn-gh, + .cell, .cell.wide, .way, .chip, .it, .frow, .glass{ + backdrop-filter:none; -webkit-backdrop-filter:none; + background:var(--bg); border-color:var(--line-2); + box-shadow:0 1px 2px rgb(20 32 74 / .06); } - .nav, .cell, .chip{transition:transform var(--t) var(--ease), border-color var(--t) var(--ease);background:var(--bg); border-color:var(--line)} - .term, .codeblock{background:var(--term-bg); border-color:var(--line-2)} - .alt{background:var(--bg-2)} + .navshell, .brandpill{box-shadow:0 2px 10px -4px rgb(20 32 74 / .18)} + .alt{background:var(--bg-2); backdrop-filter:none; -webkit-backdrop-filter:none; + box-shadow:none; border-color:var(--line)} + .term, .codeblock{ + background:var(--term-bg); backdrop-filter:none; -webkit-backdrop-filter:none; + border-color:var(--line-2); box-shadow:none; + } + .tab.is-on{border-color:var(--accent); background:var(--accent-soft)} + .cell:hover, .way:hover{box-shadow:0 2px 8px rgb(20 32 74 / .12)} } + @media (prefers-reduced-motion:reduce){ .js .rise{opacity:1; transform:none; transition:none} .js .ann, .js .hero h1, .js .sub, .js .lede, .js .herocmd, .js .cta, .js .rotator, @@ -604,7 +619,7 @@

organizing‑bookmarks

files

organizing‑folders

-

Point it at Downloads, your Desktop, or any folder you like. It finds the same file saved twice, sets aside what you have not opened in months, and tells you about the huge things you forgot were in there. Anything you touched this week is left alone.

+

Point it at Downloads, your Desktop, or any folder you like. It finds the same file saved twice, sets aside what you have not opened in months, and tells you about the huge things you forgot were in there. In Dropbox or iCloud it also clears out the duplicate copies your devices left behind when they disagreed.

notes
diff --git a/skills/browser/organizing-bookmarks/references/app-data-locations.md b/skills/browser/organizing-bookmarks/references/app-data-locations.md index bb7189c..56e7357 100644 --- a/skills/browser/organizing-bookmarks/references/app-data-locations.md +++ b/skills/browser/organizing-bookmarks/references/app-data-locations.md @@ -94,6 +94,29 @@ Bookmarks live in `places.sqlite`, table `moz_bookmarks` joined to `moz_places`. | Downloads | `~/Downloads` | `%USERPROFILE%\Downloads` | | Documents | `~/Documents` | `%USERPROFILE%\Documents` | +### Cloud drives + +Sync folders are ordinary directories, so the same skill covers them. `scan_tree.py` +reports which provider a path belongs to. + +| Provider | Where it usually lives | +|---|---| +| iCloud Drive | `~/Library/Mobile Documents/`, and `~/iCloud Drive` | +| Dropbox | `~/Dropbox` | +| OneDrive | `~/OneDrive`, `%USERPROFILE%\OneDrive` | +| Google Drive | `~/Google Drive`, `~/Library/CloudStorage/GoogleDrive-*` | +| Box | `~/Box` | +| Nextcloud | `~/Nextcloud` | + +Two hazards, both handled by `scan_tree.py`: + +- **Placeholders.** Evicted files are left as stubs. Acting on one destroys its + content. Reported as `placeholders_not_downloaded`. +- **Conflicted copies.** Every client names them differently and none clean up after + themselves. Reported as `conflicts`, each paired with the original it competes with + and flagged `same_content` so identical copies can be separated from ones holding + work that exists nowhere else. + - **Permission:** Full Disk Access on macOS for all three. Nothing on Linux or Windows. - **Parser:** `scripts/scan_tree.py` - **Deletion:** never `rm`. Use `python3 scripts/platforms.py trash `, which diff --git a/skills/files/organizing-folders/SKILL.md b/skills/files/organizing-folders/SKILL.md index 01a5566..6bd1234 100644 --- a/skills/files/organizing-folders/SKILL.md +++ b/skills/files/organizing-folders/SKILL.md @@ -1,6 +1,6 @@ --- name: organizing-folders -description: Use when the user wants to clean up, sort, dedupe, or organize a folder — Downloads, Desktop, Documents, or any folder they name. Triggers on "my downloads folder is a mess", "clean up my desktop", "organize this folder", "sort my files", "find duplicate files", "my desktop is covered in files", "too many screenshots". +description: Use when the user wants to clean up, sort, dedupe, or organize a folder — Downloads, Desktop, Documents, a cloud drive, or any folder they name. Also handles conflicted copies left behind by Dropbox, OneDrive, iCloud, and Google Drive. Triggers on "my downloads folder is a mess", "clean up my desktop", "organize this folder", "sort my files", "find duplicate files", "my desktop is covered in files", "too many screenshots", "conflicted copy", "my dropbox is a mess", "google drive is full". --- # Organizing a Folder @@ -16,6 +16,7 @@ the user points at. - **No byte-identical duplicates.** `boarding-pass.pdf` and `boarding-pass(1).pdf` are one file. +- **No unresolved conflicted copies**, if this folder is synced. See below. - **Nothing loose that is older than 90 days.** It is archived, not deleted. - **What remains is grouped**, one level deep, by kind or by project. @@ -133,15 +134,40 @@ wrong move is felt immediately, so the bar for confirmation is higher. ### Any folder inside a cloud drive -iCloud, Dropbox, OneDrive, and Google Drive keep evicted files as zero-byte -placeholders. Acting on those **destroys content**. +`scan` reports `cloud_provider` when the root is inside iCloud Drive, Dropbox, +OneDrive, Google Drive, Box, or Nextcloud. Two things change when it does. -```bash -find -name "*.icloud" -o -size 0 -name "*.*" | head -``` +**Placeholders first, before anything else.** Sync clients evict files they think you +are not using and leave a stub behind. `scan` counts these as +`placeholders_not_downloaded`. **Acting on one destroys the content**, because you are +moving or trashing a pointer, not a file. + +If the count is above zero, tell the user which files, ask them to download the folder +fully, and stop. Do not proceed on the ones that happen to be present. + +**Then conflicted copies.** This is the clutter that only exists in synced folders, and +it is the reason to point this skill at one. Two devices edited the same file while +offline, so the client kept both and named the loser something like +`Budget (Sarah's conflicted copy 2025-11-03).xlsx`. Nobody ever goes back and resolves +them. + +`scan` reports each one under `conflicts` with the copy, the original it competes with, +and `same_content`. That flag decides what you may do: + +- **`same_content: true`** means the copy is byte-for-byte identical to the original. + The conflict was spurious. Safe to propose trashing, in bulk. +- **`same_content: false`** means the two genuinely differ, and **one of them holds work + that exists nowhere else.** Never resolve these unattended. Show the user the pair, + their sizes and dates, and let them choose. If they cannot tell, propose renaming the + copy to something obvious rather than removing it. +- **`original: null`** means the file the copy competed with is already gone. Leave it, + and suggest renaming it back to the plain name. + +Report the two groups separately. "9 conflicted copies, 6 identical and safe to remove, +3 that differ and need you" is useful. A single number is not. -If placeholders exist, tell the user to download the folder fully, and stop. Moving -files also triggers a re-sync, so say that before applying. +**Moving files triggers a re-sync.** Say so before applying, because the user may be on +a metered connection or short on space on another device. ## Judgment calls diff --git a/skills/files/organizing-folders/references/app-data-locations.md b/skills/files/organizing-folders/references/app-data-locations.md index bb7189c..56e7357 100644 --- a/skills/files/organizing-folders/references/app-data-locations.md +++ b/skills/files/organizing-folders/references/app-data-locations.md @@ -94,6 +94,29 @@ Bookmarks live in `places.sqlite`, table `moz_bookmarks` joined to `moz_places`. | Downloads | `~/Downloads` | `%USERPROFILE%\Downloads` | | Documents | `~/Documents` | `%USERPROFILE%\Documents` | +### Cloud drives + +Sync folders are ordinary directories, so the same skill covers them. `scan_tree.py` +reports which provider a path belongs to. + +| Provider | Where it usually lives | +|---|---| +| iCloud Drive | `~/Library/Mobile Documents/`, and `~/iCloud Drive` | +| Dropbox | `~/Dropbox` | +| OneDrive | `~/OneDrive`, `%USERPROFILE%\OneDrive` | +| Google Drive | `~/Google Drive`, `~/Library/CloudStorage/GoogleDrive-*` | +| Box | `~/Box` | +| Nextcloud | `~/Nextcloud` | + +Two hazards, both handled by `scan_tree.py`: + +- **Placeholders.** Evicted files are left as stubs. Acting on one destroys its + content. Reported as `placeholders_not_downloaded`. +- **Conflicted copies.** Every client names them differently and none clean up after + themselves. Reported as `conflicts`, each paired with the original it competes with + and flagged `same_content` so identical copies can be separated from ones holding + work that exists nowhere else. + - **Permission:** Full Disk Access on macOS for all three. Nothing on Linux or Windows. - **Parser:** `scripts/scan_tree.py` - **Deletion:** never `rm`. Use `python3 scripts/platforms.py trash `, which diff --git a/skills/files/organizing-folders/scripts/scan_tree.py b/skills/files/organizing-folders/scripts/scan_tree.py index a08772b..5187a95 100755 --- a/skills/files/organizing-folders/scripts/scan_tree.py +++ b/skills/files/organizing-folders/scripts/scan_tree.py @@ -11,7 +11,7 @@ Refuses to touch anything on the denylist. Never follows symlinks out of root. Stdlib only. """ -import argparse, hashlib, json, os, time +import argparse, hashlib, json, os, re, time from collections import defaultdict DENY_NAMES = {".ssh", ".gnupg", ".aws", ".config", ".kube", "Library"} @@ -30,6 +30,75 @@ EXT_TO_CAT = {e: c for c, exts in CATEGORIES.items() for e in exts} +# Every sync client marks a conflict differently, and none of them clean up after +# themselves. These are the patterns they actually write. +CONFLICT_PATTERNS = [ + (re.compile(r"\(([^()]*?)'s conflicted copy \d{4}-\d{2}-\d{2}\)", re.I), "Dropbox"), + (re.compile(r"\bconflicted copy\b", re.I), "Dropbox"), + (re.compile(r"-[A-Za-z0-9]+'s conflicted copy", re.I), "Dropbox"), + (re.compile(r"\((?:[^()]*\s)?conflicted copy(?:\s[^()]*)?\)", re.I), "Dropbox"), + (re.compile(r"-\s?[A-Za-z0-9._-]+'s\s+conflict", re.I), "Dropbox"), + (re.compile(r"\(\d+\)_conflict", re.I), "Nextcloud"), + (re.compile(r"_conflict-\d{8}-\d{6}", re.I), "Nextcloud"), + (re.compile(r"\bconflicted version\b", re.I), "Box"), + (re.compile(r"-\s?[A-Za-z0-9._-]+\s?\(\d+\)(?=\.\w+$)"), None), + (re.compile(r"\(Case Conflict\)", re.I), "Dropbox"), + (re.compile(r"\bconflicted\b", re.I), None), + # OneDrive appends the machine name: "Budget-DESKTOP-A1B2C3.xlsx" + (re.compile(r"-(?:DESKTOP|LAPTOP|MacBook|MBP|PC)-[A-Z0-9]{4,}(?=\.[^.]*$)", re.I), "OneDrive"), + # Google Drive: "Budget (1).xlsx" alongside "Budget.xlsx" is handled by pairing, + # not by name alone, because (1) is also just a second download. +] + + +def conflict_of(name): + """Return the sync client that produced this name, or None. + + Named conflicts only. A trailing "(1)" is ambiguous on its own and is left to + duplicate detection, which compares content rather than guessing from a name. + """ + for pattern, client in CONFLICT_PATTERNS: + if pattern.search(name): + return client or "sync client" + return None + + +def base_name_of(name): + """Strip a conflict marker to recover the name the file is competing with. + + Patterns are applied to the whole filename, so any that anchor on the + extension still match, and the leftover separator before the dot is cleaned + up afterwards. + """ + stripped = name + for pattern, _ in CONFLICT_PATTERNS: + stripped = pattern.sub("", stripped) + stem, ext = os.path.splitext(stripped) + stem = re.sub(r"[\s._-]+$", "", stem) + return stem + ext + + +CLOUD_ROOTS = { + "iCloud Drive": ("Mobile Documents", "iCloud"), + "Dropbox": ("Dropbox",), + "OneDrive": ("OneDrive",), + "Google Drive": ("Google Drive", "GoogleDrive", "CloudStorage"), + "Box": ("Box",), + "Nextcloud": ("Nextcloud",), + "Sync": ("Sync",), +} + + +def cloud_provider(root): + """Which sync service, if any, this path lives inside.""" + real = os.path.realpath(os.path.expanduser(root)) + for name, markers in CLOUD_ROOTS.items(): + for marker in markers: + if os.sep + marker in real + os.sep or real.endswith(os.sep + marker): + return name + return None + + def denied(path, root): """True if path escapes root or touches a denylisted location.""" real_root = os.path.realpath(root) @@ -82,7 +151,9 @@ def scan(root, max_depth, min_size_mb): skipped += 1 continue ext = os.path.splitext(name)[1].lower() + conflict = conflict_of(name) files.append({ + "conflict": conflict, "path": os.path.relpath(full, root), "name": name, "ext": ext, @@ -90,6 +161,10 @@ def scan(root, max_depth, min_size_mb): "size_bytes": st.st_size, "size_mb": round(st.st_size / 1e6, 2), "age_days": int((now - st.st_mtime) / 86400), + # A zero-byte file that is not meant to be empty, or an explicit + # placeholder extension, means the content is not on this machine. + "placeholder": ext in (".icloud", ".cloud") + or (st.st_size == 0 and ext not in ("", ".gitkeep", ".keep")), }) by_cat, by_age = defaultdict(lambda: {"count": 0, "size_mb": 0.0}), defaultdict(int) @@ -114,8 +189,34 @@ def scan(root, max_depth, min_size_mb): pass dupes = {k: v for k, v in groups.items() if len(v) > 1} + # Pair each conflicted copy with the original it is competing against. + by_name = {} + for f in files: + by_name.setdefault(os.path.join(os.path.dirname(f["path"]), f["name"]), f) + conflicts = [] + for f in files: + if not f["conflict"]: + continue + original = os.path.join(os.path.dirname(f["path"]), base_name_of(f["name"])) + match = by_name.get(original) + conflicts.append({ + "copy": f["path"], + "client": f["conflict"], + "size_mb": f["size_mb"], + "age_days": f["age_days"], + "original": match["path"] if match else None, + "same_content": bool(match) and f["size_bytes"] == match["size_bytes"], + }) + + placeholders = [f["path"] for f in files if f["placeholder"]] + return { "root": root, + "cloud_provider": cloud_provider(root), + "conflicted_copies": len(conflicts), + "conflicts": conflicts[:30], + "placeholders_not_downloaded": len(placeholders), + "placeholder_examples": placeholders[:10], "total_files": len(files), "total_size_mb": round(sum(f["size_mb"] for f in files), 2), "skipped_denied_or_symlink": skipped, diff --git a/skills/notes/organizing-notes/references/app-data-locations.md b/skills/notes/organizing-notes/references/app-data-locations.md index bb7189c..56e7357 100644 --- a/skills/notes/organizing-notes/references/app-data-locations.md +++ b/skills/notes/organizing-notes/references/app-data-locations.md @@ -94,6 +94,29 @@ Bookmarks live in `places.sqlite`, table `moz_bookmarks` joined to `moz_places`. | Downloads | `~/Downloads` | `%USERPROFILE%\Downloads` | | Documents | `~/Documents` | `%USERPROFILE%\Documents` | +### Cloud drives + +Sync folders are ordinary directories, so the same skill covers them. `scan_tree.py` +reports which provider a path belongs to. + +| Provider | Where it usually lives | +|---|---| +| iCloud Drive | `~/Library/Mobile Documents/`, and `~/iCloud Drive` | +| Dropbox | `~/Dropbox` | +| OneDrive | `~/OneDrive`, `%USERPROFILE%\OneDrive` | +| Google Drive | `~/Google Drive`, `~/Library/CloudStorage/GoogleDrive-*` | +| Box | `~/Box` | +| Nextcloud | `~/Nextcloud` | + +Two hazards, both handled by `scan_tree.py`: + +- **Placeholders.** Evicted files are left as stubs. Acting on one destroys its + content. Reported as `placeholders_not_downloaded`. +- **Conflicted copies.** Every client names them differently and none clean up after + themselves. Reported as `conflicts`, each paired with the original it competes with + and flagged `same_content` so identical copies can be separated from ones holding + work that exists nowhere else. + - **Permission:** Full Disk Access on macOS for all three. Nothing on Linux or Windows. - **Parser:** `scripts/scan_tree.py` - **Deletion:** never `rm`. Use `python3 scripts/platforms.py trash `, which diff --git a/tests/test_scan_tree.py b/tests/test_scan_tree.py index ac04453..88c3274 100644 --- a/tests/test_scan_tree.py +++ b/tests/test_scan_tree.py @@ -156,3 +156,129 @@ def test_violation_is_reported_not_just_flagged(self): if __name__ == "__main__": unittest.main() + + +class TestConflictedCopies(unittest.TestCase): + """Conflicted copies are the one clutter type unique to synced folders. + + Every sync client marks them differently and none clean up after themselves, + so the names are the only signal available. + """ + + def _name(self, n): + return scan_tree.conflict_of(n) + + def test_dropbox_dated_conflict(self): + self.assertEqual(self._name("Budget (Sarah's conflicted copy 2025-11-03).xlsx"), + "Dropbox") + + def test_dropbox_case_conflict(self): + self.assertIsNotNone(self._name("Photo (Case Conflict).jpg")) + + def test_onedrive_machine_suffix(self): + self.assertEqual(self._name("Notes-DESKTOP-A1B2C3.docx"), "OneDrive") + + def test_nextcloud_conflict(self): + self.assertIsNotNone(self._name("Report_conflict-20251103-140233.pdf")) + + def test_ordinary_names_are_not_conflicts(self): + for name in ("Budget.xlsx", "holiday photo.jpg", "notes-final.docx", + "report-v2.pdf", "Screenshot 2026-08-14.png"): + with self.subTest(name=name): + self.assertIsNone(self._name(name)) + + def test_plain_numbered_copy_is_not_called_a_conflict(self): + # "(1)" is ambiguous: it is also just a second download. Content + # comparison decides that one, not the filename. + self.assertIsNone(self._name("invoice (1).pdf")) + + def test_base_name_recovers_the_original(self): + cases = { + "Budget (Sarah's conflicted copy 2025-11-03).xlsx": "Budget.xlsx", + "Notes-DESKTOP-A1B2C3.docx": "Notes.docx", + } + for conflicted, original in cases.items(): + with self.subTest(name=conflicted): + self.assertEqual(scan_tree.base_name_of(conflicted), original) + + def test_extension_survives_stripping(self): + for name in ("Budget (Sarah's conflicted copy 2025-11-03).xlsx", + "Notes-DESKTOP-A1B2C3.docx"): + with self.subTest(name=name): + self.assertTrue(scan_tree.base_name_of(name).count(".") >= 1) + + +class TestCloudScan(unittest.TestCase): + def setUp(self): + self.root = tempfile.mkdtemp() + self.dbx = os.path.join(self.root, "Dropbox", "Family") + os.makedirs(self.dbx) + self._write("Budget.xlsx", "newer version") + self._write("Budget (Sarah's conflicted copy 2025-11-03).xlsx", "older") + self._write("Notes.docx", "same bytes") + self._write("Notes-DESKTOP-A1B2C3.docx", "same bytes") + self._write("Recipes.pdf", "unrelated") + + def _write(self, name, content): + with open(os.path.join(self.dbx, name), "w") as fh: + fh.write(content) + + def scan(self): + return scan_tree.scan(self.dbx, None, 0) + + def test_detects_the_sync_provider_from_the_path(self): + self.assertEqual(self.scan()["cloud_provider"], "Dropbox") + + def test_plain_folder_reports_no_provider(self): + plain = tempfile.mkdtemp() + self.assertIsNone(scan_tree.scan(plain, None, 0)["cloud_provider"]) + + def test_counts_only_real_conflicts(self): + self.assertEqual(self.scan()["conflicted_copies"], 2) + + def test_pairs_each_copy_with_its_original(self): + by_copy = {c["copy"]: c for c in self.scan()["conflicts"]} + self.assertEqual( + by_copy["Budget (Sarah's conflicted copy 2025-11-03).xlsx"]["original"], + "Budget.xlsx") + self.assertEqual(by_copy["Notes-DESKTOP-A1B2C3.docx"]["original"], "Notes.docx") + + def test_flags_which_conflicts_are_identical(self): + # Identical to the original is safe to remove. Differing needs a human. + by_copy = {c["copy"]: c for c in self.scan()["conflicts"]} + self.assertTrue(by_copy["Notes-DESKTOP-A1B2C3.docx"]["same_content"]) + self.assertFalse( + by_copy["Budget (Sarah's conflicted copy 2025-11-03).xlsx"]["same_content"]) + + def test_unpaired_conflict_reports_no_original(self): + self._write("Orphan (conflicted copy).txt", "x") + match = [c for c in self.scan()["conflicts"] if c["copy"].startswith("Orphan")] + self.assertEqual(match[0]["original"], None) + + +class TestPlaceholders(unittest.TestCase): + """Acting on a file the sync client has evicted destroys its content.""" + + def setUp(self): + self.root = tempfile.mkdtemp() + + def _touch(self, name, content=""): + with open(os.path.join(self.root, name), "w") as fh: + fh.write(content) + + def test_icloud_placeholder_detected(self): + self._touch("Holiday.mov.icloud") + self.assertEqual(scan_tree.scan(self.root, None, 0)["placeholders_not_downloaded"], 1) + + def test_zero_byte_file_treated_as_not_downloaded(self): + self._touch("Photo.jpg") + self.assertEqual(scan_tree.scan(self.root, None, 0)["placeholders_not_downloaded"], 1) + + def test_real_files_are_not_placeholders(self): + self._touch("Photo.jpg", "actual bytes") + self.assertEqual(scan_tree.scan(self.root, None, 0)["placeholders_not_downloaded"], 0) + + def test_keep_files_are_not_placeholders(self): + self._touch(".gitkeep") + self._touch("real.txt", "x") + self.assertEqual(scan_tree.scan(self.root, None, 0)["placeholders_not_downloaded"], 0)