1 Commits

Author SHA1 Message Date
andrew f05f076464 0.6.10: Auto-delete identical numbered-sibling duplicates, no review needed
Every dedup pass previously funneled through the same manual-confirm
queue, including the obviously-safe case: two files in the same
directory with the same name and extension, differing only by the
".N" collision suffix sldl/beets inserts when a filename collides.
That's not a fuzzy match or a different version -- it's the same
download landing twice -- so it doesn't need a human in the loop the
way case-insensitive tag matches or cross-album fuzzy matches do
(different masters, DJ-mix versions, etc., which still require
confirmation).

dedup-library.sh gains --auto-apply-numbered: an is_numbered_twin()
filename check (same dir, same name modulo the numeric infix, same
extension) that deletes matching pairs unconditionally, regardless of
which pass found them -- Pass 1 only fires when the canonical name is
a beets ghost; the common case where both copies are already tracked
in beets defers to Pass 2's tag-based grouping instead, and still gets
the same free pass here. When the auto-deleted pair's survivor is
itself the numbered-named file, it's renamed back to canonical so the
library doesn't accumulate ".1."/".2." names for tracks that no longer
have a duplicate.

dedup_review_service.scan() (both the manual "Scan now" button and the
nightly scheduled job) now passes this flag and records auto-applied
deletions as pre-confirmed DedupCandidate rows for audit visibility --
they never appear as pending review items. Tag-based and cross-album
fuzzy passes are unaffected and still require manual confirmation.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-22 09:34:16 -06:00
3 changed files with 113 additions and 18 deletions
+3 -3
View File
@@ -73,10 +73,10 @@ Optional, add these later if you want them:
Pull the prebuilt image onto your Docker host:
```bash
docker pull git.kretzer.club/andrew/alembic:0.6.9
docker pull git.kretzer.club/andrew/alembic:0.6.10
```
That is the whole install. You do not need to download the source or build anything. The `0.6.9` is the version; you can pin to it so nothing changes under you, or use `latest` to always get the newest.
That is the whole install. You do not need to download the source or build anything. The `0.6.10` is the version; you can pin to it so nothing changes under you, or use `latest` to always get the newest.
(If you would rather build it yourself from source, you can, but you do not need to.)
@@ -136,7 +136,7 @@ Create a file called `docker-compose.yml` on your server (put it wherever you ke
```yaml
services:
alembic:
image: git.kretzer.club/andrew/alembic:0.6.9
image: git.kretzer.club/andrew/alembic:0.6.10
container_name: alembic
ports:
- "8420:8420"
+44 -8
View File
@@ -58,7 +58,9 @@ def _is_pair_already_pending(db, keep_path: str, delete_path: str) -> bool:
return db.execute(query).first() is not None
async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupRun | None:
async def _run_scan(
job_key: str, script_name: str, triggered_by: str, extra_args: tuple[str, ...] = ()
) -> DedupRun | None:
"""Returns None (persisting nothing) if the job never actually ran --
e.g. skipped_lock because something else was using the pipeline lock
at that moment. Same lesson as genre_review_service.run(): recording a
@@ -67,7 +69,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
pending-candidates list isn't scoped to "latest run only"."""
script = str(settings.pipeline_dir / "lib" / script_name)
job_run, output = await pipeline_runner.run_job_capture(
job_key, [script, "--json"], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
job_key, [script, "--json", *extra_args], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
)
if job_run.status != "success":
return None
@@ -76,6 +78,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
db = SessionLocal()
try:
now = time.time()
dedup_run = DedupRun(
started_at=job_run.started_at,
finished_at=job_run.finished_at,
@@ -91,7 +94,30 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
skipped_ignored = 0
skipped_duplicate = 0
auto_applied = 0
for c in candidates:
if c.get("auto_applied"):
# dedup-library.sh already deleted this pair unconditionally
# (identical numbered-sibling twin: same dir/name/ext, no
# tag/fuzzy ambiguity involved) -- record it pre-applied for
# audit history; it never shows up as a pending review item.
db.add(
DedupCandidate(
dedup_run_id=dedup_run.id,
pass_name=c.get("pass", "unknown"),
keep_path=c["keep_path"],
delete_path=c["delete_path"],
delete_id=c.get("delete_id"),
delete_size_bytes=c.get("delete_size_bytes"),
confirmed=True,
confirmed_by="auto:numbered_twin",
confirmed_at=now,
applied=True,
)
)
auto_applied += 1
db.flush()
continue
if _is_pair_ignored(db, c["keep_path"], c["delete_path"]):
skipped_ignored += 1
continue
@@ -116,8 +142,11 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
db.commit()
db.refresh(dedup_run)
skipped_total = skipped_ignored + skipped_duplicate
if skipped_total or auto_applied:
if skipped_total:
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
if auto_applied:
dedup_run.deleted = auto_applied
db.commit()
db.refresh(dedup_run)
return dedup_run
@@ -126,12 +155,19 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
async def scan(triggered_by: str = "manual") -> DedupRun | None:
"""Dry-run dedup-library.sh --json, persist every candidate deletion
into a fresh dedup_runs/dedup_candidates pair. Never deletes anything
-- the scheduled maintenance:dedup job also only ever calls this (no
--apply), matching the false-negative-biased dedup preference; actual
deletion only ever happens through confirm_and_apply() below."""
return await _run_scan("dedup:scan", _SCRIPT, triggered_by)
"""Dry-run dedup-library.sh --json (plus --auto-apply-numbered), persist
every candidate deletion into a fresh dedup_runs/dedup_candidates pair.
Everything except identical numbered-sibling twins (same dir, same name,
same extension, differing only by the ".N" collision suffix) stays a
dry-run candidate awaiting manual confirm_and_apply() below -- that's the
false-negative-biased preference for tag-based/fuzzy matches, where a
wrong auto-delete could take out a genuinely different version (a
different album pressing, a DJ-mix edit, etc.). Numbered twins have none
of that ambiguity -- it's the same download landing twice -- so
dedup-library.sh deletes those unconditionally itself and reports them
here already applied, purely for audit visibility."""
return await _run_scan("dedup:scan", _SCRIPT, triggered_by, extra_args=("--auto-apply-numbered",))
async def scan_fuzzy(triggered_by: str = "manual") -> DedupRun | None:
+64 -5
View File
@@ -34,11 +34,19 @@ set -euo pipefail
APPLY=0
JSON_MODE=0
ONLY_PATHS_FILE=""
AUTO_APPLY_NUMBERED=0
while [[ $# -gt 0 ]]; do
case "$1" in
--apply) APPLY=1; shift ;;
--json) JSON_MODE=1; shift ;;
--only-paths) ONLY_PATHS_FILE="$2"; shift 2 ;;
# Delete Pass 1 (numbered-sibling) matches unconditionally, independent of
# --apply/--only-paths, while Passes 2-4 stay dry-run/log-only. Safe to
# run unattended: a numbered-sibling pair is the SAME title, in the SAME
# directory, with the SAME extension -- unlike the tag-based/fuzzy passes,
# there's no fuzzy matching or cross-album ambiguity to get wrong, so it
# doesn't need a human in the loop. See dedup_review_service.scan().
--auto-apply-numbered) AUTO_APPLY_NUMBERED=1; shift ;;
*) shift ;;
esac
done
@@ -97,6 +105,25 @@ host_path() {
SEP=$'\x1c' # ASCII file separator — safe with any music metadata
# True if two paths are the SAME directory, SAME extension, and identical
# basenames except for a numeric ".N" infix sldl/beets inserts on a filename
# collision (e.g. "Title.flac" vs "Title.1.flac"). This is deliberately
# narrower than any tag-based pass: no fuzzy matching, no cross-album
# ambiguity, just the literal same download landing twice. Used to
# auto-delete regardless of which pass found the pair -- Pass 1 only fires
# when the canonical name is a beets ghost; the very common case where BOTH
# copies are already tracked in beets (so Pass 1 defers) surfaces instead via
# Pass 2/3's tag-based grouping, and still deserves the same free pass.
is_numbered_twin() {
local a="$1" b="$2"
[[ "${a%/*}" == "${b%/*}" ]] || return 1
local ba="${a##*/}" bb="${b##*/}"
local ext="${ba##*.}"
[[ "${ext,,}" == "${bb##*.}" ]] || return 1
[[ "$(echo "$ba" | sed -E "s/\.[0-9]+\.${ext}\$/.${ext}/")" \
== "$(echo "$bb" | sed -E "s/\.[0-9]+\.${ext}\$/.${ext}/")" ]]
}
# Rank a file: lower = better. FLAC > MP3 > other (WAV/etc.); clean filename
# > sync-conflict artifact (-DESKTOP-, (1), .1.); extended/club mix > radio
# edit at same format; then larger wins.
@@ -177,7 +204,7 @@ process_group() {
esac
local first=1
local keep_path=""
local keep_path="" keep_id="" any_auto=0
while IFS="$SEP" read -r _sk id hp; do
[[ -z "$hp" ]] && continue
local size_bytes; size_bytes=$(stat -c '%s' "$hp" 2>/dev/null || echo 0)
@@ -185,14 +212,30 @@ process_group() {
if [[ $first -eq 1 ]]; then
log " KEEP ($size_h) $hp"
keep_path="$hp"
keep_id="$id"
first=0
else
log " DELETE ($size_h) $hp"
json_emit "{\"pass\":\"${pass_name}\",\"keep_path\":\"$(json_escape "$keep_path")\",\"delete_path\":\"$(json_escape "$hp")\",\"delete_size_bytes\":${size_bytes}}"
# Auto-apply-numbered deletes numbered_sibling matches unconditionally,
# bypassing --only-paths (that gate exists only for the manual confirm
# flow, which never runs alongside this flag). All other passes stay
# dry-run/log-only here regardless.
local do_delete=0 auto_applied=0
if [[ $APPLY -eq 1 ]]; then
do_delete=1
elif [[ $AUTO_APPLY_NUMBERED -eq 1 ]] && is_numbered_twin "$keep_path" "$hp"; then
do_delete=1
auto_applied=1
any_auto=1
fi
local auto_json="false"; [[ $auto_applied -eq 1 ]] && auto_json="true"
json_emit "{\"pass\":\"${pass_name}\",\"keep_path\":\"$(json_escape "$keep_path")\",\"delete_path\":\"$(json_escape "$hp")\",\"delete_size_bytes\":${size_bytes},\"auto_applied\":${auto_json}}"
if [[ $do_delete -eq 1 ]]; then
# With --only-paths given, only delete entries the caller explicitly
# confirmed; without it, delete everything (original behavior).
if [[ -n "$ONLY_PATHS_FILE" && -z "${ONLY_PATHS[$hp]:-}" ]]; then
if [[ $auto_applied -eq 0 && -n "$ONLY_PATHS_FILE" && -z "${ONLY_PATHS[$hp]:-}" ]]; then
log " (skipped — not in --only-paths confirm list)"
elif [[ -n "$id" ]]; then
# In beets — beet remove -d removes from DB + disk. Best-effort: a
@@ -206,6 +249,18 @@ process_group() {
fi
fi
done < <(echo "$ranked" | grep -v '^$' | sort -n)
# An auto-applied numbered-twin deletion can leave the numbered-named file
# as the sole survivor (it won on size despite the dirty-filename penalty).
# Rename it back to the canonical name so the library doesn't accumulate
# ".1."/".2." filenames for tracks that no longer have a duplicate. Only
# auto_applied triggers this (never a manual --only-paths confirm, which
# may leave the sibling undeleted if the user didn't confirm it -- renaming
# then would collide); only when it's actually in beets (a ghost survivor
# has nothing to move); only when the name still has the dirty infix.
if [[ $any_auto -eq 1 && -n "$keep_id" && "$keep_path" =~ \.[0-9]+\.[A-Za-z0-9]+$ ]]; then
beet move "id:${keep_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${keep_id} failed (rename survivor)"
fi
return 0 # explicit: the loop's last command status must not leak out under set -e
}
@@ -248,16 +303,20 @@ while IFS="$SEP" read -r numbered_id numbered_cp; do
# Canonical wins: delete numbered (in beets), keep canonical (ghost)
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
"$host_canonical" "${numbered_id}${SEP}${numbered_cp}"
if [[ $APPLY -eq 1 ]]; then
if [[ $APPLY -eq 1 || $AUTO_APPLY_NUMBERED -eq 1 ]]; then
# canonical is now a ghost file; import it into beets
beet import -q -s "${canonical_cp}" >> "$LOG" 2>&1 || log " WARN: beet import failed for ${canonical_cp}"
fi
else
# Numbered wins (canonical is ghost): rm the ghost, then beet move to rename
# the numbered file to the canonical path (id query — see process_group).
# Safe to reach with just AUTO_APPLY_NUMBERED: this loop only ever builds
# genuine numbered-sibling pairs (see is_numbered_twin), and process_group
# above already deleted the ghost canonical before this runs, so the move
# target is guaranteed clear -- no rename collision.
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
"${numbered_id}${SEP}${numbered_cp}" "$host_canonical"
if [[ $APPLY -eq 1 ]]; then
if [[ $APPLY -eq 1 || $AUTO_APPLY_NUMBERED -eq 1 ]]; then
beet move "id:${numbered_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${numbered_id} failed"
fi
fi