0.6.10: Auto-delete identical numbered-sibling duplicates, no review needed

Every dedup pass previously funneled through the same manual-confirm
queue, including the obviously-safe case: two files in the same
directory with the same name and extension, differing only by the
".N" collision suffix sldl/beets inserts when a filename collides.
That's not a fuzzy match or a different version -- it's the same
download landing twice -- so it doesn't need a human in the loop the
way case-insensitive tag matches or cross-album fuzzy matches do
(different masters, DJ-mix versions, etc., which still require
confirmation).

dedup-library.sh gains --auto-apply-numbered: an is_numbered_twin()
filename check (same dir, same name modulo the numeric infix, same
extension) that deletes matching pairs unconditionally, regardless of
which pass found them -- Pass 1 only fires when the canonical name is
a beets ghost; the common case where both copies are already tracked
in beets defers to Pass 2's tag-based grouping instead, and still gets
the same free pass here. When the auto-deleted pair's survivor is
itself the numbered-named file, it's renamed back to canonical so the
library doesn't accumulate ".1."/".2." names for tracks that no longer
have a duplicate.

dedup_review_service.scan() (both the manual "Scan now" button and the
nightly scheduled job) now passes this flag and records auto-applied
deletions as pre-confirmed DedupCandidate rows for audit visibility --
they never appear as pending review items. Tag-based and cross-album
fuzzy passes are unaffected and still require manual confirmation.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
andrew
2026-07-22 09:34:16 -06:00
parent 62251d764e
commit f05f076464
3 changed files with 113 additions and 18 deletions
+46 -10
View File
@@ -58,7 +58,9 @@ def _is_pair_already_pending(db, keep_path: str, delete_path: str) -> bool:
return db.execute(query).first() is not None
async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupRun | None:
async def _run_scan(
job_key: str, script_name: str, triggered_by: str, extra_args: tuple[str, ...] = ()
) -> DedupRun | None:
"""Returns None (persisting nothing) if the job never actually ran --
e.g. skipped_lock because something else was using the pipeline lock
at that moment. Same lesson as genre_review_service.run(): recording a
@@ -67,7 +69,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
pending-candidates list isn't scoped to "latest run only"."""
script = str(settings.pipeline_dir / "lib" / script_name)
job_run, output = await pipeline_runner.run_job_capture(
job_key, [script, "--json"], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
job_key, [script, "--json", *extra_args], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
)
if job_run.status != "success":
return None
@@ -76,6 +78,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
db = SessionLocal()
try:
now = time.time()
dedup_run = DedupRun(
started_at=job_run.started_at,
finished_at=job_run.finished_at,
@@ -91,7 +94,30 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
skipped_ignored = 0
skipped_duplicate = 0
auto_applied = 0
for c in candidates:
if c.get("auto_applied"):
# dedup-library.sh already deleted this pair unconditionally
# (identical numbered-sibling twin: same dir/name/ext, no
# tag/fuzzy ambiguity involved) -- record it pre-applied for
# audit history; it never shows up as a pending review item.
db.add(
DedupCandidate(
dedup_run_id=dedup_run.id,
pass_name=c.get("pass", "unknown"),
keep_path=c["keep_path"],
delete_path=c["delete_path"],
delete_id=c.get("delete_id"),
delete_size_bytes=c.get("delete_size_bytes"),
confirmed=True,
confirmed_by="auto:numbered_twin",
confirmed_at=now,
applied=True,
)
)
auto_applied += 1
db.flush()
continue
if _is_pair_ignored(db, c["keep_path"], c["delete_path"]):
skipped_ignored += 1
continue
@@ -116,8 +142,11 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
db.commit()
db.refresh(dedup_run)
skipped_total = skipped_ignored + skipped_duplicate
if skipped_total:
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
if skipped_total or auto_applied:
if skipped_total:
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
if auto_applied:
dedup_run.deleted = auto_applied
db.commit()
db.refresh(dedup_run)
return dedup_run
@@ -126,12 +155,19 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
async def scan(triggered_by: str = "manual") -> DedupRun | None:
"""Dry-run dedup-library.sh --json, persist every candidate deletion
into a fresh dedup_runs/dedup_candidates pair. Never deletes anything
-- the scheduled maintenance:dedup job also only ever calls this (no
--apply), matching the false-negative-biased dedup preference; actual
deletion only ever happens through confirm_and_apply() below."""
return await _run_scan("dedup:scan", _SCRIPT, triggered_by)
"""Dry-run dedup-library.sh --json (plus --auto-apply-numbered), persist
every candidate deletion into a fresh dedup_runs/dedup_candidates pair.
Everything except identical numbered-sibling twins (same dir, same name,
same extension, differing only by the ".N" collision suffix) stays a
dry-run candidate awaiting manual confirm_and_apply() below -- that's the
false-negative-biased preference for tag-based/fuzzy matches, where a
wrong auto-delete could take out a genuinely different version (a
different album pressing, a DJ-mix edit, etc.). Numbered twins have none
of that ambiguity -- it's the same download landing twice -- so
dedup-library.sh deletes those unconditionally itself and reports them
here already applied, purely for audit visibility."""
return await _run_scan("dedup:scan", _SCRIPT, triggered_by, extra_args=("--auto-apply-numbered",))
async def scan_fuzzy(triggered_by: str = "manual") -> DedupRun | None: