0.6.10: Auto-delete identical numbered-sibling duplicates, no review needed
Every dedup pass previously funneled through the same manual-confirm queue, including the obviously-safe case: two files in the same directory with the same name and extension, differing only by the ".N" collision suffix sldl/beets inserts when a filename collides. That's not a fuzzy match or a different version -- it's the same download landing twice -- so it doesn't need a human in the loop the way case-insensitive tag matches or cross-album fuzzy matches do (different masters, DJ-mix versions, etc., which still require confirmation). dedup-library.sh gains --auto-apply-numbered: an is_numbered_twin() filename check (same dir, same name modulo the numeric infix, same extension) that deletes matching pairs unconditionally, regardless of which pass found them -- Pass 1 only fires when the canonical name is a beets ghost; the common case where both copies are already tracked in beets defers to Pass 2's tag-based grouping instead, and still gets the same free pass here. When the auto-deleted pair's survivor is itself the numbered-named file, it's renamed back to canonical so the library doesn't accumulate ".1."/".2." names for tracks that no longer have a duplicate. dedup_review_service.scan() (both the manual "Scan now" button and the nightly scheduled job) now passes this flag and records auto-applied deletions as pre-confirmed DedupCandidate rows for audit visibility -- they never appear as pending review items. Tag-based and cross-album fuzzy passes are unaffected and still require manual confirmation. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -58,7 +58,9 @@ def _is_pair_already_pending(db, keep_path: str, delete_path: str) -> bool:
|
||||
return db.execute(query).first() is not None
|
||||
|
||||
|
||||
async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupRun | None:
|
||||
async def _run_scan(
|
||||
job_key: str, script_name: str, triggered_by: str, extra_args: tuple[str, ...] = ()
|
||||
) -> DedupRun | None:
|
||||
"""Returns None (persisting nothing) if the job never actually ran --
|
||||
e.g. skipped_lock because something else was using the pipeline lock
|
||||
at that moment. Same lesson as genre_review_service.run(): recording a
|
||||
@@ -67,7 +69,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
pending-candidates list isn't scoped to "latest run only"."""
|
||||
script = str(settings.pipeline_dir / "lib" / script_name)
|
||||
job_run, output = await pipeline_runner.run_job_capture(
|
||||
job_key, [script, "--json"], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
|
||||
job_key, [script, "--json", *extra_args], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
|
||||
)
|
||||
if job_run.status != "success":
|
||||
return None
|
||||
@@ -76,6 +78,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
now = time.time()
|
||||
dedup_run = DedupRun(
|
||||
started_at=job_run.started_at,
|
||||
finished_at=job_run.finished_at,
|
||||
@@ -91,7 +94,30 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
|
||||
skipped_ignored = 0
|
||||
skipped_duplicate = 0
|
||||
auto_applied = 0
|
||||
for c in candidates:
|
||||
if c.get("auto_applied"):
|
||||
# dedup-library.sh already deleted this pair unconditionally
|
||||
# (identical numbered-sibling twin: same dir/name/ext, no
|
||||
# tag/fuzzy ambiguity involved) -- record it pre-applied for
|
||||
# audit history; it never shows up as a pending review item.
|
||||
db.add(
|
||||
DedupCandidate(
|
||||
dedup_run_id=dedup_run.id,
|
||||
pass_name=c.get("pass", "unknown"),
|
||||
keep_path=c["keep_path"],
|
||||
delete_path=c["delete_path"],
|
||||
delete_id=c.get("delete_id"),
|
||||
delete_size_bytes=c.get("delete_size_bytes"),
|
||||
confirmed=True,
|
||||
confirmed_by="auto:numbered_twin",
|
||||
confirmed_at=now,
|
||||
applied=True,
|
||||
)
|
||||
)
|
||||
auto_applied += 1
|
||||
db.flush()
|
||||
continue
|
||||
if _is_pair_ignored(db, c["keep_path"], c["delete_path"]):
|
||||
skipped_ignored += 1
|
||||
continue
|
||||
@@ -116,8 +142,11 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
db.commit()
|
||||
db.refresh(dedup_run)
|
||||
skipped_total = skipped_ignored + skipped_duplicate
|
||||
if skipped_total:
|
||||
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
|
||||
if skipped_total or auto_applied:
|
||||
if skipped_total:
|
||||
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
|
||||
if auto_applied:
|
||||
dedup_run.deleted = auto_applied
|
||||
db.commit()
|
||||
db.refresh(dedup_run)
|
||||
return dedup_run
|
||||
@@ -126,12 +155,19 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
|
||||
|
||||
async def scan(triggered_by: str = "manual") -> DedupRun | None:
|
||||
"""Dry-run dedup-library.sh --json, persist every candidate deletion
|
||||
into a fresh dedup_runs/dedup_candidates pair. Never deletes anything
|
||||
-- the scheduled maintenance:dedup job also only ever calls this (no
|
||||
--apply), matching the false-negative-biased dedup preference; actual
|
||||
deletion only ever happens through confirm_and_apply() below."""
|
||||
return await _run_scan("dedup:scan", _SCRIPT, triggered_by)
|
||||
"""Dry-run dedup-library.sh --json (plus --auto-apply-numbered), persist
|
||||
every candidate deletion into a fresh dedup_runs/dedup_candidates pair.
|
||||
|
||||
Everything except identical numbered-sibling twins (same dir, same name,
|
||||
same extension, differing only by the ".N" collision suffix) stays a
|
||||
dry-run candidate awaiting manual confirm_and_apply() below -- that's the
|
||||
false-negative-biased preference for tag-based/fuzzy matches, where a
|
||||
wrong auto-delete could take out a genuinely different version (a
|
||||
different album pressing, a DJ-mix edit, etc.). Numbered twins have none
|
||||
of that ambiguity -- it's the same download landing twice -- so
|
||||
dedup-library.sh deletes those unconditionally itself and reports them
|
||||
here already applied, purely for audit visibility."""
|
||||
return await _run_scan("dedup:scan", _SCRIPT, triggered_by, extra_args=("--auto-apply-numbered",))
|
||||
|
||||
|
||||
async def scan_fuzzy(triggered_by: str = "manual") -> DedupRun | None:
|
||||
|
||||
Reference in New Issue
Block a user