2075d6cf66
dedup-library.sh: additive --json (one NDJSON line per candidate deletion to stdout, alongside the unchanged human log) and --only-paths FILE (in --apply mode, only actually delete entries whose path is in FILE; without it, --apply deletes everything as before -- existing direct callers are unaffected). Without --only-paths the safety re-verification is free: --apply --only-paths re-runs all 4 passes from scratch on every invocation, so if a group's ranking changed since a scan (e.g. the old keep_path is gone), the fresh pass assigns the previously-"delete" path the KEEP role instead and the only-paths allowlist naming it is simply never consulted -- no duplicate ranking logic needed in the review service. spotify-genre.py: additive --json emitting one JSON line per genre change (dry-run or --apply) for genre_review_service to persist. pipeline_runner.run_job_capture(): like run_job() but captures stdout as text (still under the same shared lock, still writes a job_runs row) for callers that need to parse structured output rather than just log it. dedup_review_service.scan() persists dry-run candidates into dedup_runs/dedup_candidates. confirm_and_apply() re-checks confirmed candidates still exist before invoking --apply --only-paths, so nothing is ever deleted without an explicit confirm -- matches the false-negative- biased dedup preference. scheduler_service's maintenance:dedup job now goes through this (still dry-run only, every day). genre_review_service.run() wraps spotify-genre.py for both dry-run preview and the real scheduled --apply run, persisting every run's diff into genre_runs/genre_candidates either way -- genre writes keep their current auto-apply behavior (low-risk, reversible, GENRE_LOCK-protected) but are now reviewable after the fact. lock_artist_genre() gives a one-click revert path when a run gets something wrong. Added minimal routers+templates for /dedup (scan, review, confirm-and- delete) and /genres (preview, review, lock-old-genre). Verified end-to-end against REAL duplicate files (not mocked): built an actual FLAC+MP3 duplicate pair in a real beets library, ran dedup-library.sh --json and confirmed correct JSON output, verified --apply --only-paths with an empty confirm list deletes nothing and with the real confirmed path deletes exactly that file (DB + disk) while preserving the FLAC, and ran the full dedup_review_service scan->confirm->apply flow through the same fixture. genre_review_service and spotify-genre.py --json verified against mocked/direct output (spotify-genre.py's own artist-genre lookup needs a live Spotify API call, out of reach in this sandbox). Confirmed the full app boots with all five routers registered.
152 lines
5.4 KiB
Python
152 lines
5.4 KiB
Python
import json
|
|
import time
|
|
from pathlib import Path
|
|
|
|
from sqlalchemy import select
|
|
|
|
from app.db import SessionLocal
|
|
from app.models import DedupCandidate, DedupRun
|
|
from app.services import pipeline_runner
|
|
from app.settings import settings
|
|
|
|
_SCRIPT = "dedup-library.sh"
|
|
|
|
|
|
def _parse_json_lines(output: str) -> list[dict]:
|
|
candidates = []
|
|
for line in output.splitlines():
|
|
line = line.strip()
|
|
if not line.startswith("{"):
|
|
continue
|
|
try:
|
|
candidates.append(json.loads(line))
|
|
except json.JSONDecodeError:
|
|
continue
|
|
return candidates
|
|
|
|
|
|
async def scan(triggered_by: str = "manual") -> DedupRun:
|
|
"""Dry-run dedup-library.sh --json, persist every candidate deletion
|
|
into a fresh dedup_runs/dedup_candidates pair. Never deletes anything
|
|
-- the scheduled maintenance:dedup job also only ever calls this (no
|
|
--apply), matching the false-negative-biased dedup preference; actual
|
|
deletion only ever happens through confirm_and_apply() below."""
|
|
script = str(settings.pipeline_dir / "lib" / _SCRIPT)
|
|
job_run, output = await pipeline_runner.run_job_capture(
|
|
"dedup:scan", [script, "--json"], triggered_by=triggered_by
|
|
)
|
|
candidates = _parse_json_lines(output)
|
|
|
|
db = SessionLocal()
|
|
try:
|
|
dedup_run = DedupRun(
|
|
started_at=job_run.started_at,
|
|
finished_at=job_run.finished_at,
|
|
mode="dry_run",
|
|
groups_found=len({(c["pass"], c["keep_path"]) for c in candidates}),
|
|
kept=len({(c["pass"], c["keep_path"]) for c in candidates}),
|
|
deleted=0,
|
|
log_path=job_run.log_path,
|
|
)
|
|
db.add(dedup_run)
|
|
db.commit()
|
|
db.refresh(dedup_run)
|
|
|
|
for c in candidates:
|
|
db.add(
|
|
DedupCandidate(
|
|
dedup_run_id=dedup_run.id,
|
|
pass_name=c.get("pass", "unknown"),
|
|
keep_path=c["keep_path"],
|
|
delete_path=c["delete_path"],
|
|
delete_size_bytes=c.get("delete_size_bytes"),
|
|
)
|
|
)
|
|
db.commit()
|
|
db.refresh(dedup_run)
|
|
return dedup_run
|
|
finally:
|
|
db.close()
|
|
|
|
|
|
async def confirm_and_apply(candidate_ids: list[int], confirmed_by: str) -> DedupRun | None:
|
|
"""Apply only the confirmed candidate deletions.
|
|
|
|
Cheap pre-check here: skip anything where delete_path or keep_path no
|
|
longer exists (something already changed it since the scan). The real
|
|
re-verification of ranking happens for free inside dedup-library.sh
|
|
itself: --apply --only-paths re-runs all 4 passes from scratch and
|
|
recomputes keep/delete for every group before consulting the only-paths
|
|
allowlist, so if a group's ranking flipped since the scan (e.g. the old
|
|
keep_path is gone and delete_path is now the last copy), the script's
|
|
fresh pass will assign delete_path the KEEP role instead -- it never
|
|
reaches a DELETE branch for it, so --only-paths naming it is simply
|
|
never consulted. No duplicate ranking logic needed here.
|
|
"""
|
|
db = SessionLocal()
|
|
try:
|
|
candidates = [db.get(DedupCandidate, cid) for cid in candidate_ids]
|
|
candidates = [c for c in candidates if c is not None and not c.applied]
|
|
|
|
now = time.time()
|
|
confirmed_candidates = []
|
|
for c in candidates:
|
|
if not Path(c.delete_path).exists() or not Path(c.keep_path).exists():
|
|
continue
|
|
c.confirmed = True
|
|
c.confirmed_by = confirmed_by
|
|
c.confirmed_at = now
|
|
confirmed_candidates.append(c)
|
|
db.commit()
|
|
|
|
if not confirmed_candidates:
|
|
return None
|
|
|
|
confirm_file = settings.logs_dir / f"dedup-confirm-{int(now * 1000)}.txt"
|
|
confirm_file.parent.mkdir(parents=True, exist_ok=True)
|
|
confirm_file.write_text("\n".join(c.delete_path for c in confirmed_candidates) + "\n")
|
|
|
|
script = str(settings.pipeline_dir / "lib" / _SCRIPT)
|
|
job_run, _output = await pipeline_runner.run_job_capture(
|
|
"dedup:apply",
|
|
[script, "--apply", "--only-paths", str(confirm_file), "--json"],
|
|
triggered_by=f"manual:{confirmed_by}",
|
|
)
|
|
|
|
still_there = {c.delete_path for c in confirmed_candidates if Path(c.delete_path).exists()}
|
|
actually_deleted = 0
|
|
for c in confirmed_candidates:
|
|
if c.delete_path not in still_there:
|
|
c.applied = True
|
|
actually_deleted += 1
|
|
db.commit()
|
|
|
|
apply_run = DedupRun(
|
|
started_at=job_run.started_at,
|
|
finished_at=job_run.finished_at,
|
|
mode="apply",
|
|
deleted=actually_deleted,
|
|
kept=len(confirmed_candidates) - actually_deleted,
|
|
log_path=job_run.log_path,
|
|
)
|
|
db.add(apply_run)
|
|
db.commit()
|
|
db.refresh(apply_run)
|
|
return apply_run
|
|
finally:
|
|
db.close()
|
|
|
|
|
|
def list_pending_candidates(dedup_run_id: int | None = None) -> list[DedupCandidate]:
|
|
db = SessionLocal()
|
|
try:
|
|
query = select(DedupCandidate).where(
|
|
DedupCandidate.applied == False, # noqa: E712
|
|
DedupCandidate.confirmed == False, # noqa: E712
|
|
)
|
|
if dedup_run_id is not None:
|
|
query = query.where(DedupCandidate.dedup_run_id == dedup_run_id)
|
|
return list(db.execute(query).scalars())
|
|
finally:
|
|
db.close()
|