Add dedup_review_service and genre_review_service

dedup-library.sh: additive --json (one NDJSON line per candidate deletion
to stdout, alongside the unchanged human log) and --only-paths FILE (in
--apply mode, only actually delete entries whose path is in FILE; without
it, --apply deletes everything as before -- existing direct callers are
unaffected). Without --only-paths the safety re-verification is free:
--apply --only-paths re-runs all 4 passes from scratch on every invocation,
so if a group's ranking changed since a scan (e.g. the old keep_path is
gone), the fresh pass assigns the previously-"delete" path the KEEP role
instead and the only-paths allowlist naming it is simply never consulted --
no duplicate ranking logic needed in the review service.

spotify-genre.py: additive --json emitting one JSON line per genre change
(dry-run or --apply) for genre_review_service to persist.

pipeline_runner.run_job_capture(): like run_job() but captures stdout as
text (still under the same shared lock, still writes a job_runs row) for
callers that need to parse structured output rather than just log it.

dedup_review_service.scan() persists dry-run candidates into
dedup_runs/dedup_candidates. confirm_and_apply() re-checks confirmed
candidates still exist before invoking --apply --only-paths, so nothing is
ever deleted without an explicit confirm -- matches the false-negative-
biased dedup preference. scheduler_service's maintenance:dedup job now
goes through this (still dry-run only, every day).

genre_review_service.run() wraps spotify-genre.py for both dry-run preview
and the real scheduled --apply run, persisting every run's diff into
genre_runs/genre_candidates either way -- genre writes keep their current
auto-apply behavior (low-risk, reversible, GENRE_LOCK-protected) but are
now reviewable after the fact. lock_artist_genre() gives a one-click revert
path when a run gets something wrong.

Added minimal routers+templates for /dedup (scan, review, confirm-and-
delete) and /genres (preview, review, lock-old-genre).

Verified end-to-end against REAL duplicate files (not mocked): built an
actual FLAC+MP3 duplicate pair in a real beets library, ran dedup-library.sh
--json and confirmed correct JSON output, verified --apply --only-paths
with an empty confirm list deletes nothing and with the real confirmed path
deletes exactly that file (DB + disk) while preserving the FLAC, and ran
the full dedup_review_service scan->confirm->apply flow through the same
fixture. genre_review_service and spotify-genre.py --json verified against
mocked/direct output (spotify-genre.py's own artist-genre lookup needs a
live Spotify API call, out of reach in this sandbox). Confirmed the full
app boots with all five routers registered.
This commit is contained in:
andrew
2026-07-08 14:17:20 -06:00
parent 270c9c0ed1
commit 2075d6cf66
11 changed files with 575 additions and 7 deletions
+151
View File
@@ -0,0 +1,151 @@
import json
import time
from pathlib import Path
from sqlalchemy import select
from app.db import SessionLocal
from app.models import DedupCandidate, DedupRun
from app.services import pipeline_runner
from app.settings import settings
_SCRIPT = "dedup-library.sh"
def _parse_json_lines(output: str) -> list[dict]:
candidates = []
for line in output.splitlines():
line = line.strip()
if not line.startswith("{"):
continue
try:
candidates.append(json.loads(line))
except json.JSONDecodeError:
continue
return candidates
async def scan(triggered_by: str = "manual") -> DedupRun:
"""Dry-run dedup-library.sh --json, persist every candidate deletion
into a fresh dedup_runs/dedup_candidates pair. Never deletes anything
-- the scheduled maintenance:dedup job also only ever calls this (no
--apply), matching the false-negative-biased dedup preference; actual
deletion only ever happens through confirm_and_apply() below."""
script = str(settings.pipeline_dir / "lib" / _SCRIPT)
job_run, output = await pipeline_runner.run_job_capture(
"dedup:scan", [script, "--json"], triggered_by=triggered_by
)
candidates = _parse_json_lines(output)
db = SessionLocal()
try:
dedup_run = DedupRun(
started_at=job_run.started_at,
finished_at=job_run.finished_at,
mode="dry_run",
groups_found=len({(c["pass"], c["keep_path"]) for c in candidates}),
kept=len({(c["pass"], c["keep_path"]) for c in candidates}),
deleted=0,
log_path=job_run.log_path,
)
db.add(dedup_run)
db.commit()
db.refresh(dedup_run)
for c in candidates:
db.add(
DedupCandidate(
dedup_run_id=dedup_run.id,
pass_name=c.get("pass", "unknown"),
keep_path=c["keep_path"],
delete_path=c["delete_path"],
delete_size_bytes=c.get("delete_size_bytes"),
)
)
db.commit()
db.refresh(dedup_run)
return dedup_run
finally:
db.close()
async def confirm_and_apply(candidate_ids: list[int], confirmed_by: str) -> DedupRun | None:
"""Apply only the confirmed candidate deletions.
Cheap pre-check here: skip anything where delete_path or keep_path no
longer exists (something already changed it since the scan). The real
re-verification of ranking happens for free inside dedup-library.sh
itself: --apply --only-paths re-runs all 4 passes from scratch and
recomputes keep/delete for every group before consulting the only-paths
allowlist, so if a group's ranking flipped since the scan (e.g. the old
keep_path is gone and delete_path is now the last copy), the script's
fresh pass will assign delete_path the KEEP role instead -- it never
reaches a DELETE branch for it, so --only-paths naming it is simply
never consulted. No duplicate ranking logic needed here.
"""
db = SessionLocal()
try:
candidates = [db.get(DedupCandidate, cid) for cid in candidate_ids]
candidates = [c for c in candidates if c is not None and not c.applied]
now = time.time()
confirmed_candidates = []
for c in candidates:
if not Path(c.delete_path).exists() or not Path(c.keep_path).exists():
continue
c.confirmed = True
c.confirmed_by = confirmed_by
c.confirmed_at = now
confirmed_candidates.append(c)
db.commit()
if not confirmed_candidates:
return None
confirm_file = settings.logs_dir / f"dedup-confirm-{int(now * 1000)}.txt"
confirm_file.parent.mkdir(parents=True, exist_ok=True)
confirm_file.write_text("\n".join(c.delete_path for c in confirmed_candidates) + "\n")
script = str(settings.pipeline_dir / "lib" / _SCRIPT)
job_run, _output = await pipeline_runner.run_job_capture(
"dedup:apply",
[script, "--apply", "--only-paths", str(confirm_file), "--json"],
triggered_by=f"manual:{confirmed_by}",
)
still_there = {c.delete_path for c in confirmed_candidates if Path(c.delete_path).exists()}
actually_deleted = 0
for c in confirmed_candidates:
if c.delete_path not in still_there:
c.applied = True
actually_deleted += 1
db.commit()
apply_run = DedupRun(
started_at=job_run.started_at,
finished_at=job_run.finished_at,
mode="apply",
deleted=actually_deleted,
kept=len(confirmed_candidates) - actually_deleted,
log_path=job_run.log_path,
)
db.add(apply_run)
db.commit()
db.refresh(apply_run)
return apply_run
finally:
db.close()
def list_pending_candidates(dedup_run_id: int | None = None) -> list[DedupCandidate]:
db = SessionLocal()
try:
query = select(DedupCandidate).where(
DedupCandidate.applied == False, # noqa: E712
DedupCandidate.confirmed == False, # noqa: E712
)
if dedup_run_id is not None:
query = query.where(DedupCandidate.dedup_run_id == dedup_run_id)
return list(db.execute(query).scalars())
finally:
db.close()
+83
View File
@@ -0,0 +1,83 @@
import json
from app.db import SessionLocal
from app.models import GenreCandidate, GenreRun
from app.services import genre_fix, pipeline_runner
from app.settings import settings
_SCRIPT = "spotify-genre.py"
def _parse_json_lines(output: str) -> list[dict]:
candidates = []
for line in output.splitlines():
line = line.strip()
if not line.startswith("{"):
continue
try:
candidates.append(json.loads(line))
except json.JSONDecodeError:
continue
return candidates
async def run(apply: bool, force: bool = False, triggered_by: str = "schedule") -> GenreRun:
"""Run spotify-genre.py --json (dry-run or --apply), persisting the
diff into a fresh genre_runs/genre_candidates pair either way.
Unlike dedup, genre writes stay auto-apply on the weekly schedule
(low-risk/reversible, GENRE_LOCK-protected) -- this wraps that same
scheduled run so its diff becomes reviewable after the fact, rather
than being a separate preview-only mode. scheduler_service's
maintenance:spotify_genre job calls this with apply=True."""
args = ["--json"]
if apply:
args.append("--apply")
if force:
args.append("--force")
script = str(settings.pipeline_dir / "lib" / _SCRIPT)
job_run, output = await pipeline_runner.run_job_capture(
"genre:run", [script] + args, triggered_by=triggered_by
)
candidates = _parse_json_lines(output)
db = SessionLocal()
try:
genre_run = GenreRun(
started_at=job_run.started_at,
finished_at=job_run.finished_at,
mode="apply" if apply else "dry_run",
written=len(candidates) if apply else 0,
unchanged=None,
no_match=None,
no_meta=None,
log_path=job_run.log_path,
)
db.add(genre_run)
db.commit()
db.refresh(genre_run)
for c in candidates:
db.add(
GenreCandidate(
genre_run_id=genre_run.id,
artist=c.get("artist"),
title=c.get("title"),
old_genre=c.get("old_genre"),
new_genre=c.get("new_genre"),
)
)
db.commit()
db.refresh(genre_run)
return genre_run
finally:
db.close()
async def lock_artist_genre(artist: str, genre: str, changed_by: str):
"""One-click 'lock this artist's genre' action from the review UI, for
when a scheduled run got something wrong -- wraps
genre_fix.set_artist_genre (fix-genre.sh sets GENRE_LOCK=1 as part of
applying, which is what makes future spotify-genre runs leave it alone)."""
return await genre_fix.set_artist_genre(artist, genre, changed_by)
+89
View File
@@ -143,6 +143,95 @@ async def run_job(
_lock.release()
async def run_job_capture(
job_key: str,
argv: list[str],
triggered_by: str = "manual",
timeout: float | None = None,
) -> tuple[JobRun, str]:
"""Like run_job(), but captures stdout+stderr as text and returns it
alongside the JobRun, instead of only writing it to the log file --
for callers that need to parse structured (JSON-lines) output, e.g.
dedup_review_service's dry-run scan. The full output is still written
to a log file afterward so job_runs.log_path works the same as any
other job. Shares the same lock as run_job()."""
started_at = time.time()
if not await _try_acquire_nowait():
db = SessionLocal()
run = JobRun(
job_key=job_key,
started_at=started_at,
finished_at=started_at,
status="skipped_lock",
triggered_by=triggered_by,
)
db.add(run)
db.commit()
db.refresh(run)
db.close()
return run, ""
try:
log_dir = settings.logs_dir / job_key.replace(":", "_")
log_dir.mkdir(parents=True, exist_ok=True)
log_path = log_dir / f"{time.strftime('%Y%m%d-%H%M%S')}.log"
db = SessionLocal()
run = JobRun(
job_key=job_key,
started_at=started_at,
status="running",
triggered_by=triggered_by,
log_path=str(log_path),
)
db.add(run)
db.commit()
db.refresh(run)
run_id = run.id
db.close()
output_text = ""
exit_code: int | None
try:
proc = await asyncio.create_subprocess_exec(
*argv,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.STDOUT,
env=_subprocess_env(),
)
try:
stdout_bytes, _ = await asyncio.wait_for(proc.communicate(), timeout=timeout)
exit_code = proc.returncode
output_text = stdout_bytes.decode(errors="replace")
except asyncio.TimeoutError:
proc.kill()
await proc.wait()
exit_code = -1
output_text += f"\n[pipeline_runner] TIMEOUT after {timeout}s -- process killed\n"
status = "success" if exit_code == 0 else "failed"
except Exception as exc:
exit_code = None
status = "failed"
output_text += f"\n[pipeline_runner] exception before/while running: {exc!r}\n"
log_path.write_text(output_text)
finished_at = time.time()
db = SessionLocal()
run = db.get(JobRun, run_id)
run.finished_at = finished_at
run.status = status
run.exit_code = exit_code
run.summary = _summarize_log(log_path)
db.commit()
db.refresh(run)
db.close()
return run, output_text
finally:
_lock.release()
async def run_playlist(playlist_name: str, no_m3u: bool = False, triggered_by: str = "schedule") -> JobRun:
script = settings.pipeline_dir / "bin" / "run-playlist.sh"
argv = [str(script)]
+18 -3
View File
@@ -6,7 +6,7 @@ from sqlalchemy import select
from app.db import SessionLocal
from app.models import Playlist, ScheduledJob
from app.services import pipeline_runner
from app.services import dedup_review_service, genre_review_service, pipeline_runner
from app.settings import settings
TIMEZONE = "America/Edmonton"
@@ -44,6 +44,21 @@ def _beet(job_key: str, args: list[str]):
return _run
async def _dedup_scan(triggered_by: str = "schedule"):
"""Dry-run only, ever, on the schedule -- see module docstring. Populates
dedup_candidates for review; deletion only ever happens through a
confirmed dedup_review_service.confirm_and_apply() call from the UI."""
return await dedup_review_service.scan(triggered_by=triggered_by)
async def _genre_run(triggered_by: str = "schedule"):
"""Keeps its current auto-apply behavior (unlike dedup) -- genre writes
are low-risk/reversible and GENRE_LOCK-protected -- but now goes through
genre_review_service so every run's diff is persisted for after-the-fact
review instead of only living in a log file."""
return await genre_review_service.run(apply=True, force=True, triggered_by=triggered_by)
async def _log_rotation(triggered_by: str = "schedule"):
"""Replaces `find /var/log/sldl -name '*.log' -mtime +30 -delete`."""
cutoff = time.time() - 30 * 86400
@@ -68,7 +83,7 @@ MAINTENANCE_JOBS: dict[str, tuple[dict, callable]] = {
),
"maintenance:dedup": (
dict(minute=55, hour=8),
_lib("maintenance:dedup", "dedup-library.sh"), # no --apply: dry-run only
_dedup_scan, # no --apply, ever, on schedule -- see _dedup_scan docstring
),
"maintenance:gen_djmix_playlist": (
dict(minute=57, hour=8),
@@ -125,7 +140,7 @@ MAINTENANCE_JOBS: dict[str, tuple[dict, callable]] = {
),
"maintenance:spotify_genre": (
dict(minute=50, hour=8, day_of_week="sun"),
_lib("maintenance:spotify_genre", "spotify-genre.py", ["--apply", "--force"]),
_genre_run, # goes through genre_review_service -- see its docstring
),
"maintenance:beets_update_sync": (
dict(minute=53, hour=8, day_of_week="sun"),