2075d6cf66
dedup-library.sh: additive --json (one NDJSON line per candidate deletion to stdout, alongside the unchanged human log) and --only-paths FILE (in --apply mode, only actually delete entries whose path is in FILE; without it, --apply deletes everything as before -- existing direct callers are unaffected). Without --only-paths the safety re-verification is free: --apply --only-paths re-runs all 4 passes from scratch on every invocation, so if a group's ranking changed since a scan (e.g. the old keep_path is gone), the fresh pass assigns the previously-"delete" path the KEEP role instead and the only-paths allowlist naming it is simply never consulted -- no duplicate ranking logic needed in the review service. spotify-genre.py: additive --json emitting one JSON line per genre change (dry-run or --apply) for genre_review_service to persist. pipeline_runner.run_job_capture(): like run_job() but captures stdout as text (still under the same shared lock, still writes a job_runs row) for callers that need to parse structured output rather than just log it. dedup_review_service.scan() persists dry-run candidates into dedup_runs/dedup_candidates. confirm_and_apply() re-checks confirmed candidates still exist before invoking --apply --only-paths, so nothing is ever deleted without an explicit confirm -- matches the false-negative- biased dedup preference. scheduler_service's maintenance:dedup job now goes through this (still dry-run only, every day). genre_review_service.run() wraps spotify-genre.py for both dry-run preview and the real scheduled --apply run, persisting every run's diff into genre_runs/genre_candidates either way -- genre writes keep their current auto-apply behavior (low-risk, reversible, GENRE_LOCK-protected) but are now reviewable after the fact. lock_artist_genre() gives a one-click revert path when a run gets something wrong. Added minimal routers+templates for /dedup (scan, review, confirm-and- delete) and /genres (preview, review, lock-old-genre). Verified end-to-end against REAL duplicate files (not mocked): built an actual FLAC+MP3 duplicate pair in a real beets library, ran dedup-library.sh --json and confirmed correct JSON output, verified --apply --only-paths with an empty confirm list deletes nothing and with the real confirmed path deletes exactly that file (DB + disk) while preserving the FLAC, and ran the full dedup_review_service scan->confirm->apply flow through the same fixture. genre_review_service and spotify-genre.py --json verified against mocked/direct output (spotify-genre.py's own artist-genre lookup needs a live Spotify API call, out of reach in this sandbox). Confirmed the full app boots with all five routers registered.
84 lines
2.7 KiB
Python
84 lines
2.7 KiB
Python
import json
|
|
|
|
from app.db import SessionLocal
|
|
from app.models import GenreCandidate, GenreRun
|
|
from app.services import genre_fix, pipeline_runner
|
|
from app.settings import settings
|
|
|
|
_SCRIPT = "spotify-genre.py"
|
|
|
|
|
|
def _parse_json_lines(output: str) -> list[dict]:
|
|
candidates = []
|
|
for line in output.splitlines():
|
|
line = line.strip()
|
|
if not line.startswith("{"):
|
|
continue
|
|
try:
|
|
candidates.append(json.loads(line))
|
|
except json.JSONDecodeError:
|
|
continue
|
|
return candidates
|
|
|
|
|
|
async def run(apply: bool, force: bool = False, triggered_by: str = "schedule") -> GenreRun:
|
|
"""Run spotify-genre.py --json (dry-run or --apply), persisting the
|
|
diff into a fresh genre_runs/genre_candidates pair either way.
|
|
|
|
Unlike dedup, genre writes stay auto-apply on the weekly schedule
|
|
(low-risk/reversible, GENRE_LOCK-protected) -- this wraps that same
|
|
scheduled run so its diff becomes reviewable after the fact, rather
|
|
than being a separate preview-only mode. scheduler_service's
|
|
maintenance:spotify_genre job calls this with apply=True."""
|
|
args = ["--json"]
|
|
if apply:
|
|
args.append("--apply")
|
|
if force:
|
|
args.append("--force")
|
|
|
|
script = str(settings.pipeline_dir / "lib" / _SCRIPT)
|
|
job_run, output = await pipeline_runner.run_job_capture(
|
|
"genre:run", [script] + args, triggered_by=triggered_by
|
|
)
|
|
candidates = _parse_json_lines(output)
|
|
|
|
db = SessionLocal()
|
|
try:
|
|
genre_run = GenreRun(
|
|
started_at=job_run.started_at,
|
|
finished_at=job_run.finished_at,
|
|
mode="apply" if apply else "dry_run",
|
|
written=len(candidates) if apply else 0,
|
|
unchanged=None,
|
|
no_match=None,
|
|
no_meta=None,
|
|
log_path=job_run.log_path,
|
|
)
|
|
db.add(genre_run)
|
|
db.commit()
|
|
db.refresh(genre_run)
|
|
|
|
for c in candidates:
|
|
db.add(
|
|
GenreCandidate(
|
|
genre_run_id=genre_run.id,
|
|
artist=c.get("artist"),
|
|
title=c.get("title"),
|
|
old_genre=c.get("old_genre"),
|
|
new_genre=c.get("new_genre"),
|
|
)
|
|
)
|
|
db.commit()
|
|
db.refresh(genre_run)
|
|
return genre_run
|
|
finally:
|
|
db.close()
|
|
|
|
|
|
async def lock_artist_genre(artist: str, genre: str, changed_by: str):
|
|
"""One-click 'lock this artist's genre' action from the review UI, for
|
|
when a scheduled run got something wrong -- wraps
|
|
genre_fix.set_artist_genre (fix-genre.sh sets GENRE_LOCK=1 as part of
|
|
applying, which is what makes future spotify-genre runs leave it alone)."""
|
|
return await genre_fix.set_artist_genre(artist, genre, changed_by)
|