Add beets_service, spotify_client, status_service
beets_service: read-only SQLite queries against the beets DB (mode=ro, matches the WAL setup in db.py's enable_beets_db_wal). Decodes the `path` column, which beets stores as a BLOB not TEXT. stats() gives dashboard/ migration-verification parity with `beet stats`. spotify_client: client-credentials OAuth (token cached in-process), paginated playlist-track fetch. Verified with mocked httpx responses: pagination across pages, null-track filtering (removed/local tracks), ISRC extraction, and token reuse across calls. status_service: reconciles a playlist's live Spotify tracklist against beets (IN_LIBRARY, by ISRC or normalized artist+title), sldl's per-playlist _sldl.m3u8 index (DOWNLOADED_PENDING_IMPORT), and the tag-guard quarantine dir (QUARANTINED, matched by filename since quarantined files have no tags by definition) -- everything else is NOT_YET_ATTEMPTED. ATTEMPTED_NO_MATCH is deliberately not implemented (the only signal is grepping sldl's per-run text logs, which the migration plan already flags as unreliable rather than authoritative). Verified end-to-end against a fake beets DB + sldl index + quarantine dir: all four statuses reconcile correctly, including both the ISRC-match and normalized-artist+title-fallback paths.
This commit is contained in:
@@ -0,0 +1,121 @@
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from app.services import beets_service, spotify_client
|
||||
from app.settings import settings
|
||||
|
||||
IN_LIBRARY = "IN_LIBRARY"
|
||||
QUARANTINED = "QUARANTINED"
|
||||
DOWNLOADED_PENDING_IMPORT = "DOWNLOADED_PENDING_IMPORT"
|
||||
NOT_YET_ATTEMPTED = "NOT_YET_ATTEMPTED"
|
||||
|
||||
ALL_STATUSES = [IN_LIBRARY, QUARANTINED, DOWNLOADED_PENDING_IMPORT, NOT_YET_ATTEMPTED]
|
||||
|
||||
# ATTEMPTED_NO_MATCH (a track sldl searched for and found nothing) is
|
||||
# deliberately not implemented: the only signal for it is grepping sldl's
|
||||
# per-run text logs for "no results" lines, which the migration plan flags
|
||||
# as best-effort/unreliable rather than authoritative. Tracks that were
|
||||
# searched-and-missed and tracks that haven't been searched yet both show
|
||||
# as NOT_YET_ATTEMPTED for now.
|
||||
|
||||
_STRIP_PARENS_RE = re.compile(r"\s*\([^)]*\)")
|
||||
_STRIP_BRACKETS_RE = re.compile(r"\s*\[[^\]]*\]")
|
||||
_FEAT_RE = re.compile(r"\s*(feat\.?|ft\.?)\s.*", re.IGNORECASE)
|
||||
_PUNCT_RE = re.compile(r"[\s\-_,;:&'\"!?.()\[\]/]+")
|
||||
|
||||
|
||||
def _normalize(s: str) -> str:
|
||||
s = (s or "").lower()
|
||||
s = _STRIP_PARENS_RE.sub("", s)
|
||||
s = _STRIP_BRACKETS_RE.sub("", s)
|
||||
s = _FEAT_RE.sub("", s)
|
||||
s = _PUNCT_RE.sub("", s)
|
||||
return s
|
||||
|
||||
|
||||
def _artist_title_key(artist: str, title: str) -> str:
|
||||
return f"{_normalize(artist)}|{_normalize(title)}"
|
||||
|
||||
|
||||
def _key_from_filename_stem(stem: str) -> str | None:
|
||||
"""'Artist - Title' -> matching key. sldl and quarantined files both
|
||||
follow this convention (sldl's name-format template; quarantined files
|
||||
are untagged sldl downloads, so the filename is usually all that's
|
||||
left to match on)."""
|
||||
if " - " not in stem:
|
||||
return None
|
||||
artist, title = stem.split(" - ", 1)
|
||||
return _artist_title_key(artist, title)
|
||||
|
||||
|
||||
def _sldl_index(playlist_name: str) -> set[str]:
|
||||
m3u8_path = settings.music_data_dir / "sldl-dropbox" / playlist_name / "_sldl.m3u8"
|
||||
if not m3u8_path.exists():
|
||||
return set()
|
||||
keys = set()
|
||||
for line in m3u8_path.read_text(errors="replace").splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
key = _key_from_filename_stem(Path(line).stem)
|
||||
if key:
|
||||
keys.add(key)
|
||||
return keys
|
||||
|
||||
|
||||
def _quarantine_index() -> set[str]:
|
||||
quarantine_dir = settings.music_data_dir / "Songs" / "untagged"
|
||||
if not quarantine_dir.exists():
|
||||
return set()
|
||||
keys = set()
|
||||
for f in quarantine_dir.rglob("*"):
|
||||
if not f.is_file():
|
||||
continue
|
||||
key = _key_from_filename_stem(f.stem)
|
||||
if key:
|
||||
keys.add(key)
|
||||
return keys
|
||||
|
||||
|
||||
def _beets_index(playlist_name: str) -> dict[str, dict]:
|
||||
index: dict[str, dict] = {}
|
||||
for item in beets_service.query_items(grouping=playlist_name):
|
||||
key = _artist_title_key(item.get("artist", ""), item.get("title", ""))
|
||||
index[key] = item
|
||||
if item.get("isrc"):
|
||||
index[f"isrc:{item['isrc']}"] = item
|
||||
return index
|
||||
|
||||
|
||||
def playlist_status(db: Session, playlist_name: str, spotify_url: str) -> dict:
|
||||
"""Reconcile one playlist's Spotify tracklist against beets (in-library),
|
||||
sldl's per-playlist index (downloaded, not yet imported), and the
|
||||
quarantine dir (downloaded, missing tags). Returns
|
||||
{"tracks": [{...track, "status": ...}], "counts": {status: count}}."""
|
||||
spotify_tracks = spotify_client.get_playlist_tracks(db, spotify_url)
|
||||
beets_index = _beets_index(playlist_name)
|
||||
sldl_index = _sldl_index(playlist_name)
|
||||
quarantine_index = _quarantine_index()
|
||||
|
||||
counts = {status: 0 for status in ALL_STATUSES}
|
||||
tracks = []
|
||||
|
||||
for track in spotify_tracks:
|
||||
key = _artist_title_key(track["artist"], track["title"])
|
||||
isrc_key = f"isrc:{track['isrc']}" if track.get("isrc") else None
|
||||
|
||||
if (isrc_key and isrc_key in beets_index) or key in beets_index:
|
||||
status = IN_LIBRARY
|
||||
elif key in quarantine_index:
|
||||
status = QUARANTINED
|
||||
elif key in sldl_index:
|
||||
status = DOWNLOADED_PENDING_IMPORT
|
||||
else:
|
||||
status = NOT_YET_ATTEMPTED
|
||||
|
||||
tracks.append({**track, "status": status})
|
||||
counts[status] += 1
|
||||
|
||||
return {"tracks": tracks, "counts": counts}
|
||||
Reference in New Issue
Block a user