Stage 0: migrate pipeline scripts from /opt/sldl, scaffold repo
Moves all ~30 pipeline scripts, configs, and the vendored sldl binary into this repo (source /opt/sldl left untouched). Removes all docker exec/docker compose dependencies now that beets and sldl run in-process/as a subprocess of this container instead of via soulbeet/on-demand sldl containers. Replaces hardcoded host paths, Navidrome credentials, and Spotify credential sourcing with env-var-driven paths and shared credential loaders. Adds Dockerfile, entrypoint.sh, requirements.txt, docker-compose.snippet.yml, and the initial app DB schema.
This commit is contained in:
Executable
+350
@@ -0,0 +1,350 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
spotify-genre.py — write GENRE tags onto every audio file in the library
|
||||
using Spotify's per-artist genre data, filtered against a whitelist.
|
||||
|
||||
Spotify only has genre at the ARTIST level, but their taxonomy is curated
|
||||
and clean (e.g. "tropical house", "tech house", "synthwave"). We match
|
||||
each artist's reported genres against our whitelist to pick up to 3
|
||||
canonical display strings, then write to every track by that artist.
|
||||
|
||||
Whitelist matching is fuzzy-normalized: lowercased, dashes/spaces stripped.
|
||||
So Spotify's "tropical house" matches our whitelist's "Tropical House"
|
||||
(or rolls into "House" if Tropical House isn't in the list).
|
||||
|
||||
Per-artist caching: each unique artist is looked up once, even if they
|
||||
have 100 tracks in the library.
|
||||
|
||||
Falls back gracefully:
|
||||
- Artist not on Spotify -> no change
|
||||
- Found but no whitelist match -> no change
|
||||
|
||||
Usage:
|
||||
spotify-genre.py # dry run, show plan
|
||||
spotify-genre.py --apply # write tags
|
||||
spotify-genre.py --apply --force # overwrite existing GENRE tags
|
||||
spotify-genre.py --apply --query '...' # subset by beets query
|
||||
"""
|
||||
import sys, os, re, json, argparse, base64, urllib.parse, urllib.request, subprocess
|
||||
from pathlib import Path
|
||||
from mutagen import File as MFile
|
||||
from mutagen.id3 import ID3, ID3NoHeaderError, TCON
|
||||
from mutagen.flac import FLAC
|
||||
from mutagen.mp4 import MP4
|
||||
from mutagen.oggopus import OggOpus
|
||||
from mutagen.oggvorbis import OggVorbis
|
||||
|
||||
# OGG/OPUS use Vorbis-comment-style tags (lowercase keys, multi-value), same
|
||||
# shape as FLAC. Treat them as a family so the read/write paths are shared.
|
||||
VORBIS_LIKE = (FLAC, OggOpus, OggVorbis)
|
||||
|
||||
ALEMBIC_CONFIG_DIR = os.environ.get("ALEMBIC_CONFIG_DIR", "/config")
|
||||
|
||||
|
||||
def _load_spotify_creds():
|
||||
"""Spotify creds, rendered by alembic's credential_service to _spotify.env."""
|
||||
path = f"{ALEMBIC_CONFIG_DIR}/pipeline/_spotify.env"
|
||||
env = {}
|
||||
with open(path) as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#") or "=" not in line:
|
||||
continue
|
||||
k, v = line.split("=", 1)
|
||||
env[k] = v.strip().strip("'").strip('"')
|
||||
return env["SPOTIFY_CLIENT_ID"], env["SPOTIFY_CLIENT_SECRET"]
|
||||
|
||||
SPOTIFY_CID, SPOTIFY_CSEC = _load_spotify_creds()
|
||||
WHITELIST_FILE = f"{ALEMBIC_CONFIG_DIR}/pipeline/genres-whitelist.txt"
|
||||
MAX_GENRES = 3
|
||||
SEPARATOR = "; "
|
||||
|
||||
|
||||
def load_whitelist():
|
||||
"""Load whitelist and build a normalized→display mapping for fuzzy matching."""
|
||||
out = {}
|
||||
with open(WHITELIST_FILE) as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
normalized = re.sub(r"[\s\-_&/.]+", "", line).lower()
|
||||
out[normalized] = line
|
||||
return out
|
||||
|
||||
|
||||
_token = None
|
||||
def _spotify_token():
|
||||
global _token
|
||||
if _token: return _token
|
||||
creds = base64.b64encode(f"{SPOTIFY_CID}:{SPOTIFY_CSEC}".encode()).decode()
|
||||
req = urllib.request.Request(
|
||||
"https://accounts.spotify.com/api/token",
|
||||
data=urllib.parse.urlencode({"grant_type":"client_credentials"}).encode(),
|
||||
headers={"Authorization":f"Basic {creds}",
|
||||
"Content-Type":"application/x-www-form-urlencoded"})
|
||||
with urllib.request.urlopen(req, timeout=15) as r:
|
||||
_token = json.loads(r.read())["access_token"]
|
||||
return _token
|
||||
|
||||
|
||||
# Cache: artist_name → list[str] of Spotify genres (or None if not found)
|
||||
_artist_genre_cache = {}
|
||||
|
||||
|
||||
def spotify_artist_genres(artist_name):
|
||||
"""Return Spotify's genre list for the artist, or None if not found.
|
||||
Cached per-artist so 1500 tracks for ~300 artists = ~300 API calls."""
|
||||
if artist_name in _artist_genre_cache:
|
||||
return _artist_genre_cache[artist_name]
|
||||
|
||||
# Search artist by name (limit=1, take best match)
|
||||
qs = urllib.parse.urlencode({
|
||||
"q": f'artist:"{artist_name}"',
|
||||
"type": "artist",
|
||||
"limit": 1,
|
||||
})
|
||||
url = f"https://api.spotify.com/v1/search?{qs}"
|
||||
req = urllib.request.Request(url, headers={"Authorization": f"Bearer {_spotify_token()}"})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=15) as r:
|
||||
data = json.loads(r.read())
|
||||
except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError, TimeoutError):
|
||||
_artist_genre_cache[artist_name] = None
|
||||
return None
|
||||
|
||||
items = data.get("artists", {}).get("items", [])
|
||||
if not items:
|
||||
_artist_genre_cache[artist_name] = None
|
||||
return None
|
||||
|
||||
# Validate: Spotify's fuzzy search will silently match short/ambiguous
|
||||
# names to the wrong artist (e.g. "G Jones" → "George Jones" the country
|
||||
# singer, "T78" → some random T78). Reject the result unless the returned
|
||||
# name matches what we asked for after normalizing whitespace/punct.
|
||||
def _norm(s):
|
||||
return re.sub(r"[\s\-_&/.()\[\]'\"]+", "", s or "").lower()
|
||||
requested = _norm(artist_name)
|
||||
returned = _norm(items[0].get("name", ""))
|
||||
if requested and returned and requested != returned:
|
||||
_artist_genre_cache[artist_name] = None
|
||||
return None
|
||||
|
||||
genres = items[0].get("genres") or []
|
||||
_artist_genre_cache[artist_name] = genres
|
||||
return genres
|
||||
|
||||
|
||||
# Map Spotify-style spellings to the normalized form of their whitelist entry.
|
||||
# Spotify uses "drum and bass" / "rock and roll" / "r&b" — our whitelist uses
|
||||
# "Drum & Bass" / "Rock & Roll" / "R&B". The "and"/"&" difference prevents the
|
||||
# roll-up match below, so we explicitly resolve common variants first.
|
||||
ALIASES = {
|
||||
"drumandbass": "drumbass",
|
||||
"rockandroll": "rockroll",
|
||||
"randb": "rb",
|
||||
"rnb": "rb",
|
||||
"hiphop": "hiphop",
|
||||
"ukgarage": "ukgarage",
|
||||
}
|
||||
|
||||
|
||||
def match_whitelist(genres, whitelist):
|
||||
"""Pick up to MAX_GENRES whitelist entries from Spotify's genre list,
|
||||
in Spotify's reported order (most-relevant first). Returns list of
|
||||
display strings.
|
||||
|
||||
For specific genres like 'tropical house' that aren't in the whitelist,
|
||||
we also check if any whitelist entry's normalized form is contained
|
||||
within the genre's normalized form — so 'tropical house' rolls up to
|
||||
'House' if Tropical House isn't a separate whitelist entry."""
|
||||
out = []
|
||||
seen = set()
|
||||
for g in genres or []:
|
||||
norm = re.sub(r"[\s\-_&/.]+", "", g).lower()
|
||||
# Resolve known aliases (e.g. "drum and bass" → "drumbass") before lookup
|
||||
norm = ALIASES.get(norm, norm)
|
||||
# Direct hit
|
||||
display = whitelist.get(norm)
|
||||
if display:
|
||||
if display not in seen:
|
||||
out.append(display); seen.add(display)
|
||||
continue
|
||||
# Roll-up: e.g. "tropicalhouse" contains "house"
|
||||
for wl_norm, wl_display in whitelist.items():
|
||||
if wl_norm and wl_norm in norm and len(wl_norm) >= 4:
|
||||
if wl_display not in seen:
|
||||
out.append(wl_display); seen.add(wl_display)
|
||||
break
|
||||
if len(out) >= MAX_GENRES:
|
||||
break
|
||||
return out[:MAX_GENRES]
|
||||
|
||||
|
||||
def is_genre_locked(filepath):
|
||||
"""Return True if the file has GENRE_LOCK=1, meaning manual genre should not be overwritten."""
|
||||
try:
|
||||
m = MFile(filepath)
|
||||
except Exception:
|
||||
return False
|
||||
if not m or not m.tags:
|
||||
return False
|
||||
try:
|
||||
if isinstance(m, VORBIS_LIKE):
|
||||
val = (m.tags.get("genre_lock") or m.tags.get("GENRE_LOCK") or [None])[0]
|
||||
return val == "1"
|
||||
else:
|
||||
# MP3: check TXXX:GENRE_LOCK
|
||||
t = getattr(m, "tags", None)
|
||||
if t is not None:
|
||||
for frame in t.getall("TXXX"):
|
||||
if frame.desc.upper() == "GENRE_LOCK" and frame.text and frame.text[0] == "1":
|
||||
return True
|
||||
except (AttributeError, IndexError, TypeError):
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def get_track_meta(filepath):
|
||||
"""Read (artist, title, current_genre) from the file. Returns (None,None,None) on failure."""
|
||||
try:
|
||||
m = MFile(filepath)
|
||||
except Exception:
|
||||
return None, None, None
|
||||
if not m or not m.tags:
|
||||
return None, None, None
|
||||
|
||||
artist = title = genre = None
|
||||
try:
|
||||
if isinstance(m, VORBIS_LIKE):
|
||||
# FLAC/OGG/OPUS share Vorbis-comment-style tags
|
||||
artist = (m.tags.get("artist") or [None])[0]
|
||||
title = (m.tags.get("title") or [None])[0]
|
||||
genre = (m.tags.get("genre") or [None])[0]
|
||||
elif isinstance(m, MP4):
|
||||
artist = (m.tags.get("©ART") or [None])[0]
|
||||
title = (m.tags.get("©nam") or [None])[0]
|
||||
genre = (m.tags.get("©gen") or [None])[0]
|
||||
else:
|
||||
# MP3 ID3v2 — read via .text[0] of the frame to avoid str() returning frame metadata
|
||||
t = getattr(m, "tags", None)
|
||||
if t is not None:
|
||||
if "TPE1" in t and t["TPE1"].text: artist = str(t["TPE1"].text[0])
|
||||
if "TIT2" in t and t["TIT2"].text: title = str(t["TIT2"].text[0])
|
||||
if "TCON" in t and t["TCON"].text: genre = str(t["TCON"].text[0])
|
||||
except (AttributeError, IndexError, TypeError):
|
||||
pass
|
||||
return artist, title, genre
|
||||
|
||||
|
||||
def write_genre(filepath, value):
|
||||
"""Write GENRE tag on the file."""
|
||||
try:
|
||||
m = MFile(filepath)
|
||||
except Exception:
|
||||
return False
|
||||
if isinstance(m, VORBIS_LIKE):
|
||||
m.tags["GENRE"] = value # mutagen Vorbis-style tags are case-insensitive
|
||||
m.save()
|
||||
elif isinstance(m, MP4):
|
||||
m.tags["©gen"] = value
|
||||
m.save()
|
||||
else:
|
||||
try:
|
||||
tags = ID3(filepath)
|
||||
except ID3NoHeaderError:
|
||||
tags = ID3()
|
||||
tags.delall("TCON")
|
||||
tags.add(TCON(encoding=3, text=value))
|
||||
tags.save(filepath)
|
||||
return True
|
||||
|
||||
|
||||
def list_tracks(query=None):
|
||||
"""Use beets (in-process, same container) to list tracks; paths are
|
||||
already valid in this container's filesystem — no translation needed."""
|
||||
args = ["beet", "ls", "-f", "$path"]
|
||||
if query:
|
||||
args.append(query)
|
||||
r = subprocess.run(args, capture_output=True, text=True, timeout=120)
|
||||
return [line for line in r.stdout.splitlines() if line.strip()]
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--apply", action="store_true", help="Actually write tags")
|
||||
ap.add_argument("--force", action="store_true",
|
||||
help="Overwrite existing GENRE tags (default: only fill empty)")
|
||||
ap.add_argument("--query", help="Beets query to limit which tracks are processed")
|
||||
ap.add_argument("--limit", type=int, help="Stop after N tracks (debug)")
|
||||
args = ap.parse_args()
|
||||
|
||||
whitelist = load_whitelist()
|
||||
print(f"[spotify-genre] whitelist has {len(whitelist)} entries")
|
||||
|
||||
paths = list_tracks(args.query)
|
||||
if args.limit:
|
||||
paths = paths[:args.limit]
|
||||
print(f"[spotify-genre] {len(paths)} tracks to scan")
|
||||
|
||||
written = unchanged = no_match = no_meta = 0
|
||||
for i, p in enumerate(paths, 1):
|
||||
if i % 100 == 0:
|
||||
print(f" ...{i}/{len(paths)}", flush=True)
|
||||
if not os.path.exists(p):
|
||||
continue
|
||||
|
||||
artist, title, current_genre = get_track_meta(p)
|
||||
if not artist or not title:
|
||||
no_meta += 1
|
||||
continue
|
||||
|
||||
# Skip if manually genre-locked (set via fix-genre.sh or direct GENRE_LOCK tag)
|
||||
if is_genre_locked(p):
|
||||
unchanged += 1
|
||||
continue
|
||||
|
||||
# Skip if already has whitelist-canonical genre and we're not forcing
|
||||
if current_genre and not args.force:
|
||||
cur_parts = [g.strip() for g in re.split(r"[;,]", current_genre)]
|
||||
cur_normed = {re.sub(r"[\s\-_&/.]+", "", g).lower() for g in cur_parts}
|
||||
if cur_normed & set(whitelist.keys()):
|
||||
unchanged += 1
|
||||
continue
|
||||
|
||||
# Multi-artist "X; Y" → use first as primary
|
||||
primary_artist = re.split(r"[;,/&]", artist)[0].strip()
|
||||
genres = spotify_artist_genres(primary_artist)
|
||||
if genres is None or not genres:
|
||||
no_match += 1
|
||||
continue
|
||||
|
||||
matched = match_whitelist(genres, whitelist)
|
||||
if not matched:
|
||||
no_match += 1
|
||||
continue
|
||||
|
||||
new_value = SEPARATOR.join(matched)
|
||||
if current_genre == new_value:
|
||||
unchanged += 1
|
||||
continue
|
||||
|
||||
if args.apply:
|
||||
ok = write_genre(p, new_value)
|
||||
if ok:
|
||||
written += 1
|
||||
# Quiet — only print every 50th
|
||||
if written % 50 == 0:
|
||||
print(f" [{written}] {primary_artist} - {title}: {new_value}", flush=True)
|
||||
else:
|
||||
print(f" WOULD-SET {primary_artist} - {title} → {new_value}")
|
||||
written += 1 # count as 'planned'
|
||||
|
||||
print(f"\n[spotify-genre] {('written' if args.apply else 'plan')}: {written}, "
|
||||
f"unchanged: {unchanged}, no-match: {no_match}, no-meta: {no_meta}")
|
||||
if args.apply:
|
||||
print("[spotify-genre] now run 'beet update' to sync the DB")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user