R7: consolidate duplicated Spotify-token and NDJSON-parse logic

- The client-credentials token fetch was copy-pasted across spotify-retag.py,
  spotify-genre.py, and fix-track-metadata.py (and the app's spotify_client).
  Add pipeline/lib/_spotify_auth.get_token (cached per id/secret); the three
  scripts now source their own credentials but delegate the request to it. The
  scripts run with pipeline/lib on sys.path, so the plain `from _spotify_auth
  import get_token` resolves.
- The identical _parse_json_lines helper in dedup_review_service and
  genre_review_service is now a single app/services/_ndjson.parse_json_lines.

Verified: unit test of the token helper (cache + request), the NDJSON parser
(tests/test_ndjson.py), full suite green (40), and a live spotify-genre dry-run
that fetched a token and queried Spotify through the shared helper.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
andrew
2026-07-10 16:08:15 -06:00
parent e49d12a72b
commit bcd07750bf
8 changed files with 83 additions and 66 deletions
+35
View File
@@ -0,0 +1,35 @@
"""Shared Spotify Client-Credentials token fetch for the pipeline scripts.
The scripts (spotify-retag.py, spotify-genre.py, fix-track-metadata.py) each
run as `python /app/pipeline/lib/<name>.py`, so this directory is on sys.path
and they can `from _spotify_auth import get_token`. Each still sources its own
credentials (env vars vs a rendered .env file); only the token request itself,
which was copy-pasted four ways, lives here.
"""
import base64
import json
import urllib.parse
import urllib.request
_TOKEN_URL = "https://accounts.spotify.com/api/token"
_cache: dict[tuple[str, str], str] = {}
def get_token(client_id: str, client_secret: str) -> str:
"""Return a client-credentials access token, cached per (id, secret) for
the life of the process."""
key = (client_id, client_secret)
if key in _cache:
return _cache[key]
creds = base64.b64encode(f"{client_id}:{client_secret}".encode()).decode()
req = urllib.request.Request(
_TOKEN_URL,
data=urllib.parse.urlencode({"grant_type": "client_credentials"}).encode(),
headers={
"Authorization": f"Basic {creds}",
"Content-Type": "application/x-www-form-urlencoded",
},
)
with urllib.request.urlopen(req, timeout=15) as r:
_cache[key] = json.loads(r.read())["access_token"]
return _cache[key]
+3 -8
View File
@@ -36,6 +36,8 @@ Supported URLs:
"""
import sys, os, re, json, base64, argparse, subprocess, unicodedata, urllib.parse, urllib.request
from _spotify_auth import get_token
from mutagen import File as MFile
from mutagen.id3 import ID3, ID3NoHeaderError, TPE1, TPE2, TALB, TIT2, TRCK, TDRC, TCON, APIC
from mutagen.flac import FLAC, Picture
@@ -87,14 +89,7 @@ def _spotify_token():
continue
k, v = line.split("=", 1)
env[k] = v.strip().strip("'").strip('"')
cid, csec = env["SPOTIFY_CLIENT_ID"], env["SPOTIFY_CLIENT_SECRET"]
creds = base64.b64encode(f"{cid}:{csec}".encode()).decode()
req = urllib.request.Request(
"https://accounts.spotify.com/api/token",
data=urllib.parse.urlencode({"grant_type": "client_credentials"}).encode(),
headers={"Authorization": f"Basic {creds}",
"Content-Type": "application/x-www-form-urlencoded"})
return json.loads(urllib.request.urlopen(req, timeout=15).read())["access_token"]
return get_token(env["SPOTIFY_CLIENT_ID"], env["SPOTIFY_CLIENT_SECRET"])
def _spotify_get(path, token):
+4 -13
View File
@@ -25,8 +25,10 @@ Usage:
spotify-genre.py --apply --force # overwrite existing GENRE tags
spotify-genre.py --apply --query '...' # subset by beets query
"""
import sys, os, re, json, argparse, base64, urllib.parse, urllib.request, subprocess
import sys, os, re, json, argparse, urllib.parse, urllib.request, subprocess
from pathlib import Path
from _spotify_auth import get_token
from mutagen import File as MFile
from mutagen.id3 import ID3, ID3NoHeaderError, TCON
from mutagen.flac import FLAC
@@ -73,19 +75,8 @@ def load_whitelist():
return out
_token = None
def _spotify_token():
global _token
if _token: return _token
creds = base64.b64encode(f"{SPOTIFY_CID}:{SPOTIFY_CSEC}".encode()).decode()
req = urllib.request.Request(
"https://accounts.spotify.com/api/token",
data=urllib.parse.urlencode({"grant_type":"client_credentials"}).encode(),
headers={"Authorization":f"Basic {creds}",
"Content-Type":"application/x-www-form-urlencoded"})
with urllib.request.urlopen(req, timeout=15) as r:
_token = json.loads(r.read())["access_token"]
return _token
return get_token(SPOTIFY_CID, SPOTIFY_CSEC)
# Cache: artist_name → list[str] of Spotify genres (or None if not found)
+3 -15
View File
@@ -31,27 +31,15 @@ Matching strategy (per file):
use it. If multiple, score each by artist-string overlap and pick the
best. If nothing close, leave the file alone.
"""
import sys, os, re, json, base64, urllib.request, urllib.parse, subprocess
import sys, os, re, json, urllib.request, urllib.parse, subprocess
from pathlib import Path
from difflib import SequenceMatcher
from _spotify_auth import get_token
API = "https://api.spotify.com/v1"
def get_token(cid: str, csec: str) -> str:
creds = base64.b64encode(f"{cid}:{csec}".encode()).decode()
req = urllib.request.Request(
"https://accounts.spotify.com/api/token",
data=urllib.parse.urlencode({"grant_type": "client_credentials"}).encode(),
headers={
"Authorization": f"Basic {creds}",
"Content-Type": "application/x-www-form-urlencoded",
},
)
with urllib.request.urlopen(req, timeout=15) as r:
return json.loads(r.read())["access_token"]
def fetch_playlist(url: str, token: str) -> list[dict]:
m = re.search(r"playlist[/:]([A-Za-z0-9]+)", url)
if not m: