be33664be2
The daily digest is now formatted with Telegram HTML: bold section headers
with <pre>-aligned columns, colored circle emoji as status dots, a branded
header line, and a top blockquote callout (all clear vs N issues). All free
text is &/</> escaped so a stray angle bracket in a log snippet can't break
the parse and swallow the whole message. notify-telegram.sh gains an --html
flag (parse_mode=HTML) used by the digest send; plain-text callers are
unchanged. Truncation is now tag-safe in HTML mode: cut on a line boundary,
re-close an open <pre>, and use an escaped marker -- the old literal
"...<truncated>" tail would itself have 400'd the send.
Two real bugs surfaced while testing:
- pipeline-status.sh pins its own PATH, which lacked /opt/venv/bin where
beet lives, so every beet probe ("added today", "Library by format", the
mp3-now count) has been silently empty behind 2>/dev/null since the cron
migration. PATH now includes the venv; both sections show real numbers.
- strip-watermark-art.py aborted its entire weekly run when metaflac stalled
on ONE file (seen today: a healthy 190KB cover took >15s under disk
contention, TimeoutExpired killed the job). Per-file timeout is now 60s
and a timeout skips that file with a warning instead of failing the run.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
186 lines
6.6 KiB
Python
Executable File
186 lines
6.6 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""
|
|
strip-watermark-art.py — find embedded cover art that appears across multiple
|
|
unrelated albums (common Soulseek-uploader watermarks like djsoundtop.com,
|
|
iptorrents.com, torrentday.com, electronicfresh.com, etc.). Strip those
|
|
images from the offending files; beets' fetchart can re-fetch real cover
|
|
art on the next pass.
|
|
|
|
Heuristic: a real cover.jpg appears in tracks of one (albumartist, album)
|
|
pair. A watermark image appears across many. Anything where the same image
|
|
hash appears in >= --threshold distinct albums is flagged.
|
|
|
|
Handles both FLAC (via metaflac) and MP3 (via mutagen, apt-installed
|
|
python3-mutagen).
|
|
|
|
Usage:
|
|
strip-watermark-art.py # dry run
|
|
strip-watermark-art.py --apply # strip suspicious art
|
|
strip-watermark-art.py --threshold 5 # tune (default 3)
|
|
"""
|
|
import sys, os, hashlib, subprocess, tempfile, argparse
|
|
from pathlib import Path
|
|
from collections import defaultdict
|
|
from mutagen.id3 import ID3, ID3NoHeaderError, APIC
|
|
from mutagen.mp3 import MP3
|
|
|
|
LIBRARY = f"{os.environ.get('MUSIC_DATA_DIR', '/data/music')}/Library"
|
|
|
|
|
|
def build_album_index():
|
|
"""Single beets call → {path: (albumartist, album)} dict. beets runs
|
|
in-process in this same container, so paths need no translation."""
|
|
print("[strip-art] loading beets album index...")
|
|
r = subprocess.run(
|
|
["beet", "ls", "-f", "$path‖$albumartist‖$album"],
|
|
capture_output=True, text=True, timeout=120
|
|
)
|
|
idx = {}
|
|
for line in r.stdout.splitlines():
|
|
parts = line.split("‖", 2)
|
|
if len(parts) != 3:
|
|
continue
|
|
path, aa, al = parts
|
|
idx[path] = (aa, al)
|
|
print(f"[strip-art] indexed {len(idx)} library tracks")
|
|
return idx
|
|
|
|
|
|
def extract_flac_picture(flac_path):
|
|
"""Returns (sha256_hash, size) of the first embedded picture, or None."""
|
|
with tempfile.NamedTemporaryFile(delete=False, suffix=".pic") as tf:
|
|
tmp = tf.name
|
|
try:
|
|
try:
|
|
r = subprocess.run(
|
|
["metaflac", f"--export-picture-to={tmp}", flac_path],
|
|
capture_output=True, timeout=60
|
|
)
|
|
except subprocess.TimeoutExpired:
|
|
# A transient IO stall on one file must not abort the whole weekly
|
|
# run (seen 2026-07-15: a healthy 190KB cover took >15s under disk
|
|
# contention and the raised TimeoutExpired failed the entire job).
|
|
print(f"[strip-art] WARN: metaflac timed out on {flac_path}, skipping file")
|
|
return None
|
|
if r.returncode != 0 or not os.path.exists(tmp):
|
|
return None
|
|
sz = os.path.getsize(tmp)
|
|
if sz == 0:
|
|
return None
|
|
with open(tmp, "rb") as f:
|
|
return hashlib.sha256(f.read()).hexdigest(), sz
|
|
finally:
|
|
try: os.unlink(tmp)
|
|
except FileNotFoundError: pass
|
|
|
|
|
|
def extract_mp3_picture(mp3_path):
|
|
"""Returns (sha256_hash, size) of the first APIC payload, or None."""
|
|
try:
|
|
tags = ID3(mp3_path)
|
|
except (ID3NoHeaderError, Exception):
|
|
return None
|
|
for k in tags.keys():
|
|
if not k.startswith("APIC"):
|
|
continue
|
|
frame = tags[k]
|
|
data = frame.data
|
|
if data and len(data) > 100:
|
|
return hashlib.sha256(data).hexdigest(), len(data)
|
|
return None
|
|
|
|
|
|
def strip_mp3_pictures(mp3_path):
|
|
"""Remove all APIC frames from an MP3. Returns True on success."""
|
|
try:
|
|
tags = ID3(mp3_path)
|
|
keys = [k for k in tags.keys() if k.startswith("APIC")]
|
|
for k in keys:
|
|
del tags[k]
|
|
tags.save(mp3_path)
|
|
return True
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--apply", action="store_true",
|
|
help="Actually strip suspicious art (default: dry run)")
|
|
ap.add_argument("--threshold", type=int, default=3,
|
|
help="Min distinct albums sharing an image to flag it (default 3)")
|
|
args = ap.parse_args()
|
|
|
|
album_idx = build_album_index()
|
|
|
|
# hash -> {albums: set of (aa, al), files: [paths], size: int}
|
|
images = defaultdict(lambda: {"albums": set(), "files": [], "size": 0})
|
|
audio_files = (
|
|
list(Path(LIBRARY).rglob("*.flac")) +
|
|
list(Path(LIBRARY).rglob("*.mp3"))
|
|
)
|
|
print(f"[strip-art] scanning {len(audio_files)} audio files for embedded art...")
|
|
|
|
for i, p in enumerate(audio_files, 1):
|
|
if i % 200 == 0:
|
|
print(f" ...{i}/{len(audio_files)}", flush=True)
|
|
sfx = p.suffix.lower()
|
|
if sfx == ".flac":
|
|
res = extract_flac_picture(str(p))
|
|
elif sfx == ".mp3":
|
|
res = extract_mp3_picture(str(p))
|
|
else:
|
|
continue
|
|
if not res:
|
|
continue
|
|
h, sz = res
|
|
aa, al = album_idx.get(str(p), ("", ""))
|
|
images[h]["albums"].add((aa, al))
|
|
images[h]["files"].append(str(p))
|
|
images[h]["size"] = sz
|
|
|
|
suspicious = {h: info for h, info in images.items()
|
|
if len(info["albums"]) >= args.threshold}
|
|
|
|
if not suspicious:
|
|
print(f"\n[strip-art] no images shared across >= {args.threshold} distinct albums.")
|
|
print("[strip-art] If watermarks remain, lower --threshold or check MP3s manually.")
|
|
return 0
|
|
|
|
print(f"\n[strip-art] {len(suspicious)} suspicious image(s):\n")
|
|
for h, info in sorted(suspicious.items(), key=lambda kv: -len(kv[1]["files"])):
|
|
print(f" {h[:16]}… {info['size']} bytes "
|
|
f"in {len(info['files'])} files across {len(info['albums'])} albums:")
|
|
for aa, al in sorted(info["albums"])[:6]:
|
|
print(f" {aa or '?'} / {al or '?'}")
|
|
if len(info["albums"]) > 6:
|
|
print(f" ...and {len(info['albums']) - 6} more")
|
|
print()
|
|
|
|
if not args.apply:
|
|
n = sum(len(s["files"]) for s in suspicious.values())
|
|
print(f"[strip-art] DRY RUN — re-run with --apply to strip from {n} FLAC files")
|
|
return 0
|
|
|
|
stripped = 0
|
|
for info in suspicious.values():
|
|
for f in info["files"]:
|
|
if f.lower().endswith(".flac"):
|
|
r = subprocess.run(
|
|
["metaflac", "--remove", "--block-type=PICTURE", f],
|
|
capture_output=True
|
|
)
|
|
if r.returncode == 0:
|
|
stripped += 1
|
|
elif f.lower().endswith(".mp3"):
|
|
if strip_mp3_pictures(f):
|
|
stripped += 1
|
|
|
|
print(f"\n[strip-art] stripped picture blocks from {stripped} files")
|
|
print("[strip-art] next: re-fetch real cover art with:")
|
|
print(" beet fetchart")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|