Compare commits
4 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| f05f076464 | |||
| 62251d764e | |||
| d756aab7fc | |||
| 24738d815f |
@@ -73,10 +73,10 @@ Optional, add these later if you want them:
|
|||||||
Pull the prebuilt image onto your Docker host:
|
Pull the prebuilt image onto your Docker host:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
docker pull git.kretzer.club/andrew/alembic:0.6.6
|
docker pull git.kretzer.club/andrew/alembic:0.6.10
|
||||||
```
|
```
|
||||||
|
|
||||||
That is the whole install. You do not need to download the source or build anything. The `0.6.6` is the version; you can pin to it so nothing changes under you, or use `latest` to always get the newest.
|
That is the whole install. You do not need to download the source or build anything. The `0.6.10` is the version; you can pin to it so nothing changes under you, or use `latest` to always get the newest.
|
||||||
|
|
||||||
(If you would rather build it yourself from source, you can, but you do not need to.)
|
(If you would rather build it yourself from source, you can, but you do not need to.)
|
||||||
|
|
||||||
@@ -136,7 +136,7 @@ Create a file called `docker-compose.yml` on your server (put it wherever you ke
|
|||||||
```yaml
|
```yaml
|
||||||
services:
|
services:
|
||||||
alembic:
|
alembic:
|
||||||
image: git.kretzer.club/andrew/alembic:0.6.6
|
image: git.kretzer.club/andrew/alembic:0.6.10
|
||||||
container_name: alembic
|
container_name: alembic
|
||||||
ports:
|
ports:
|
||||||
- "8420:8420"
|
- "8420:8420"
|
||||||
|
|||||||
@@ -58,7 +58,9 @@ def _is_pair_already_pending(db, keep_path: str, delete_path: str) -> bool:
|
|||||||
return db.execute(query).first() is not None
|
return db.execute(query).first() is not None
|
||||||
|
|
||||||
|
|
||||||
async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupRun | None:
|
async def _run_scan(
|
||||||
|
job_key: str, script_name: str, triggered_by: str, extra_args: tuple[str, ...] = ()
|
||||||
|
) -> DedupRun | None:
|
||||||
"""Returns None (persisting nothing) if the job never actually ran --
|
"""Returns None (persisting nothing) if the job never actually ran --
|
||||||
e.g. skipped_lock because something else was using the pipeline lock
|
e.g. skipped_lock because something else was using the pipeline lock
|
||||||
at that moment. Same lesson as genre_review_service.run(): recording a
|
at that moment. Same lesson as genre_review_service.run(): recording a
|
||||||
@@ -67,7 +69,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
pending-candidates list isn't scoped to "latest run only"."""
|
pending-candidates list isn't scoped to "latest run only"."""
|
||||||
script = str(settings.pipeline_dir / "lib" / script_name)
|
script = str(settings.pipeline_dir / "lib" / script_name)
|
||||||
job_run, output = await pipeline_runner.run_job_capture(
|
job_run, output = await pipeline_runner.run_job_capture(
|
||||||
job_key, [script, "--json"], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
|
job_key, [script, "--json", *extra_args], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
|
||||||
)
|
)
|
||||||
if job_run.status != "success":
|
if job_run.status != "success":
|
||||||
return None
|
return None
|
||||||
@@ -76,6 +78,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
|
|
||||||
db = SessionLocal()
|
db = SessionLocal()
|
||||||
try:
|
try:
|
||||||
|
now = time.time()
|
||||||
dedup_run = DedupRun(
|
dedup_run = DedupRun(
|
||||||
started_at=job_run.started_at,
|
started_at=job_run.started_at,
|
||||||
finished_at=job_run.finished_at,
|
finished_at=job_run.finished_at,
|
||||||
@@ -91,7 +94,30 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
|
|
||||||
skipped_ignored = 0
|
skipped_ignored = 0
|
||||||
skipped_duplicate = 0
|
skipped_duplicate = 0
|
||||||
|
auto_applied = 0
|
||||||
for c in candidates:
|
for c in candidates:
|
||||||
|
if c.get("auto_applied"):
|
||||||
|
# dedup-library.sh already deleted this pair unconditionally
|
||||||
|
# (identical numbered-sibling twin: same dir/name/ext, no
|
||||||
|
# tag/fuzzy ambiguity involved) -- record it pre-applied for
|
||||||
|
# audit history; it never shows up as a pending review item.
|
||||||
|
db.add(
|
||||||
|
DedupCandidate(
|
||||||
|
dedup_run_id=dedup_run.id,
|
||||||
|
pass_name=c.get("pass", "unknown"),
|
||||||
|
keep_path=c["keep_path"],
|
||||||
|
delete_path=c["delete_path"],
|
||||||
|
delete_id=c.get("delete_id"),
|
||||||
|
delete_size_bytes=c.get("delete_size_bytes"),
|
||||||
|
confirmed=True,
|
||||||
|
confirmed_by="auto:numbered_twin",
|
||||||
|
confirmed_at=now,
|
||||||
|
applied=True,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
auto_applied += 1
|
||||||
|
db.flush()
|
||||||
|
continue
|
||||||
if _is_pair_ignored(db, c["keep_path"], c["delete_path"]):
|
if _is_pair_ignored(db, c["keep_path"], c["delete_path"]):
|
||||||
skipped_ignored += 1
|
skipped_ignored += 1
|
||||||
continue
|
continue
|
||||||
@@ -116,8 +142,11 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
db.commit()
|
db.commit()
|
||||||
db.refresh(dedup_run)
|
db.refresh(dedup_run)
|
||||||
skipped_total = skipped_ignored + skipped_duplicate
|
skipped_total = skipped_ignored + skipped_duplicate
|
||||||
if skipped_total:
|
if skipped_total or auto_applied:
|
||||||
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
|
if skipped_total:
|
||||||
|
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
|
||||||
|
if auto_applied:
|
||||||
|
dedup_run.deleted = auto_applied
|
||||||
db.commit()
|
db.commit()
|
||||||
db.refresh(dedup_run)
|
db.refresh(dedup_run)
|
||||||
return dedup_run
|
return dedup_run
|
||||||
@@ -126,12 +155,19 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
|
|
||||||
|
|
||||||
async def scan(triggered_by: str = "manual") -> DedupRun | None:
|
async def scan(triggered_by: str = "manual") -> DedupRun | None:
|
||||||
"""Dry-run dedup-library.sh --json, persist every candidate deletion
|
"""Dry-run dedup-library.sh --json (plus --auto-apply-numbered), persist
|
||||||
into a fresh dedup_runs/dedup_candidates pair. Never deletes anything
|
every candidate deletion into a fresh dedup_runs/dedup_candidates pair.
|
||||||
-- the scheduled maintenance:dedup job also only ever calls this (no
|
|
||||||
--apply), matching the false-negative-biased dedup preference; actual
|
Everything except identical numbered-sibling twins (same dir, same name,
|
||||||
deletion only ever happens through confirm_and_apply() below."""
|
same extension, differing only by the ".N" collision suffix) stays a
|
||||||
return await _run_scan("dedup:scan", _SCRIPT, triggered_by)
|
dry-run candidate awaiting manual confirm_and_apply() below -- that's the
|
||||||
|
false-negative-biased preference for tag-based/fuzzy matches, where a
|
||||||
|
wrong auto-delete could take out a genuinely different version (a
|
||||||
|
different album pressing, a DJ-mix edit, etc.). Numbered twins have none
|
||||||
|
of that ambiguity -- it's the same download landing twice -- so
|
||||||
|
dedup-library.sh deletes those unconditionally itself and reports them
|
||||||
|
here already applied, purely for audit visibility."""
|
||||||
|
return await _run_scan("dedup:scan", _SCRIPT, triggered_by, extra_args=("--auto-apply-numbered",))
|
||||||
|
|
||||||
|
|
||||||
async def scan_fuzzy(triggered_by: str = "manual") -> DedupRun | None:
|
async def scan_fuzzy(triggered_by: str = "manual") -> DedupRun | None:
|
||||||
|
|||||||
@@ -96,6 +96,21 @@ else
|
|||||||
log "Fetched $TRACK_COUNT track(s) from Spotify"
|
log "Fetched $TRACK_COUNT track(s) from Spotify"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# ==== Seed the pinned sldl skip index if it's missing or stranded ====
|
||||||
|
# The conf pins index-path to $DROPBOX/_index.csv (see _template.conf: sldl's
|
||||||
|
# default index location is derived from the input, so an input change strands
|
||||||
|
# the old index and the whole playlist re-downloads -- that's what produced
|
||||||
|
# the 2026-07-16 duplicate flood). If the pinned file is missing, or an index
|
||||||
|
# in one of sldl's old input-named subfolders is newer (i.e. sldl last wrote
|
||||||
|
# somewhere else), fold them all into the pinned location first.
|
||||||
|
INDEX_FILE="$DROPBOX/_index.csv"
|
||||||
|
if [[ ! -s "$INDEX_FILE" ]] || \
|
||||||
|
[[ -n "$(find "$DROPBOX" -mindepth 2 -maxdepth 2 -name _index.csv -newer "$INDEX_FILE" -print -quit 2>/dev/null)" ]]; then
|
||||||
|
log "Seeding pinned sldl index at $INDEX_FILE from prior indexes"
|
||||||
|
python3 "${PIPELINE_DIR:-/app/pipeline}/lib/merge-sldl-indexes.py" "$DROPBOX" "$INDEX_FILE" >> "$LOG" 2>&1 \
|
||||||
|
|| log "WARNING: index merge failed -- sldl may re-download tracks the library already has"
|
||||||
|
fi
|
||||||
|
|
||||||
# ==== Run sldl (vendored binary, subprocess of this container) ====
|
# ==== Run sldl (vendored binary, subprocess of this container) ====
|
||||||
log "Running sldl for: $PLAYLIST_NAME"
|
log "Running sldl for: $PLAYLIST_NAME"
|
||||||
|
|
||||||
|
|||||||
@@ -31,6 +31,18 @@ path = SLDL_DROPBOX_ROOT/PLAYLIST_NAME
|
|||||||
playlist-path = SLDL_DROPBOX_ROOT/PLAYLIST_NAME/_sldl.m3u8
|
playlist-path = SLDL_DROPBOX_ROOT/PLAYLIST_NAME/_sldl.m3u8
|
||||||
write-playlist = true
|
write-playlist = true
|
||||||
|
|
||||||
|
# ==== Skip index (pinned!) ====
|
||||||
|
# sldl's already-downloaded index defaults to {path}/{playlist-name}/_index.csv
|
||||||
|
# where {playlist-name} is derived from the INPUT: the Spotify playlist's
|
||||||
|
# display name for spotify input, the CSV filename for csv input. beets moves
|
||||||
|
# every download out of the dropbox, so this index is the ONLY thing standing
|
||||||
|
# between a nightly run and re-downloading the whole playlist. When 0.6.2
|
||||||
|
# switched input from Spotify URL to CSV, the derived name changed, sldl
|
||||||
|
# started a fresh empty index in a new subfolder, and every playlist
|
||||||
|
# re-downloaded in full on 2026-07-16. Pinning the path here decouples the
|
||||||
|
# index from input naming so that can never happen again.
|
||||||
|
index-path = SLDL_DROPBOX_ROOT/PLAYLIST_NAME/_index.csv
|
||||||
|
|
||||||
# ==== Naming ====
|
# ==== Naming ====
|
||||||
name-format = {artist} - {title}
|
name-format = {artist} - {title}
|
||||||
|
|
||||||
|
|||||||
@@ -34,11 +34,19 @@ set -euo pipefail
|
|||||||
APPLY=0
|
APPLY=0
|
||||||
JSON_MODE=0
|
JSON_MODE=0
|
||||||
ONLY_PATHS_FILE=""
|
ONLY_PATHS_FILE=""
|
||||||
|
AUTO_APPLY_NUMBERED=0
|
||||||
while [[ $# -gt 0 ]]; do
|
while [[ $# -gt 0 ]]; do
|
||||||
case "$1" in
|
case "$1" in
|
||||||
--apply) APPLY=1; shift ;;
|
--apply) APPLY=1; shift ;;
|
||||||
--json) JSON_MODE=1; shift ;;
|
--json) JSON_MODE=1; shift ;;
|
||||||
--only-paths) ONLY_PATHS_FILE="$2"; shift 2 ;;
|
--only-paths) ONLY_PATHS_FILE="$2"; shift 2 ;;
|
||||||
|
# Delete Pass 1 (numbered-sibling) matches unconditionally, independent of
|
||||||
|
# --apply/--only-paths, while Passes 2-4 stay dry-run/log-only. Safe to
|
||||||
|
# run unattended: a numbered-sibling pair is the SAME title, in the SAME
|
||||||
|
# directory, with the SAME extension -- unlike the tag-based/fuzzy passes,
|
||||||
|
# there's no fuzzy matching or cross-album ambiguity to get wrong, so it
|
||||||
|
# doesn't need a human in the loop. See dedup_review_service.scan().
|
||||||
|
--auto-apply-numbered) AUTO_APPLY_NUMBERED=1; shift ;;
|
||||||
*) shift ;;
|
*) shift ;;
|
||||||
esac
|
esac
|
||||||
done
|
done
|
||||||
@@ -78,11 +86,44 @@ fi
|
|||||||
# Counters: tracked via log grep at the end (avoids bash subshell pitfalls).
|
# Counters: tracked via log grep at the end (avoids bash subshell pitfalls).
|
||||||
# Functions write KEEP/DELETE lines with consistent prefixes; summary greps them.
|
# Functions write KEEP/DELETE lines with consistent prefixes; summary greps them.
|
||||||
|
|
||||||
# Convert container path (/music/...) to host path (${MUSIC_DATA_DIR:-/data/music}/Library/...)
|
# Resolve a beets-displayed path to a real path in this container. Since the
|
||||||
host_path() { echo "${MUSIC_DATA_DIR:-/data/music}/Library${1#/music}"; }
|
# 2026-07-08 path rewrite, beets displays ${MUSIC_DATA_DIR:-/data/music}/Library/...
|
||||||
|
# paths that are directly usable — pass those through UNCHANGED. Only legacy
|
||||||
|
# /music/... display paths (the pre-rewrite transitional mount) still need the
|
||||||
|
# prefix swap. Blindly prepending the library dir to an already-correct path
|
||||||
|
# (the old behavior) produced /data/music/Library/data/music/Library/... —
|
||||||
|
# nonexistent, so every candidate failed the -f check and every pass reported
|
||||||
|
# 0 groups from 2026-07-08 until this fix.
|
||||||
|
host_path() {
|
||||||
|
local p="$1"
|
||||||
|
if [[ "$p" == /music/* ]]; then
|
||||||
|
echo "${MUSIC_DATA_DIR:-/data/music}/Library${p#/music}"
|
||||||
|
else
|
||||||
|
echo "$p"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
SEP=$'\x1c' # ASCII file separator — safe with any music metadata
|
SEP=$'\x1c' # ASCII file separator — safe with any music metadata
|
||||||
|
|
||||||
|
# True if two paths are the SAME directory, SAME extension, and identical
|
||||||
|
# basenames except for a numeric ".N" infix sldl/beets inserts on a filename
|
||||||
|
# collision (e.g. "Title.flac" vs "Title.1.flac"). This is deliberately
|
||||||
|
# narrower than any tag-based pass: no fuzzy matching, no cross-album
|
||||||
|
# ambiguity, just the literal same download landing twice. Used to
|
||||||
|
# auto-delete regardless of which pass found the pair -- Pass 1 only fires
|
||||||
|
# when the canonical name is a beets ghost; the very common case where BOTH
|
||||||
|
# copies are already tracked in beets (so Pass 1 defers) surfaces instead via
|
||||||
|
# Pass 2/3's tag-based grouping, and still deserves the same free pass.
|
||||||
|
is_numbered_twin() {
|
||||||
|
local a="$1" b="$2"
|
||||||
|
[[ "${a%/*}" == "${b%/*}" ]] || return 1
|
||||||
|
local ba="${a##*/}" bb="${b##*/}"
|
||||||
|
local ext="${ba##*.}"
|
||||||
|
[[ "${ext,,}" == "${bb##*.}" ]] || return 1
|
||||||
|
[[ "$(echo "$ba" | sed -E "s/\.[0-9]+\.${ext}\$/.${ext}/")" \
|
||||||
|
== "$(echo "$bb" | sed -E "s/\.[0-9]+\.${ext}\$/.${ext}/")" ]]
|
||||||
|
}
|
||||||
|
|
||||||
# Rank a file: lower = better. FLAC > MP3 > other (WAV/etc.); clean filename
|
# Rank a file: lower = better. FLAC > MP3 > other (WAV/etc.); clean filename
|
||||||
# > sync-conflict artifact (-DESKTOP-, (1), .1.); extended/club mix > radio
|
# > sync-conflict artifact (-DESKTOP-, (1), .1.); extended/club mix > radio
|
||||||
# edit at same format; then larger wins.
|
# edit at same format; then larger wins.
|
||||||
@@ -163,7 +204,7 @@ process_group() {
|
|||||||
esac
|
esac
|
||||||
|
|
||||||
local first=1
|
local first=1
|
||||||
local keep_path=""
|
local keep_path="" keep_id="" any_auto=0
|
||||||
while IFS="$SEP" read -r _sk id hp; do
|
while IFS="$SEP" read -r _sk id hp; do
|
||||||
[[ -z "$hp" ]] && continue
|
[[ -z "$hp" ]] && continue
|
||||||
local size_bytes; size_bytes=$(stat -c '%s' "$hp" 2>/dev/null || echo 0)
|
local size_bytes; size_bytes=$(stat -c '%s' "$hp" 2>/dev/null || echo 0)
|
||||||
@@ -171,14 +212,30 @@ process_group() {
|
|||||||
if [[ $first -eq 1 ]]; then
|
if [[ $first -eq 1 ]]; then
|
||||||
log " KEEP ($size_h) $hp"
|
log " KEEP ($size_h) $hp"
|
||||||
keep_path="$hp"
|
keep_path="$hp"
|
||||||
|
keep_id="$id"
|
||||||
first=0
|
first=0
|
||||||
else
|
else
|
||||||
log " DELETE ($size_h) $hp"
|
log " DELETE ($size_h) $hp"
|
||||||
json_emit "{\"pass\":\"${pass_name}\",\"keep_path\":\"$(json_escape "$keep_path")\",\"delete_path\":\"$(json_escape "$hp")\",\"delete_size_bytes\":${size_bytes}}"
|
|
||||||
|
# Auto-apply-numbered deletes numbered_sibling matches unconditionally,
|
||||||
|
# bypassing --only-paths (that gate exists only for the manual confirm
|
||||||
|
# flow, which never runs alongside this flag). All other passes stay
|
||||||
|
# dry-run/log-only here regardless.
|
||||||
|
local do_delete=0 auto_applied=0
|
||||||
if [[ $APPLY -eq 1 ]]; then
|
if [[ $APPLY -eq 1 ]]; then
|
||||||
|
do_delete=1
|
||||||
|
elif [[ $AUTO_APPLY_NUMBERED -eq 1 ]] && is_numbered_twin "$keep_path" "$hp"; then
|
||||||
|
do_delete=1
|
||||||
|
auto_applied=1
|
||||||
|
any_auto=1
|
||||||
|
fi
|
||||||
|
local auto_json="false"; [[ $auto_applied -eq 1 ]] && auto_json="true"
|
||||||
|
json_emit "{\"pass\":\"${pass_name}\",\"keep_path\":\"$(json_escape "$keep_path")\",\"delete_path\":\"$(json_escape "$hp")\",\"delete_size_bytes\":${size_bytes},\"auto_applied\":${auto_json}}"
|
||||||
|
|
||||||
|
if [[ $do_delete -eq 1 ]]; then
|
||||||
# With --only-paths given, only delete entries the caller explicitly
|
# With --only-paths given, only delete entries the caller explicitly
|
||||||
# confirmed; without it, delete everything (original behavior).
|
# confirmed; without it, delete everything (original behavior).
|
||||||
if [[ -n "$ONLY_PATHS_FILE" && -z "${ONLY_PATHS[$hp]:-}" ]]; then
|
if [[ $auto_applied -eq 0 && -n "$ONLY_PATHS_FILE" && -z "${ONLY_PATHS[$hp]:-}" ]]; then
|
||||||
log " (skipped — not in --only-paths confirm list)"
|
log " (skipped — not in --only-paths confirm list)"
|
||||||
elif [[ -n "$id" ]]; then
|
elif [[ -n "$id" ]]; then
|
||||||
# In beets — beet remove -d removes from DB + disk. Best-effort: a
|
# In beets — beet remove -d removes from DB + disk. Best-effort: a
|
||||||
@@ -192,6 +249,18 @@ process_group() {
|
|||||||
fi
|
fi
|
||||||
fi
|
fi
|
||||||
done < <(echo "$ranked" | grep -v '^$' | sort -n)
|
done < <(echo "$ranked" | grep -v '^$' | sort -n)
|
||||||
|
|
||||||
|
# An auto-applied numbered-twin deletion can leave the numbered-named file
|
||||||
|
# as the sole survivor (it won on size despite the dirty-filename penalty).
|
||||||
|
# Rename it back to the canonical name so the library doesn't accumulate
|
||||||
|
# ".1."/".2." filenames for tracks that no longer have a duplicate. Only
|
||||||
|
# auto_applied triggers this (never a manual --only-paths confirm, which
|
||||||
|
# may leave the sibling undeleted if the user didn't confirm it -- renaming
|
||||||
|
# then would collide); only when it's actually in beets (a ghost survivor
|
||||||
|
# has nothing to move); only when the name still has the dirty infix.
|
||||||
|
if [[ $any_auto -eq 1 && -n "$keep_id" && "$keep_path" =~ \.[0-9]+\.[A-Za-z0-9]+$ ]]; then
|
||||||
|
beet move "id:${keep_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${keep_id} failed (rename survivor)"
|
||||||
|
fi
|
||||||
return 0 # explicit: the loop's last command status must not leak out under set -e
|
return 0 # explicit: the loop's last command status must not leak out under set -e
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -202,9 +271,10 @@ process_group() {
|
|||||||
log ""
|
log ""
|
||||||
log "[$(date -Iseconds)] === Pass 1: numbered siblings ==="
|
log "[$(date -Iseconds)] === Pass 1: numbered siblings ==="
|
||||||
|
|
||||||
# Dump id + container path from beets once. beets normalizes $path for
|
# Dump id + display path from beets once. beets normalizes $path for display
|
||||||
# display (always /music/-prefixed) even though the DB stores a mix of
|
# (real /data/music/Library/... paths since the 2026-07-08 rewrite; /music/...
|
||||||
# absolute and relative — so the display paths are safe to compare/convert,
|
# on any legacy entry) even though the DB stores a mix of absolute and
|
||||||
|
# relative — so the display paths are safe to compare/convert via host_path(),
|
||||||
# and the ids are what we hand to beet remove/move.
|
# and the ids are what we hand to beet remove/move.
|
||||||
beet ls -f "\$id${SEP}\$path" 2>/dev/null > /tmp/beets-id-paths.txt
|
beet ls -f "\$id${SEP}\$path" 2>/dev/null > /tmp/beets-id-paths.txt
|
||||||
cut -d"$SEP" -f2- /tmp/beets-id-paths.txt > /tmp/beets-all-paths.txt
|
cut -d"$SEP" -f2- /tmp/beets-id-paths.txt > /tmp/beets-all-paths.txt
|
||||||
@@ -231,18 +301,22 @@ while IFS="$SEP" read -r numbered_id numbered_cp; do
|
|||||||
|
|
||||||
if [[ $score_canonical -le $score_numbered ]]; then
|
if [[ $score_canonical -le $score_numbered ]]; then
|
||||||
# Canonical wins: delete numbered (in beets), keep canonical (ghost)
|
# Canonical wins: delete numbered (in beets), keep canonical (ghost)
|
||||||
process_group "P1 numbered-sibling: ${numbered_cp##/music/}" \
|
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
|
||||||
"$host_canonical" "${numbered_id}${SEP}${numbered_cp}"
|
"$host_canonical" "${numbered_id}${SEP}${numbered_cp}"
|
||||||
if [[ $APPLY -eq 1 ]]; then
|
if [[ $APPLY -eq 1 || $AUTO_APPLY_NUMBERED -eq 1 ]]; then
|
||||||
# canonical is now a ghost file; import it into beets
|
# canonical is now a ghost file; import it into beets
|
||||||
beet import -q -s "${canonical_cp}" >> "$LOG" 2>&1 || log " WARN: beet import failed for ${canonical_cp}"
|
beet import -q -s "${canonical_cp}" >> "$LOG" 2>&1 || log " WARN: beet import failed for ${canonical_cp}"
|
||||||
fi
|
fi
|
||||||
else
|
else
|
||||||
# Numbered wins (canonical is ghost): rm the ghost, then beet move to rename
|
# Numbered wins (canonical is ghost): rm the ghost, then beet move to rename
|
||||||
# the numbered file to the canonical path (id query — see process_group).
|
# the numbered file to the canonical path (id query — see process_group).
|
||||||
process_group "P1 numbered-sibling: ${numbered_cp##/music/}" \
|
# Safe to reach with just AUTO_APPLY_NUMBERED: this loop only ever builds
|
||||||
|
# genuine numbered-sibling pairs (see is_numbered_twin), and process_group
|
||||||
|
# above already deleted the ghost canonical before this runs, so the move
|
||||||
|
# target is guaranteed clear -- no rename collision.
|
||||||
|
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
|
||||||
"${numbered_id}${SEP}${numbered_cp}" "$host_canonical"
|
"${numbered_id}${SEP}${numbered_cp}" "$host_canonical"
|
||||||
if [[ $APPLY -eq 1 ]]; then
|
if [[ $APPLY -eq 1 || $AUTO_APPLY_NUMBERED -eq 1 ]]; then
|
||||||
beet move "id:${numbered_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${numbered_id} failed"
|
beet move "id:${numbered_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${numbered_id} failed"
|
||||||
fi
|
fi
|
||||||
fi
|
fi
|
||||||
|
|||||||
@@ -57,11 +57,12 @@ OUT_DIR = f"{_MUSIC_DATA_DIR}/playlists-laptop"
|
|||||||
# Absolute Windows path to the laptop's library root.
|
# Absolute Windows path to the laptop's library root.
|
||||||
LIBRARY_LAPTOP_ROOT = r"C:\Music\Library"
|
LIBRARY_LAPTOP_ROOT = r"C:\Music\Library"
|
||||||
|
|
||||||
# Beets stores paths with this prefix (the container-side mount of the
|
# Prefixes beets may display library paths under, tried in order: the real
|
||||||
# library); strip it to get the path relative to the library root. This is
|
# library dir (everything since the 2026-07-08 path rewrite) and the legacy
|
||||||
# still "/music/" through the migration's Stage 0-3 transitional mount;
|
# transitional mount (any stale pre-rewrite entry). Strip whichever matches
|
||||||
# revisit at Stage 4 once beets' directory: becomes MUSIC_DATA_DIR/Library.
|
# to get the path relative to the library root; when neither matches the
|
||||||
BEETS_LIBRARY_PREFIX = "/music/"
|
# track is outside the library and is skipped.
|
||||||
|
BEETS_LIBRARY_PREFIXES = (f"{_MUSIC_DATA_DIR}/Library/", "/music/")
|
||||||
|
|
||||||
M3U_EXT = ".m3u"
|
M3U_EXT = ".m3u"
|
||||||
# Older formats from earlier iterations of this script — swept by reconcile.
|
# Older formats from earlier iterations of this script — swept by reconcile.
|
||||||
@@ -157,9 +158,12 @@ def build_beets_index() -> dict[tuple[str, str], list[tuple[str, str, str, str]]
|
|||||||
if len(parts) != 4:
|
if len(parts) != 4:
|
||||||
continue
|
continue
|
||||||
aa, alb, ti, path = parts
|
aa, alb, ti, path = parts
|
||||||
if not path.startswith(BEETS_LIBRARY_PREFIX):
|
rel = next(
|
||||||
|
(path[len(p):] for p in BEETS_LIBRARY_PREFIXES if path.startswith(p)),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
if rel is None:
|
||||||
continue
|
continue
|
||||||
rel = path[len(BEETS_LIBRARY_PREFIX):]
|
|
||||||
idx.setdefault((_normkey(alb), _normkey(ti)), []).append((aa, alb, ti, rel))
|
idx.setdefault((_normkey(alb), _normkey(ti)), []).append((aa, alb, ti, rel))
|
||||||
return idx
|
return idx
|
||||||
|
|
||||||
|
|||||||
@@ -49,8 +49,10 @@ _ALEMBIC_CONFIG_DIR = os.environ.get("ALEMBIC_CONFIG_DIR", "/config")
|
|||||||
_MUSIC_DATA_DIR = os.environ.get("MUSIC_DATA_DIR", "/data/music")
|
_MUSIC_DATA_DIR = os.environ.get("MUSIC_DATA_DIR", "/data/music")
|
||||||
|
|
||||||
LIBRARY = f"{_MUSIC_DATA_DIR}/Library"
|
LIBRARY = f"{_MUSIC_DATA_DIR}/Library"
|
||||||
# Still "/music" through the migration's Stage 0-3 transitional beets mount;
|
# Legacy prefix from the migration's transitional beets mount. Since the
|
||||||
# revisit at Stage 4 once beets' directory: becomes MUSIC_DATA_DIR/Library.
|
# 2026-07-08 path rewrite beets displays real LIBRARY paths, so this only
|
||||||
|
# matters for accepting pasted /music/... paths as input and as a fallback
|
||||||
|
# beets query form for any stale pre-rewrite DB entry.
|
||||||
CONTAINER_PREFIX = "/music"
|
CONTAINER_PREFIX = "/music"
|
||||||
SPOTIFY_ENV = f"{_ALEMBIC_CONFIG_DIR}/pipeline/_spotify.env"
|
SPOTIFY_ENV = f"{_ALEMBIC_CONFIG_DIR}/pipeline/_spotify.env"
|
||||||
SPOTIFY_GENRE = f"{os.environ.get('PIPELINE_DIR', '/app/pipeline')}/lib/spotify-genre.py"
|
SPOTIFY_GENRE = f"{os.environ.get('PIPELINE_DIR', '/app/pipeline')}/lib/spotify-genre.py"
|
||||||
@@ -316,15 +318,19 @@ def resolve_input_path(arg):
|
|||||||
|
|
||||||
|
|
||||||
def beets_id_for(host_path):
|
def beets_id_for(host_path):
|
||||||
cp = host_to_container(host_path)
|
# Real path first (how beets displays everything since the 2026-07-08
|
||||||
r = subprocess.run(
|
# path rewrite), legacy /music/... form second for any stale entry.
|
||||||
["beet", "ls", "-f", "$id|||$path", f"path:{cp}"],
|
# Querying only the legacy form (the old behavior) matched nothing after
|
||||||
capture_output=True, text=True, check=True)
|
# the rewrite, so retag-from-url silently skipped beet update/move.
|
||||||
for line in r.stdout.splitlines():
|
for cp in (host_path, host_to_container(host_path)):
|
||||||
if "|||" in line:
|
r = subprocess.run(
|
||||||
id_str, p = line.split("|||", 1)
|
["beet", "ls", "-f", "$id|||$path", f"path:{cp}"],
|
||||||
if p == cp:
|
capture_output=True, text=True, check=True)
|
||||||
return int(id_str)
|
for line in r.stdout.splitlines():
|
||||||
|
if "|||" in line:
|
||||||
|
id_str, p = line.split("|||", 1)
|
||||||
|
if p == cp:
|
||||||
|
return int(id_str)
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,121 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Fold every sldl skip index under a playlist's dropbox into one pinned file.
|
||||||
|
|
||||||
|
sldl names its per-playlist index folder after the *input*: the Spotify
|
||||||
|
playlist's display name for spotify input, the CSV filename stem for csv
|
||||||
|
input. So when 0.6.2 switched input from Spotify URL to CSV, sldl started a
|
||||||
|
fresh empty index in a new subfolder, saw no download history, and
|
||||||
|
re-downloaded every playlist in full (2026-07-16). The rendered confs now pin
|
||||||
|
index-path to <dropbox>/<playlist>/_index.csv; this script seeds that pinned
|
||||||
|
file from all the indexes sldl left behind (run-playlist.sh calls it whenever
|
||||||
|
the pinned file is missing or older than a stranded one).
|
||||||
|
|
||||||
|
Usage: merge-sldl-indexes.py <playlist_dropbox_dir> <output_index_csv>
|
||||||
|
|
||||||
|
Merge rules, per (artist, title) lowercased:
|
||||||
|
- the row from the newest index file wins;
|
||||||
|
- EXCEPT when that row is a failure (state 2) and any older index has the
|
||||||
|
track as downloaded (state 1 or 3): then the newest row's identity
|
||||||
|
(artist/album/title/length -- matching what the current input will
|
||||||
|
present; Spotify length rounding drifted between the old extractor and
|
||||||
|
our CSV) is kept but marked state 3 (already downloaded), so sldl does
|
||||||
|
not re-fetch a track the library already holds;
|
||||||
|
- rows only present in older indexes are kept as-is (tracks since removed
|
||||||
|
from the playlist; harmless, and they keep their history if re-added).
|
||||||
|
"""
|
||||||
|
|
||||||
|
import csv
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
HEADER = ["filepath", "artist", "album", "title", "length", "tracktype", "state", "failurereason"]
|
||||||
|
DOWNLOADED_STATES = {"1", "3"} # 1 = downloaded this run, 3 = found in index previously
|
||||||
|
FAILED_STATE = "2"
|
||||||
|
|
||||||
|
|
||||||
|
def read_rows(path: Path) -> list[dict]:
|
||||||
|
rows = []
|
||||||
|
with open(path, newline="", encoding="utf-8") as fh:
|
||||||
|
reader = csv.DictReader(fh)
|
||||||
|
for row in reader:
|
||||||
|
if row.get("artist") is None or row.get("title") is None:
|
||||||
|
continue
|
||||||
|
rows.append({k: (row.get(k) or "") for k in HEADER})
|
||||||
|
return rows
|
||||||
|
|
||||||
|
|
||||||
|
def norm_key(row: dict) -> tuple:
|
||||||
|
# Primary artist only: sldl's old Spotify extractor recorded just the
|
||||||
|
# first artist, while spotify-playlist-csv.py joins all of them with
|
||||||
|
# ", " -- keying on the full string would miss every multi-artist track
|
||||||
|
# when recovering history across the two index generations. First
|
||||||
|
# comma-segment matches both forms (and both sides of a comma-in-name
|
||||||
|
# artist like "Tyler, The Creator" truncate identically).
|
||||||
|
return (row["artist"].split(",")[0].strip().lower(), row["title"].strip().lower())
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
if len(sys.argv) != 3:
|
||||||
|
print(__doc__, file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
dropbox = Path(sys.argv[1])
|
||||||
|
out_path = Path(sys.argv[2])
|
||||||
|
if not dropbox.is_dir():
|
||||||
|
print(f"[merge-sldl-indexes] not a directory: {dropbox}", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
|
# Every index at the dropbox root or one level down (sldl's input-named
|
||||||
|
# subfolders), including the pinned output itself if it already exists --
|
||||||
|
# newest first, so the most recent record of each track wins.
|
||||||
|
candidates = sorted(
|
||||||
|
set(dropbox.glob("_index.csv")) | set(dropbox.glob("*/_index.csv")),
|
||||||
|
key=lambda p: p.stat().st_mtime,
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
if not candidates:
|
||||||
|
print(f"[merge-sldl-indexes] no _index.csv found under {dropbox}; nothing to seed")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
merged: dict[tuple, dict] = {}
|
||||||
|
recovered = 0
|
||||||
|
for path in candidates:
|
||||||
|
try:
|
||||||
|
rows = read_rows(path)
|
||||||
|
except (OSError, csv.Error) as exc:
|
||||||
|
print(f"[merge-sldl-indexes] skipping unreadable {path}: {exc}", file=sys.stderr)
|
||||||
|
continue
|
||||||
|
print(f"[merge-sldl-indexes] {path}: {len(rows)} rows")
|
||||||
|
for row in rows:
|
||||||
|
key = norm_key(row)
|
||||||
|
kept = merged.get(key)
|
||||||
|
if kept is None:
|
||||||
|
merged[key] = row
|
||||||
|
elif kept["state"] == FAILED_STATE and row["state"] in DOWNLOADED_STATES:
|
||||||
|
# Newest attempt failed but an older index proves we already
|
||||||
|
# have this track: keep the newest identity fields, take the
|
||||||
|
# old filepath (informational only; skip-mode index never
|
||||||
|
# checks the file on disk), and mark it downloaded.
|
||||||
|
kept["filepath"] = row["filepath"]
|
||||||
|
kept["state"] = "3"
|
||||||
|
kept["failurereason"] = "0"
|
||||||
|
recovered += 1
|
||||||
|
|
||||||
|
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
tmp = out_path.with_suffix(".csv.tmp")
|
||||||
|
with open(tmp, "w", newline="", encoding="utf-8") as fh:
|
||||||
|
writer = csv.DictWriter(fh, fieldnames=HEADER)
|
||||||
|
writer.writeheader()
|
||||||
|
writer.writerows(merged.values())
|
||||||
|
tmp.replace(out_path)
|
||||||
|
|
||||||
|
downloaded = sum(1 for r in merged.values() if r["state"] in DOWNLOADED_STATES)
|
||||||
|
print(
|
||||||
|
f"[merge-sldl-indexes] wrote {out_path}: {len(merged)} tracks "
|
||||||
|
f"({downloaded} downloaded, {recovered} recovered from older indexes)"
|
||||||
|
)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
@@ -18,8 +18,10 @@
|
|||||||
# swallow the whole message.
|
# swallow the whole message.
|
||||||
#
|
#
|
||||||
# Reports on:
|
# Reports on:
|
||||||
# - Sibling service reachability (slskd, navidrome) via HTTP, not docker ps —
|
# - Sibling service reachability (navidrome) via HTTP, not docker ps —
|
||||||
# alembic has no Docker socket access
|
# alembic has no Docker socket access. slskd is deliberately NOT checked:
|
||||||
|
# it is not part of the pipeline (sldl is its own Soulseek client) — it
|
||||||
|
# runs on the host purely as the user's own file-sharing presence.
|
||||||
# - Today's runs: playlist syncs, Bandcamp sync, manual imports, dedup
|
# - Today's runs: playlist syncs, Bandcamp sync, manual imports, dedup
|
||||||
# - Weekly maintenance freshness: strip-mb-tags, strip-watermark-art,
|
# - Weekly maintenance freshness: strip-mb-tags, strip-watermark-art,
|
||||||
# scrub-watermark-text, clean-sldl-index, spotify-genre
|
# scrub-watermark-text, clean-sldl-index, spotify-genre
|
||||||
@@ -188,7 +190,6 @@ dedup_today_status() {
|
|||||||
mark_warn "$name" "unreachable at $url"
|
mark_warn "$name" "unreachable at $url"
|
||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
check_http slskd "http://gluetun:5030/"
|
|
||||||
check_http navidrome "http://navidrome:4533/rest/ping.view"
|
check_http navidrome "http://navidrome:4533/rest/ping.view"
|
||||||
|
|
||||||
# 2. Today's runs
|
# 2. Today's runs
|
||||||
|
|||||||
@@ -195,7 +195,18 @@ while IFS= read -r -d '' new_file; do
|
|||||||
existing_id="${match%%$'\x1f'*}"
|
existing_id="${match%%$'\x1f'*}"
|
||||||
existing_container="${match#*$'\x1f'}"
|
existing_container="${match#*$'\x1f'}"
|
||||||
|
|
||||||
existing="${MUSIC_DATA_DIR:-/data/music}/Library${existing_container#/music}"
|
# beets displays real ${MUSIC_DATA_DIR}/Library/... paths since the
|
||||||
|
# 2026-07-08 path rewrite -- use them as-is; only a legacy /music/...
|
||||||
|
# display path still needs the prefix swap. Blindly prepending (the old
|
||||||
|
# behavior) doubled the prefix, so every match logged [stale-db], the MP3
|
||||||
|
# never got replaced, and the leftover-import step at the bottom imported
|
||||||
|
# the staged FLAC as a NEW track -- producing exactly the flac+mp3 twins
|
||||||
|
# this script exists to prevent.
|
||||||
|
if [[ "$existing_container" == /music/* ]]; then
|
||||||
|
existing="${MUSIC_DATA_DIR:-/data/music}/Library${existing_container#/music}"
|
||||||
|
else
|
||||||
|
existing="$existing_container"
|
||||||
|
fi
|
||||||
if [[ ! -f "$existing" ]]; then
|
if [[ ! -f "$existing" ]]; then
|
||||||
echo "[stale-db] beets has $existing_container but file missing" | tee -a "$LOG"
|
echo "[stale-db] beets has $existing_container but file missing" | tee -a "$LOG"
|
||||||
continue
|
continue
|
||||||
|
|||||||
Reference in New Issue
Block a user