Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 6114e6dc7a | |||
| f05f076464 | |||
| 62251d764e | |||
| d756aab7fc | |||
| 24738d815f |
@@ -73,10 +73,10 @@ Optional, add these later if you want them:
|
||||
Pull the prebuilt image onto your Docker host:
|
||||
|
||||
```bash
|
||||
docker pull git.kretzer.club/andrew/alembic:0.6.6
|
||||
docker pull git.kretzer.club/andrew/alembic:0.6.11
|
||||
```
|
||||
|
||||
That is the whole install. You do not need to download the source or build anything. The `0.6.6` is the version; you can pin to it so nothing changes under you, or use `latest` to always get the newest.
|
||||
That is the whole install. You do not need to download the source or build anything. The `0.6.11` is the version; you can pin to it so nothing changes under you, or use `latest` to always get the newest.
|
||||
|
||||
(If you would rather build it yourself from source, you can, but you do not need to.)
|
||||
|
||||
@@ -136,7 +136,7 @@ Create a file called `docker-compose.yml` on your server (put it wherever you ke
|
||||
```yaml
|
||||
services:
|
||||
alembic:
|
||||
image: git.kretzer.club/andrew/alembic:0.6.6
|
||||
image: git.kretzer.club/andrew/alembic:0.6.11
|
||||
container_name: alembic
|
||||
ports:
|
||||
- "8420:8420"
|
||||
|
||||
@@ -27,6 +27,38 @@ def _script_for_pass(pass_name: str) -> str:
|
||||
return _FUZZY_SCRIPT if pass_name == _FUZZY_PASS else _SCRIPT
|
||||
|
||||
|
||||
def _prune_stale_pending(db) -> int:
|
||||
"""Mark pending candidates whose delete_path or keep_path no longer
|
||||
exists as resolved, instead of leaving them to surface forever.
|
||||
|
||||
A pending row can go stale without ever being confirmed through the UI --
|
||||
e.g. a later scan's numbered-twin/format-upgrade auto-apply already
|
||||
removed one side, or upgrade-mp3-to-flac replaced it, or someone deleted
|
||||
it by hand. confirm_and_apply() already does this "already_gone" check
|
||||
for candidates a user explicitly acts on; this generalizes it to run at
|
||||
the start of every scan, so the pending list reflects current reality
|
||||
instead of stale groups that something else already resolved."""
|
||||
now = time.time()
|
||||
rows = db.execute(
|
||||
select(DedupCandidate).where(
|
||||
DedupCandidate.applied == False, # noqa: E712
|
||||
DedupCandidate.confirmed == False, # noqa: E712
|
||||
DedupCandidate.ignored == False, # noqa: E712
|
||||
)
|
||||
).scalars().all()
|
||||
pruned = 0
|
||||
for c in rows:
|
||||
if not Path(c.delete_path).exists() or not Path(c.keep_path).exists():
|
||||
c.confirmed = True
|
||||
c.confirmed_by = "auto:stale"
|
||||
c.confirmed_at = now
|
||||
c.applied = True
|
||||
pruned += 1
|
||||
if pruned:
|
||||
db.commit()
|
||||
return pruned
|
||||
|
||||
|
||||
def _is_pair_ignored(db, path_a: str, path_b: str) -> bool:
|
||||
"""True if this path pair was ever marked 'keep both', regardless of
|
||||
which path was on the keep/delete side that time -- a later scan can
|
||||
@@ -58,7 +90,9 @@ def _is_pair_already_pending(db, keep_path: str, delete_path: str) -> bool:
|
||||
return db.execute(query).first() is not None
|
||||
|
||||
|
||||
async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupRun | None:
|
||||
async def _run_scan(
|
||||
job_key: str, script_name: str, triggered_by: str, extra_args: tuple[str, ...] = ()
|
||||
) -> DedupRun | None:
|
||||
"""Returns None (persisting nothing) if the job never actually ran --
|
||||
e.g. skipped_lock because something else was using the pipeline lock
|
||||
at that moment. Same lesson as genre_review_service.run(): recording a
|
||||
@@ -67,7 +101,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
pending-candidates list isn't scoped to "latest run only"."""
|
||||
script = str(settings.pipeline_dir / "lib" / script_name)
|
||||
job_run, output = await pipeline_runner.run_job_capture(
|
||||
job_key, [script, "--json"], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
|
||||
job_key, [script, "--json", *extra_args], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
|
||||
)
|
||||
if job_run.status != "success":
|
||||
return None
|
||||
@@ -76,6 +110,8 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
_prune_stale_pending(db)
|
||||
now = time.time()
|
||||
dedup_run = DedupRun(
|
||||
started_at=job_run.started_at,
|
||||
finished_at=job_run.finished_at,
|
||||
@@ -91,7 +127,30 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
|
||||
skipped_ignored = 0
|
||||
skipped_duplicate = 0
|
||||
auto_applied = 0
|
||||
for c in candidates:
|
||||
if c.get("auto_applied"):
|
||||
# dedup-library.sh already deleted this pair unconditionally
|
||||
# (identical numbered-sibling twin: same dir/name/ext, no
|
||||
# tag/fuzzy ambiguity involved) -- record it pre-applied for
|
||||
# audit history; it never shows up as a pending review item.
|
||||
db.add(
|
||||
DedupCandidate(
|
||||
dedup_run_id=dedup_run.id,
|
||||
pass_name=c.get("pass", "unknown"),
|
||||
keep_path=c["keep_path"],
|
||||
delete_path=c["delete_path"],
|
||||
delete_id=c.get("delete_id"),
|
||||
delete_size_bytes=c.get("delete_size_bytes"),
|
||||
confirmed=True,
|
||||
confirmed_by="auto:numbered_twin",
|
||||
confirmed_at=now,
|
||||
applied=True,
|
||||
)
|
||||
)
|
||||
auto_applied += 1
|
||||
db.flush()
|
||||
continue
|
||||
if _is_pair_ignored(db, c["keep_path"], c["delete_path"]):
|
||||
skipped_ignored += 1
|
||||
continue
|
||||
@@ -116,8 +175,11 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
db.commit()
|
||||
db.refresh(dedup_run)
|
||||
skipped_total = skipped_ignored + skipped_duplicate
|
||||
if skipped_total:
|
||||
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
|
||||
if skipped_total or auto_applied:
|
||||
if skipped_total:
|
||||
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
|
||||
if auto_applied:
|
||||
dedup_run.deleted = auto_applied
|
||||
db.commit()
|
||||
db.refresh(dedup_run)
|
||||
return dedup_run
|
||||
@@ -126,12 +188,23 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
||||
|
||||
|
||||
async def scan(triggered_by: str = "manual") -> DedupRun | None:
|
||||
"""Dry-run dedup-library.sh --json, persist every candidate deletion
|
||||
into a fresh dedup_runs/dedup_candidates pair. Never deletes anything
|
||||
-- the scheduled maintenance:dedup job also only ever calls this (no
|
||||
--apply), matching the false-negative-biased dedup preference; actual
|
||||
deletion only ever happens through confirm_and_apply() below."""
|
||||
return await _run_scan("dedup:scan", _SCRIPT, triggered_by)
|
||||
"""Dry-run dedup-library.sh --json (plus --auto-apply-numbered), persist
|
||||
every candidate deletion into a fresh dedup_runs/dedup_candidates pair.
|
||||
|
||||
Two kinds of match need no human judgment and get deleted unconditionally
|
||||
by dedup-library.sh itself (see is_numbered_twin/is_format_upgrade there):
|
||||
identical numbered-sibling twins (same dir, same name, same extension,
|
||||
differing only by the ".N" collision suffix), and format upgrades within
|
||||
an EXACT tag match from Pass 2/3 (case-insensitive/normalized already
|
||||
proved same song via identical tags, so a differing extension there is
|
||||
just "better format vs. worse"). Everything else -- including Pass 4's
|
||||
cross-album fuzzy matches, where a wrong auto-delete could take out a
|
||||
genuinely different version (a different album pressing, a DJ-mix edit,
|
||||
etc.) -- stays a dry-run candidate awaiting manual confirm_and_apply()
|
||||
below; that's the false-negative-biased preference for anything with real
|
||||
ambiguity. Auto-applied deletions are reported here already applied,
|
||||
purely for audit visibility."""
|
||||
return await _run_scan("dedup:scan", _SCRIPT, triggered_by, extra_args=("--auto-apply-numbered",))
|
||||
|
||||
|
||||
async def scan_fuzzy(triggered_by: str = "manual") -> DedupRun | None:
|
||||
|
||||
@@ -96,6 +96,21 @@ else
|
||||
log "Fetched $TRACK_COUNT track(s) from Spotify"
|
||||
fi
|
||||
|
||||
# ==== Seed the pinned sldl skip index if it's missing or stranded ====
|
||||
# The conf pins index-path to $DROPBOX/_index.csv (see _template.conf: sldl's
|
||||
# default index location is derived from the input, so an input change strands
|
||||
# the old index and the whole playlist re-downloads -- that's what produced
|
||||
# the 2026-07-16 duplicate flood). If the pinned file is missing, or an index
|
||||
# in one of sldl's old input-named subfolders is newer (i.e. sldl last wrote
|
||||
# somewhere else), fold them all into the pinned location first.
|
||||
INDEX_FILE="$DROPBOX/_index.csv"
|
||||
if [[ ! -s "$INDEX_FILE" ]] || \
|
||||
[[ -n "$(find "$DROPBOX" -mindepth 2 -maxdepth 2 -name _index.csv -newer "$INDEX_FILE" -print -quit 2>/dev/null)" ]]; then
|
||||
log "Seeding pinned sldl index at $INDEX_FILE from prior indexes"
|
||||
python3 "${PIPELINE_DIR:-/app/pipeline}/lib/merge-sldl-indexes.py" "$DROPBOX" "$INDEX_FILE" >> "$LOG" 2>&1 \
|
||||
|| log "WARNING: index merge failed -- sldl may re-download tracks the library already has"
|
||||
fi
|
||||
|
||||
# ==== Run sldl (vendored binary, subprocess of this container) ====
|
||||
log "Running sldl for: $PLAYLIST_NAME"
|
||||
|
||||
|
||||
@@ -31,6 +31,18 @@ path = SLDL_DROPBOX_ROOT/PLAYLIST_NAME
|
||||
playlist-path = SLDL_DROPBOX_ROOT/PLAYLIST_NAME/_sldl.m3u8
|
||||
write-playlist = true
|
||||
|
||||
# ==== Skip index (pinned!) ====
|
||||
# sldl's already-downloaded index defaults to {path}/{playlist-name}/_index.csv
|
||||
# where {playlist-name} is derived from the INPUT: the Spotify playlist's
|
||||
# display name for spotify input, the CSV filename for csv input. beets moves
|
||||
# every download out of the dropbox, so this index is the ONLY thing standing
|
||||
# between a nightly run and re-downloading the whole playlist. When 0.6.2
|
||||
# switched input from Spotify URL to CSV, the derived name changed, sldl
|
||||
# started a fresh empty index in a new subfolder, and every playlist
|
||||
# re-downloaded in full on 2026-07-16. Pinning the path here decouples the
|
||||
# index from input naming so that can never happen again.
|
||||
index-path = SLDL_DROPBOX_ROOT/PLAYLIST_NAME/_index.csv
|
||||
|
||||
# ==== Naming ====
|
||||
name-format = {artist} - {title}
|
||||
|
||||
|
||||
+113
-12
@@ -34,11 +34,26 @@ set -euo pipefail
|
||||
APPLY=0
|
||||
JSON_MODE=0
|
||||
ONLY_PATHS_FILE=""
|
||||
AUTO_APPLY_NUMBERED=0
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--apply) APPLY=1; shift ;;
|
||||
--json) JSON_MODE=1; shift ;;
|
||||
--only-paths) ONLY_PATHS_FILE="$2"; shift 2 ;;
|
||||
# Delete two kinds of match unconditionally, independent of
|
||||
# --apply/--only-paths, while everything else stays dry-run/log-only:
|
||||
# - numbered-sibling twins (see is_numbered_twin): same dir, same name,
|
||||
# same extension, just the ".N" collision suffix -- the same download
|
||||
# landing twice.
|
||||
# - format upgrades within an exact tag match (see is_format_upgrade):
|
||||
# Pass 2/3 already proved same song via identical tags; a different
|
||||
# extension there just means "better format vs. worse", never a
|
||||
# different version.
|
||||
# Excludes Pass 4 (cross-album fuzzy) entirely -- matching across
|
||||
# different albums/directories is exactly where a different master or
|
||||
# DJ-mix edit can share tags without being interchangeable, so it always
|
||||
# needs a human. See dedup_review_service.scan().
|
||||
--auto-apply-numbered) AUTO_APPLY_NUMBERED=1; shift ;;
|
||||
*) shift ;;
|
||||
esac
|
||||
done
|
||||
@@ -78,11 +93,63 @@ fi
|
||||
# Counters: tracked via log grep at the end (avoids bash subshell pitfalls).
|
||||
# Functions write KEEP/DELETE lines with consistent prefixes; summary greps them.
|
||||
|
||||
# Convert container path (/music/...) to host path (${MUSIC_DATA_DIR:-/data/music}/Library/...)
|
||||
host_path() { echo "${MUSIC_DATA_DIR:-/data/music}/Library${1#/music}"; }
|
||||
# Resolve a beets-displayed path to a real path in this container. Since the
|
||||
# 2026-07-08 path rewrite, beets displays ${MUSIC_DATA_DIR:-/data/music}/Library/...
|
||||
# paths that are directly usable — pass those through UNCHANGED. Only legacy
|
||||
# /music/... display paths (the pre-rewrite transitional mount) still need the
|
||||
# prefix swap. Blindly prepending the library dir to an already-correct path
|
||||
# (the old behavior) produced /data/music/Library/data/music/Library/... —
|
||||
# nonexistent, so every candidate failed the -f check and every pass reported
|
||||
# 0 groups from 2026-07-08 until this fix.
|
||||
host_path() {
|
||||
local p="$1"
|
||||
if [[ "$p" == /music/* ]]; then
|
||||
echo "${MUSIC_DATA_DIR:-/data/music}/Library${p#/music}"
|
||||
else
|
||||
echo "$p"
|
||||
fi
|
||||
}
|
||||
|
||||
SEP=$'\x1c' # ASCII file separator — safe with any music metadata
|
||||
|
||||
# True if two paths are the SAME directory, SAME extension, and identical
|
||||
# basenames except for a numeric ".N" infix sldl/beets inserts on a filename
|
||||
# collision (e.g. "Title.flac" vs "Title.1.flac"). This is deliberately
|
||||
# narrower than any tag-based pass: no fuzzy matching, no cross-album
|
||||
# ambiguity, just the literal same download landing twice. Used to
|
||||
# auto-delete regardless of which pass found the pair -- Pass 1 only fires
|
||||
# when the canonical name is a beets ghost; the very common case where BOTH
|
||||
# copies are already tracked in beets (so Pass 1 defers) surfaces instead via
|
||||
# Pass 2/3's tag-based grouping, and still deserves the same free pass.
|
||||
is_numbered_twin() {
|
||||
local a="$1" b="$2"
|
||||
[[ "${a%/*}" == "${b%/*}" ]] || return 1
|
||||
local ba="${a##*/}" bb="${b##*/}"
|
||||
local ext="${ba##*.}"
|
||||
[[ "${ext,,}" == "${bb##*.}" ]] || return 1
|
||||
[[ "$(echo "$ba" | sed -E "s/\.[0-9]+\.${ext}\$/.${ext}/")" \
|
||||
== "$(echo "$bb" | sed -E "s/\.[0-9]+\.${ext}\$/.${ext}/")" ]]
|
||||
}
|
||||
|
||||
# True if this pair is a same-song format upgrade found by an EXACT tag match
|
||||
# (Pass 2 case-insensitive or Pass 3 normalized -- both require identical
|
||||
# albumartist+album+title after case/punctuation folding, so "same song" is
|
||||
# already proven by the pass itself) where the two files simply have
|
||||
# different extensions. rank_file() always ranks FLAC ahead of any other
|
||||
# format regardless of size/dirty-name/etc, so within a tag-exact group a
|
||||
# cross-extension pair unconditionally means "the format-preferred copy vs.
|
||||
# a lesser one" -- no judgment call. Deliberately excludes Pass 4
|
||||
# (cross_album_fuzzy): that pass matches by artist+title-base ACROSS
|
||||
# different albums/directories, which is exactly where a different master,
|
||||
# DJ-mix edit, or reissue can share tags without being an interchangeable
|
||||
# copy -- those still need a human to look at album/context before deleting
|
||||
# either side.
|
||||
is_format_upgrade() {
|
||||
local pass="$1" keep="$2" delete="$3"
|
||||
[[ "$pass" == "case_insensitive" || "$pass" == "normalized" ]] || return 1
|
||||
[[ "${keep##*.}" != "${delete##*.}" ]]
|
||||
}
|
||||
|
||||
# Rank a file: lower = better. FLAC > MP3 > other (WAV/etc.); clean filename
|
||||
# > sync-conflict artifact (-DESKTOP-, (1), .1.); extended/club mix > radio
|
||||
# edit at same format; then larger wins.
|
||||
@@ -163,7 +230,7 @@ process_group() {
|
||||
esac
|
||||
|
||||
local first=1
|
||||
local keep_path=""
|
||||
local keep_path="" keep_id="" any_auto=0
|
||||
while IFS="$SEP" read -r _sk id hp; do
|
||||
[[ -z "$hp" ]] && continue
|
||||
local size_bytes; size_bytes=$(stat -c '%s' "$hp" 2>/dev/null || echo 0)
|
||||
@@ -171,14 +238,31 @@ process_group() {
|
||||
if [[ $first -eq 1 ]]; then
|
||||
log " KEEP ($size_h) $hp"
|
||||
keep_path="$hp"
|
||||
keep_id="$id"
|
||||
first=0
|
||||
else
|
||||
log " DELETE ($size_h) $hp"
|
||||
json_emit "{\"pass\":\"${pass_name}\",\"keep_path\":\"$(json_escape "$keep_path")\",\"delete_path\":\"$(json_escape "$hp")\",\"delete_size_bytes\":${size_bytes}}"
|
||||
|
||||
# Auto-apply-numbered deletes numbered_sibling matches unconditionally,
|
||||
# bypassing --only-paths (that gate exists only for the manual confirm
|
||||
# flow, which never runs alongside this flag). All other passes stay
|
||||
# dry-run/log-only here regardless.
|
||||
local do_delete=0 auto_applied=0
|
||||
if [[ $APPLY -eq 1 ]]; then
|
||||
do_delete=1
|
||||
elif [[ $AUTO_APPLY_NUMBERED -eq 1 ]] \
|
||||
&& { is_numbered_twin "$keep_path" "$hp" || is_format_upgrade "$pass_name" "$keep_path" "$hp"; }; then
|
||||
do_delete=1
|
||||
auto_applied=1
|
||||
any_auto=1
|
||||
fi
|
||||
local auto_json="false"; [[ $auto_applied -eq 1 ]] && auto_json="true"
|
||||
json_emit "{\"pass\":\"${pass_name}\",\"keep_path\":\"$(json_escape "$keep_path")\",\"delete_path\":\"$(json_escape "$hp")\",\"delete_size_bytes\":${size_bytes},\"auto_applied\":${auto_json}}"
|
||||
|
||||
if [[ $do_delete -eq 1 ]]; then
|
||||
# With --only-paths given, only delete entries the caller explicitly
|
||||
# confirmed; without it, delete everything (original behavior).
|
||||
if [[ -n "$ONLY_PATHS_FILE" && -z "${ONLY_PATHS[$hp]:-}" ]]; then
|
||||
if [[ $auto_applied -eq 0 && -n "$ONLY_PATHS_FILE" && -z "${ONLY_PATHS[$hp]:-}" ]]; then
|
||||
log " (skipped — not in --only-paths confirm list)"
|
||||
elif [[ -n "$id" ]]; then
|
||||
# In beets — beet remove -d removes from DB + disk. Best-effort: a
|
||||
@@ -192,6 +276,18 @@ process_group() {
|
||||
fi
|
||||
fi
|
||||
done < <(echo "$ranked" | grep -v '^$' | sort -n)
|
||||
|
||||
# An auto-applied numbered-twin deletion can leave the numbered-named file
|
||||
# as the sole survivor (it won on size despite the dirty-filename penalty).
|
||||
# Rename it back to the canonical name so the library doesn't accumulate
|
||||
# ".1."/".2." filenames for tracks that no longer have a duplicate. Only
|
||||
# auto_applied triggers this (never a manual --only-paths confirm, which
|
||||
# may leave the sibling undeleted if the user didn't confirm it -- renaming
|
||||
# then would collide); only when it's actually in beets (a ghost survivor
|
||||
# has nothing to move); only when the name still has the dirty infix.
|
||||
if [[ $any_auto -eq 1 && -n "$keep_id" && "$keep_path" =~ \.[0-9]+\.[A-Za-z0-9]+$ ]]; then
|
||||
beet move "id:${keep_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${keep_id} failed (rename survivor)"
|
||||
fi
|
||||
return 0 # explicit: the loop's last command status must not leak out under set -e
|
||||
}
|
||||
|
||||
@@ -202,9 +298,10 @@ process_group() {
|
||||
log ""
|
||||
log "[$(date -Iseconds)] === Pass 1: numbered siblings ==="
|
||||
|
||||
# Dump id + container path from beets once. beets normalizes $path for
|
||||
# display (always /music/-prefixed) even though the DB stores a mix of
|
||||
# absolute and relative — so the display paths are safe to compare/convert,
|
||||
# Dump id + display path from beets once. beets normalizes $path for display
|
||||
# (real /data/music/Library/... paths since the 2026-07-08 rewrite; /music/...
|
||||
# on any legacy entry) even though the DB stores a mix of absolute and
|
||||
# relative — so the display paths are safe to compare/convert via host_path(),
|
||||
# and the ids are what we hand to beet remove/move.
|
||||
beet ls -f "\$id${SEP}\$path" 2>/dev/null > /tmp/beets-id-paths.txt
|
||||
cut -d"$SEP" -f2- /tmp/beets-id-paths.txt > /tmp/beets-all-paths.txt
|
||||
@@ -231,18 +328,22 @@ while IFS="$SEP" read -r numbered_id numbered_cp; do
|
||||
|
||||
if [[ $score_canonical -le $score_numbered ]]; then
|
||||
# Canonical wins: delete numbered (in beets), keep canonical (ghost)
|
||||
process_group "P1 numbered-sibling: ${numbered_cp##/music/}" \
|
||||
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
|
||||
"$host_canonical" "${numbered_id}${SEP}${numbered_cp}"
|
||||
if [[ $APPLY -eq 1 ]]; then
|
||||
if [[ $APPLY -eq 1 || $AUTO_APPLY_NUMBERED -eq 1 ]]; then
|
||||
# canonical is now a ghost file; import it into beets
|
||||
beet import -q -s "${canonical_cp}" >> "$LOG" 2>&1 || log " WARN: beet import failed for ${canonical_cp}"
|
||||
fi
|
||||
else
|
||||
# Numbered wins (canonical is ghost): rm the ghost, then beet move to rename
|
||||
# the numbered file to the canonical path (id query — see process_group).
|
||||
process_group "P1 numbered-sibling: ${numbered_cp##/music/}" \
|
||||
# Safe to reach with just AUTO_APPLY_NUMBERED: this loop only ever builds
|
||||
# genuine numbered-sibling pairs (see is_numbered_twin), and process_group
|
||||
# above already deleted the ghost canonical before this runs, so the move
|
||||
# target is guaranteed clear -- no rename collision.
|
||||
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
|
||||
"${numbered_id}${SEP}${numbered_cp}" "$host_canonical"
|
||||
if [[ $APPLY -eq 1 ]]; then
|
||||
if [[ $APPLY -eq 1 || $AUTO_APPLY_NUMBERED -eq 1 ]]; then
|
||||
beet move "id:${numbered_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${numbered_id} failed"
|
||||
fi
|
||||
fi
|
||||
|
||||
@@ -57,11 +57,12 @@ OUT_DIR = f"{_MUSIC_DATA_DIR}/playlists-laptop"
|
||||
# Absolute Windows path to the laptop's library root.
|
||||
LIBRARY_LAPTOP_ROOT = r"C:\Music\Library"
|
||||
|
||||
# Beets stores paths with this prefix (the container-side mount of the
|
||||
# library); strip it to get the path relative to the library root. This is
|
||||
# still "/music/" through the migration's Stage 0-3 transitional mount;
|
||||
# revisit at Stage 4 once beets' directory: becomes MUSIC_DATA_DIR/Library.
|
||||
BEETS_LIBRARY_PREFIX = "/music/"
|
||||
# Prefixes beets may display library paths under, tried in order: the real
|
||||
# library dir (everything since the 2026-07-08 path rewrite) and the legacy
|
||||
# transitional mount (any stale pre-rewrite entry). Strip whichever matches
|
||||
# to get the path relative to the library root; when neither matches the
|
||||
# track is outside the library and is skipped.
|
||||
BEETS_LIBRARY_PREFIXES = (f"{_MUSIC_DATA_DIR}/Library/", "/music/")
|
||||
|
||||
M3U_EXT = ".m3u"
|
||||
# Older formats from earlier iterations of this script — swept by reconcile.
|
||||
@@ -157,9 +158,12 @@ def build_beets_index() -> dict[tuple[str, str], list[tuple[str, str, str, str]]
|
||||
if len(parts) != 4:
|
||||
continue
|
||||
aa, alb, ti, path = parts
|
||||
if not path.startswith(BEETS_LIBRARY_PREFIX):
|
||||
rel = next(
|
||||
(path[len(p):] for p in BEETS_LIBRARY_PREFIXES if path.startswith(p)),
|
||||
None,
|
||||
)
|
||||
if rel is None:
|
||||
continue
|
||||
rel = path[len(BEETS_LIBRARY_PREFIX):]
|
||||
idx.setdefault((_normkey(alb), _normkey(ti)), []).append((aa, alb, ti, rel))
|
||||
return idx
|
||||
|
||||
|
||||
@@ -49,8 +49,10 @@ _ALEMBIC_CONFIG_DIR = os.environ.get("ALEMBIC_CONFIG_DIR", "/config")
|
||||
_MUSIC_DATA_DIR = os.environ.get("MUSIC_DATA_DIR", "/data/music")
|
||||
|
||||
LIBRARY = f"{_MUSIC_DATA_DIR}/Library"
|
||||
# Still "/music" through the migration's Stage 0-3 transitional beets mount;
|
||||
# revisit at Stage 4 once beets' directory: becomes MUSIC_DATA_DIR/Library.
|
||||
# Legacy prefix from the migration's transitional beets mount. Since the
|
||||
# 2026-07-08 path rewrite beets displays real LIBRARY paths, so this only
|
||||
# matters for accepting pasted /music/... paths as input and as a fallback
|
||||
# beets query form for any stale pre-rewrite DB entry.
|
||||
CONTAINER_PREFIX = "/music"
|
||||
SPOTIFY_ENV = f"{_ALEMBIC_CONFIG_DIR}/pipeline/_spotify.env"
|
||||
SPOTIFY_GENRE = f"{os.environ.get('PIPELINE_DIR', '/app/pipeline')}/lib/spotify-genre.py"
|
||||
@@ -316,15 +318,19 @@ def resolve_input_path(arg):
|
||||
|
||||
|
||||
def beets_id_for(host_path):
|
||||
cp = host_to_container(host_path)
|
||||
r = subprocess.run(
|
||||
["beet", "ls", "-f", "$id|||$path", f"path:{cp}"],
|
||||
capture_output=True, text=True, check=True)
|
||||
for line in r.stdout.splitlines():
|
||||
if "|||" in line:
|
||||
id_str, p = line.split("|||", 1)
|
||||
if p == cp:
|
||||
return int(id_str)
|
||||
# Real path first (how beets displays everything since the 2026-07-08
|
||||
# path rewrite), legacy /music/... form second for any stale entry.
|
||||
# Querying only the legacy form (the old behavior) matched nothing after
|
||||
# the rewrite, so retag-from-url silently skipped beet update/move.
|
||||
for cp in (host_path, host_to_container(host_path)):
|
||||
r = subprocess.run(
|
||||
["beet", "ls", "-f", "$id|||$path", f"path:{cp}"],
|
||||
capture_output=True, text=True, check=True)
|
||||
for line in r.stdout.splitlines():
|
||||
if "|||" in line:
|
||||
id_str, p = line.split("|||", 1)
|
||||
if p == cp:
|
||||
return int(id_str)
|
||||
return None
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fold every sldl skip index under a playlist's dropbox into one pinned file.
|
||||
|
||||
sldl names its per-playlist index folder after the *input*: the Spotify
|
||||
playlist's display name for spotify input, the CSV filename stem for csv
|
||||
input. So when 0.6.2 switched input from Spotify URL to CSV, sldl started a
|
||||
fresh empty index in a new subfolder, saw no download history, and
|
||||
re-downloaded every playlist in full (2026-07-16). The rendered confs now pin
|
||||
index-path to <dropbox>/<playlist>/_index.csv; this script seeds that pinned
|
||||
file from all the indexes sldl left behind (run-playlist.sh calls it whenever
|
||||
the pinned file is missing or older than a stranded one).
|
||||
|
||||
Usage: merge-sldl-indexes.py <playlist_dropbox_dir> <output_index_csv>
|
||||
|
||||
Merge rules, per (artist, title) lowercased:
|
||||
- the row from the newest index file wins;
|
||||
- EXCEPT when that row is a failure (state 2) and any older index has the
|
||||
track as downloaded (state 1 or 3): then the newest row's identity
|
||||
(artist/album/title/length -- matching what the current input will
|
||||
present; Spotify length rounding drifted between the old extractor and
|
||||
our CSV) is kept but marked state 3 (already downloaded), so sldl does
|
||||
not re-fetch a track the library already holds;
|
||||
- rows only present in older indexes are kept as-is (tracks since removed
|
||||
from the playlist; harmless, and they keep their history if re-added).
|
||||
"""
|
||||
|
||||
import csv
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
HEADER = ["filepath", "artist", "album", "title", "length", "tracktype", "state", "failurereason"]
|
||||
DOWNLOADED_STATES = {"1", "3"} # 1 = downloaded this run, 3 = found in index previously
|
||||
FAILED_STATE = "2"
|
||||
|
||||
|
||||
def read_rows(path: Path) -> list[dict]:
|
||||
rows = []
|
||||
with open(path, newline="", encoding="utf-8") as fh:
|
||||
reader = csv.DictReader(fh)
|
||||
for row in reader:
|
||||
if row.get("artist") is None or row.get("title") is None:
|
||||
continue
|
||||
rows.append({k: (row.get(k) or "") for k in HEADER})
|
||||
return rows
|
||||
|
||||
|
||||
def norm_key(row: dict) -> tuple:
|
||||
# Primary artist only: sldl's old Spotify extractor recorded just the
|
||||
# first artist, while spotify-playlist-csv.py joins all of them with
|
||||
# ", " -- keying on the full string would miss every multi-artist track
|
||||
# when recovering history across the two index generations. First
|
||||
# comma-segment matches both forms (and both sides of a comma-in-name
|
||||
# artist like "Tyler, The Creator" truncate identically).
|
||||
return (row["artist"].split(",")[0].strip().lower(), row["title"].strip().lower())
|
||||
|
||||
|
||||
def main() -> int:
|
||||
if len(sys.argv) != 3:
|
||||
print(__doc__, file=sys.stderr)
|
||||
return 2
|
||||
|
||||
dropbox = Path(sys.argv[1])
|
||||
out_path = Path(sys.argv[2])
|
||||
if not dropbox.is_dir():
|
||||
print(f"[merge-sldl-indexes] not a directory: {dropbox}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
# Every index at the dropbox root or one level down (sldl's input-named
|
||||
# subfolders), including the pinned output itself if it already exists --
|
||||
# newest first, so the most recent record of each track wins.
|
||||
candidates = sorted(
|
||||
set(dropbox.glob("_index.csv")) | set(dropbox.glob("*/_index.csv")),
|
||||
key=lambda p: p.stat().st_mtime,
|
||||
reverse=True,
|
||||
)
|
||||
if not candidates:
|
||||
print(f"[merge-sldl-indexes] no _index.csv found under {dropbox}; nothing to seed")
|
||||
return 0
|
||||
|
||||
merged: dict[tuple, dict] = {}
|
||||
recovered = 0
|
||||
for path in candidates:
|
||||
try:
|
||||
rows = read_rows(path)
|
||||
except (OSError, csv.Error) as exc:
|
||||
print(f"[merge-sldl-indexes] skipping unreadable {path}: {exc}", file=sys.stderr)
|
||||
continue
|
||||
print(f"[merge-sldl-indexes] {path}: {len(rows)} rows")
|
||||
for row in rows:
|
||||
key = norm_key(row)
|
||||
kept = merged.get(key)
|
||||
if kept is None:
|
||||
merged[key] = row
|
||||
elif kept["state"] == FAILED_STATE and row["state"] in DOWNLOADED_STATES:
|
||||
# Newest attempt failed but an older index proves we already
|
||||
# have this track: keep the newest identity fields, take the
|
||||
# old filepath (informational only; skip-mode index never
|
||||
# checks the file on disk), and mark it downloaded.
|
||||
kept["filepath"] = row["filepath"]
|
||||
kept["state"] = "3"
|
||||
kept["failurereason"] = "0"
|
||||
recovered += 1
|
||||
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = out_path.with_suffix(".csv.tmp")
|
||||
with open(tmp, "w", newline="", encoding="utf-8") as fh:
|
||||
writer = csv.DictWriter(fh, fieldnames=HEADER)
|
||||
writer.writeheader()
|
||||
writer.writerows(merged.values())
|
||||
tmp.replace(out_path)
|
||||
|
||||
downloaded = sum(1 for r in merged.values() if r["state"] in DOWNLOADED_STATES)
|
||||
print(
|
||||
f"[merge-sldl-indexes] wrote {out_path}: {len(merged)} tracks "
|
||||
f"({downloaded} downloaded, {recovered} recovered from older indexes)"
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -18,8 +18,10 @@
|
||||
# swallow the whole message.
|
||||
#
|
||||
# Reports on:
|
||||
# - Sibling service reachability (slskd, navidrome) via HTTP, not docker ps —
|
||||
# alembic has no Docker socket access
|
||||
# - Sibling service reachability (navidrome) via HTTP, not docker ps —
|
||||
# alembic has no Docker socket access. slskd is deliberately NOT checked:
|
||||
# it is not part of the pipeline (sldl is its own Soulseek client) — it
|
||||
# runs on the host purely as the user's own file-sharing presence.
|
||||
# - Today's runs: playlist syncs, Bandcamp sync, manual imports, dedup
|
||||
# - Weekly maintenance freshness: strip-mb-tags, strip-watermark-art,
|
||||
# scrub-watermark-text, clean-sldl-index, spotify-genre
|
||||
@@ -188,7 +190,6 @@ dedup_today_status() {
|
||||
mark_warn "$name" "unreachable at $url"
|
||||
fi
|
||||
}
|
||||
check_http slskd "http://gluetun:5030/"
|
||||
check_http navidrome "http://navidrome:4533/rest/ping.view"
|
||||
|
||||
# 2. Today's runs
|
||||
|
||||
@@ -195,7 +195,18 @@ while IFS= read -r -d '' new_file; do
|
||||
existing_id="${match%%$'\x1f'*}"
|
||||
existing_container="${match#*$'\x1f'}"
|
||||
|
||||
existing="${MUSIC_DATA_DIR:-/data/music}/Library${existing_container#/music}"
|
||||
# beets displays real ${MUSIC_DATA_DIR}/Library/... paths since the
|
||||
# 2026-07-08 path rewrite -- use them as-is; only a legacy /music/...
|
||||
# display path still needs the prefix swap. Blindly prepending (the old
|
||||
# behavior) doubled the prefix, so every match logged [stale-db], the MP3
|
||||
# never got replaced, and the leftover-import step at the bottom imported
|
||||
# the staged FLAC as a NEW track -- producing exactly the flac+mp3 twins
|
||||
# this script exists to prevent.
|
||||
if [[ "$existing_container" == /music/* ]]; then
|
||||
existing="${MUSIC_DATA_DIR:-/data/music}/Library${existing_container#/music}"
|
||||
else
|
||||
existing="$existing_container"
|
||||
fi
|
||||
if [[ ! -f "$existing" ]]; then
|
||||
echo "[stale-db] beets has $existing_container but file missing" | tee -a "$LOG"
|
||||
continue
|
||||
|
||||
Reference in New Issue
Block a user