Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 068e7e9534 | |||
| e07e931c45 | |||
| 6114e6dc7a | |||
| f05f076464 | |||
| 62251d764e |
@@ -73,10 +73,10 @@ Optional, add these later if you want them:
|
|||||||
Pull the prebuilt image onto your Docker host:
|
Pull the prebuilt image onto your Docker host:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
docker pull git.kretzer.club/andrew/alembic:0.6.8
|
docker pull git.kretzer.club/andrew/alembic:0.6.13
|
||||||
```
|
```
|
||||||
|
|
||||||
That is the whole install. You do not need to download the source or build anything. The `0.6.8` is the version; you can pin to it so nothing changes under you, or use `latest` to always get the newest.
|
That is the whole install. You do not need to download the source or build anything. The `0.6.13` is the version; you can pin to it so nothing changes under you, or use `latest` to always get the newest.
|
||||||
|
|
||||||
(If you would rather build it yourself from source, you can, but you do not need to.)
|
(If you would rather build it yourself from source, you can, but you do not need to.)
|
||||||
|
|
||||||
@@ -136,7 +136,7 @@ Create a file called `docker-compose.yml` on your server (put it wherever you ke
|
|||||||
```yaml
|
```yaml
|
||||||
services:
|
services:
|
||||||
alembic:
|
alembic:
|
||||||
image: git.kretzer.club/andrew/alembic:0.6.8
|
image: git.kretzer.club/andrew/alembic:0.6.13
|
||||||
container_name: alembic
|
container_name: alembic
|
||||||
ports:
|
ports:
|
||||||
- "8420:8420"
|
- "8420:8420"
|
||||||
|
|||||||
+68
-5
@@ -1,3 +1,6 @@
|
|||||||
|
import os
|
||||||
|
import re
|
||||||
|
|
||||||
from fastapi import APIRouter, BackgroundTasks, Depends, Request
|
from fastapi import APIRouter, BackgroundTasks, Depends, Request
|
||||||
from fastapi.responses import RedirectResponse
|
from fastapi.responses import RedirectResponse
|
||||||
from fastapi.templating import Jinja2Templates
|
from fastapi.templating import Jinja2Templates
|
||||||
@@ -30,16 +33,76 @@ def _display_path(path: str) -> str:
|
|||||||
return path[len(_LIBRARY_PREFIX):] if path.startswith(_LIBRARY_PREFIX) else path
|
return path[len(_LIBRARY_PREFIX):] if path.startswith(_LIBRARY_PREFIX) else path
|
||||||
|
|
||||||
|
|
||||||
|
# Compilation-style import paths carry a "$track " prefix on the filename
|
||||||
|
# (see pipeline/configs/beets config.yaml paths: comp/albumtype_soundtrack);
|
||||||
|
# singleton paths (the vast majority of this library) don't. Strip it either
|
||||||
|
# way so the title reads clean.
|
||||||
|
_TRACK_PREFIX_RE = re.compile(r"^\d{1,3}[\s.\-]+")
|
||||||
|
|
||||||
|
# rank_file() in dedup-library.sh always ranks FLAC above every other format
|
||||||
|
# and WAV below MP3 (a download glitch, never desirable) -- mirror that
|
||||||
|
# judgment in the format pill's color so it reads as "this one's the keeper"
|
||||||
|
# at a glance, not just a bare extension string.
|
||||||
|
_FORMAT_BADGE = {"flac": "badge-success", "wav": "badge-warning"}
|
||||||
|
|
||||||
|
|
||||||
|
def _split_display(path: str) -> dict:
|
||||||
|
"""Break a library display path into (artist, album, title, ext).
|
||||||
|
|
||||||
|
Beets' singleton path template is always
|
||||||
|
%the{$albumartist}/$album/$title (pipeline/configs/beets config.yaml,
|
||||||
|
paths:) -- every track in this library lands at exactly that shape, so
|
||||||
|
the last two segments are reliably album/filename. A path that doesn't
|
||||||
|
fit (an unexpected root-level file) degrades to showing the raw display
|
||||||
|
path as the title instead of guessing at structure that isn't there.
|
||||||
|
"""
|
||||||
|
display = _display_path(path)
|
||||||
|
parts = display.split("/")
|
||||||
|
if len(parts) >= 2:
|
||||||
|
artist, album, filename = parts[0], parts[-2], parts[-1]
|
||||||
|
title, ext = os.path.splitext(filename)
|
||||||
|
title = _TRACK_PREFIX_RE.sub("", title)
|
||||||
|
else:
|
||||||
|
artist, album = "", ""
|
||||||
|
title, ext = os.path.splitext(display)
|
||||||
|
return {
|
||||||
|
"path": path,
|
||||||
|
"display": display,
|
||||||
|
"artist": artist,
|
||||||
|
"album": album,
|
||||||
|
"title": title,
|
||||||
|
"ext": ext.lstrip(".").lower(),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _file_size(path: str) -> int | None:
|
||||||
|
try:
|
||||||
|
return os.path.getsize(path)
|
||||||
|
except OSError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def _to_row(c) -> dict:
|
def _to_row(c) -> dict:
|
||||||
|
keep = _split_display(c.keep_path)
|
||||||
|
delete = _split_display(c.delete_path)
|
||||||
|
keep["size_bytes"] = _file_size(c.keep_path)
|
||||||
|
keep["badge"] = _FORMAT_BADGE.get(keep["ext"], "badge-muted")
|
||||||
|
delete["size_bytes"] = c.delete_size_bytes if c.delete_size_bytes is not None else _file_size(c.delete_path)
|
||||||
|
delete["badge"] = _FORMAT_BADGE.get(delete["ext"], "badge-muted")
|
||||||
return {
|
return {
|
||||||
"id": c.id,
|
"id": c.id,
|
||||||
"pass_name": c.pass_name,
|
"pass_name": c.pass_name,
|
||||||
"pass_label": _pass_label(c.pass_name),
|
"pass_label": _pass_label(c.pass_name),
|
||||||
"keep_path": c.keep_path,
|
"keep": keep,
|
||||||
"delete_path": c.delete_path,
|
"delete": delete,
|
||||||
"keep_display": _display_path(c.keep_path),
|
# Same-tag passes (numbered_sibling/case_insensitive/normalized) share
|
||||||
"delete_display": _display_path(c.delete_path),
|
# artist/title almost always; cross_album_fuzzy and fuzzy_audio can
|
||||||
"delete_size_bytes": c.delete_size_bytes,
|
# legitimately differ (different credit, different album entirely) --
|
||||||
|
# the template only collapses shared context onto one line when it's
|
||||||
|
# actually shared, otherwise shows both sides in full.
|
||||||
|
"same_title": keep["title"].lower() == delete["title"].lower(),
|
||||||
|
"same_artist": keep["artist"].lower() == delete["artist"].lower(),
|
||||||
|
"same_album": keep["album"].lower() == delete["album"].lower(),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -27,6 +27,38 @@ def _script_for_pass(pass_name: str) -> str:
|
|||||||
return _FUZZY_SCRIPT if pass_name == _FUZZY_PASS else _SCRIPT
|
return _FUZZY_SCRIPT if pass_name == _FUZZY_PASS else _SCRIPT
|
||||||
|
|
||||||
|
|
||||||
|
def _prune_stale_pending(db) -> int:
|
||||||
|
"""Mark pending candidates whose delete_path or keep_path no longer
|
||||||
|
exists as resolved, instead of leaving them to surface forever.
|
||||||
|
|
||||||
|
A pending row can go stale without ever being confirmed through the UI --
|
||||||
|
e.g. a later scan's numbered-twin/format-upgrade auto-apply already
|
||||||
|
removed one side, or upgrade-mp3-to-flac replaced it, or someone deleted
|
||||||
|
it by hand. confirm_and_apply() already does this "already_gone" check
|
||||||
|
for candidates a user explicitly acts on; this generalizes it to run at
|
||||||
|
the start of every scan, so the pending list reflects current reality
|
||||||
|
instead of stale groups that something else already resolved."""
|
||||||
|
now = time.time()
|
||||||
|
rows = db.execute(
|
||||||
|
select(DedupCandidate).where(
|
||||||
|
DedupCandidate.applied == False, # noqa: E712
|
||||||
|
DedupCandidate.confirmed == False, # noqa: E712
|
||||||
|
DedupCandidate.ignored == False, # noqa: E712
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
pruned = 0
|
||||||
|
for c in rows:
|
||||||
|
if not Path(c.delete_path).exists() or not Path(c.keep_path).exists():
|
||||||
|
c.confirmed = True
|
||||||
|
c.confirmed_by = "auto:stale"
|
||||||
|
c.confirmed_at = now
|
||||||
|
c.applied = True
|
||||||
|
pruned += 1
|
||||||
|
if pruned:
|
||||||
|
db.commit()
|
||||||
|
return pruned
|
||||||
|
|
||||||
|
|
||||||
def _is_pair_ignored(db, path_a: str, path_b: str) -> bool:
|
def _is_pair_ignored(db, path_a: str, path_b: str) -> bool:
|
||||||
"""True if this path pair was ever marked 'keep both', regardless of
|
"""True if this path pair was ever marked 'keep both', regardless of
|
||||||
which path was on the keep/delete side that time -- a later scan can
|
which path was on the keep/delete side that time -- a later scan can
|
||||||
@@ -58,7 +90,9 @@ def _is_pair_already_pending(db, keep_path: str, delete_path: str) -> bool:
|
|||||||
return db.execute(query).first() is not None
|
return db.execute(query).first() is not None
|
||||||
|
|
||||||
|
|
||||||
async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupRun | None:
|
async def _run_scan(
|
||||||
|
job_key: str, script_name: str, triggered_by: str, extra_args: tuple[str, ...] = ()
|
||||||
|
) -> DedupRun | None:
|
||||||
"""Returns None (persisting nothing) if the job never actually ran --
|
"""Returns None (persisting nothing) if the job never actually ran --
|
||||||
e.g. skipped_lock because something else was using the pipeline lock
|
e.g. skipped_lock because something else was using the pipeline lock
|
||||||
at that moment. Same lesson as genre_review_service.run(): recording a
|
at that moment. Same lesson as genre_review_service.run(): recording a
|
||||||
@@ -67,7 +101,7 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
pending-candidates list isn't scoped to "latest run only"."""
|
pending-candidates list isn't scoped to "latest run only"."""
|
||||||
script = str(settings.pipeline_dir / "lib" / script_name)
|
script = str(settings.pipeline_dir / "lib" / script_name)
|
||||||
job_run, output = await pipeline_runner.run_job_capture(
|
job_run, output = await pipeline_runner.run_job_capture(
|
||||||
job_key, [script, "--json"], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
|
job_key, [script, "--json", *extra_args], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
|
||||||
)
|
)
|
||||||
if job_run.status != "success":
|
if job_run.status != "success":
|
||||||
return None
|
return None
|
||||||
@@ -76,6 +110,8 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
|
|
||||||
db = SessionLocal()
|
db = SessionLocal()
|
||||||
try:
|
try:
|
||||||
|
_prune_stale_pending(db)
|
||||||
|
now = time.time()
|
||||||
dedup_run = DedupRun(
|
dedup_run = DedupRun(
|
||||||
started_at=job_run.started_at,
|
started_at=job_run.started_at,
|
||||||
finished_at=job_run.finished_at,
|
finished_at=job_run.finished_at,
|
||||||
@@ -91,7 +127,30 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
|
|
||||||
skipped_ignored = 0
|
skipped_ignored = 0
|
||||||
skipped_duplicate = 0
|
skipped_duplicate = 0
|
||||||
|
auto_applied = 0
|
||||||
for c in candidates:
|
for c in candidates:
|
||||||
|
if c.get("auto_applied"):
|
||||||
|
# dedup-library.sh already deleted this pair unconditionally
|
||||||
|
# (identical numbered-sibling twin: same dir/name/ext, no
|
||||||
|
# tag/fuzzy ambiguity involved) -- record it pre-applied for
|
||||||
|
# audit history; it never shows up as a pending review item.
|
||||||
|
db.add(
|
||||||
|
DedupCandidate(
|
||||||
|
dedup_run_id=dedup_run.id,
|
||||||
|
pass_name=c.get("pass", "unknown"),
|
||||||
|
keep_path=c["keep_path"],
|
||||||
|
delete_path=c["delete_path"],
|
||||||
|
delete_id=c.get("delete_id"),
|
||||||
|
delete_size_bytes=c.get("delete_size_bytes"),
|
||||||
|
confirmed=True,
|
||||||
|
confirmed_by="auto:numbered_twin",
|
||||||
|
confirmed_at=now,
|
||||||
|
applied=True,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
auto_applied += 1
|
||||||
|
db.flush()
|
||||||
|
continue
|
||||||
if _is_pair_ignored(db, c["keep_path"], c["delete_path"]):
|
if _is_pair_ignored(db, c["keep_path"], c["delete_path"]):
|
||||||
skipped_ignored += 1
|
skipped_ignored += 1
|
||||||
continue
|
continue
|
||||||
@@ -116,8 +175,11 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
db.commit()
|
db.commit()
|
||||||
db.refresh(dedup_run)
|
db.refresh(dedup_run)
|
||||||
skipped_total = skipped_ignored + skipped_duplicate
|
skipped_total = skipped_ignored + skipped_duplicate
|
||||||
if skipped_total:
|
if skipped_total or auto_applied:
|
||||||
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
|
if skipped_total:
|
||||||
|
dedup_run.kept = (dedup_run.kept or 0) + skipped_total
|
||||||
|
if auto_applied:
|
||||||
|
dedup_run.deleted = auto_applied
|
||||||
db.commit()
|
db.commit()
|
||||||
db.refresh(dedup_run)
|
db.refresh(dedup_run)
|
||||||
return dedup_run
|
return dedup_run
|
||||||
@@ -126,12 +188,23 @@ async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupR
|
|||||||
|
|
||||||
|
|
||||||
async def scan(triggered_by: str = "manual") -> DedupRun | None:
|
async def scan(triggered_by: str = "manual") -> DedupRun | None:
|
||||||
"""Dry-run dedup-library.sh --json, persist every candidate deletion
|
"""Dry-run dedup-library.sh --json (plus --auto-apply-numbered), persist
|
||||||
into a fresh dedup_runs/dedup_candidates pair. Never deletes anything
|
every candidate deletion into a fresh dedup_runs/dedup_candidates pair.
|
||||||
-- the scheduled maintenance:dedup job also only ever calls this (no
|
|
||||||
--apply), matching the false-negative-biased dedup preference; actual
|
Two kinds of match need no human judgment and get deleted unconditionally
|
||||||
deletion only ever happens through confirm_and_apply() below."""
|
by dedup-library.sh itself (see is_numbered_twin/is_format_upgrade there):
|
||||||
return await _run_scan("dedup:scan", _SCRIPT, triggered_by)
|
identical numbered-sibling twins (same dir, same name, same extension,
|
||||||
|
differing only by the ".N" collision suffix), and format upgrades within
|
||||||
|
an EXACT tag match from Pass 2/3 (case-insensitive/normalized already
|
||||||
|
proved same song via identical tags, so a differing extension there is
|
||||||
|
just "better format vs. worse"). Everything else -- including Pass 4's
|
||||||
|
cross-album fuzzy matches, where a wrong auto-delete could take out a
|
||||||
|
genuinely different version (a different album pressing, a DJ-mix edit,
|
||||||
|
etc.) -- stays a dry-run candidate awaiting manual confirm_and_apply()
|
||||||
|
below; that's the false-negative-biased preference for anything with real
|
||||||
|
ambiguity. Auto-applied deletions are reported here already applied,
|
||||||
|
purely for audit visibility."""
|
||||||
|
return await _run_scan("dedup:scan", _SCRIPT, triggered_by, extra_args=("--auto-apply-numbered",))
|
||||||
|
|
||||||
|
|
||||||
async def scan_fuzzy(triggered_by: str = "manual") -> DedupRun | None:
|
async def scan_fuzzy(triggered_by: str = "manual") -> DedupRun | None:
|
||||||
|
|||||||
@@ -155,8 +155,19 @@ MAINTENANCE_JOBS: dict[str, tuple[dict, callable]] = {
|
|||||||
_log_rotation,
|
_log_rotation,
|
||||||
),
|
),
|
||||||
# ==== Weekly (Sunday) ====
|
# ==== Weekly (Sunday) ====
|
||||||
|
# strip_mb_tags runs first here at 01:00 -- not 08:30 like the rest of
|
||||||
|
# this chain -- because its runtime isn't stable: 10m19s on 2026-07-12,
|
||||||
|
# 39.6min on 2026-07-19 (MusicBrainz lookup latency scales with library
|
||||||
|
# size and isn't under our control). At 08:30 that variance repeatedly
|
||||||
|
# starved every job behind it: strip_watermark_art waited the full
|
||||||
|
# SCHEDULED_LOCK_WAIT_SECONDS and gave up every week from 07-12 onward,
|
||||||
|
# and on 07-19 the starvation cascaded all the way through genre:run,
|
||||||
|
# normalize_casing, beets_update_sync, dedup:scan, and both gen_*_playlist
|
||||||
|
# jobs. 01:00 sits in the dead zone before the first playlist sync (05:00)
|
||||||
|
# and well after log_rotation (00:00), so even a run several times slower
|
||||||
|
# than 07-19's is guaranteed to release the lock long before 08:30.
|
||||||
"maintenance:strip_mb_tags": (
|
"maintenance:strip_mb_tags": (
|
||||||
dict(minute=30, hour=8, day_of_week="sun"),
|
dict(minute=0, hour=1, day_of_week="sun"),
|
||||||
_lib("maintenance:strip_mb_tags", "strip-mb-tags.sh"),
|
_lib("maintenance:strip_mb_tags", "strip-mb-tags.sh"),
|
||||||
),
|
),
|
||||||
"maintenance:strip_watermark_art": (
|
"maintenance:strip_watermark_art": (
|
||||||
|
|||||||
+61
-5
@@ -475,11 +475,60 @@ tbody tr:hover { background: rgba(167, 139, 250, 0.055); }
|
|||||||
tbody td.actions-cell { display: flex; gap: 0.5rem; flex-wrap: wrap; }
|
tbody td.actions-cell { display: flex; gap: 0.5rem; flex-wrap: wrap; }
|
||||||
|
|
||||||
.dedup-table { table-layout: fixed; }
|
.dedup-table { table-layout: fixed; }
|
||||||
.dedup-table .path-cell {
|
|
||||||
overflow: hidden;
|
/* Each candidate is its own <tbody> (title row + detail row) so the pair
|
||||||
text-overflow: ellipsis;
|
highlights together on hover, and so :hover can bubble from either row
|
||||||
white-space: nowrap;
|
up to the shared group without JS. That means every detail row is now
|
||||||
max-width: 0; /* forces the cell to respect the colgroup width instead of the content's natural width */
|
a tbody's last-child, which would otherwise strip the separator between
|
||||||
|
one candidate and the next (the generic `tbody tr:last-child` rule above
|
||||||
|
assumed one shared tbody) -- restore it here, and only drop it for the
|
||||||
|
actual last group in the table. */
|
||||||
|
.dedup-table .dedup-row-group:hover td { background: rgba(167, 139, 250, 0.06); }
|
||||||
|
.dedup-table .dedup-detail-row td { border-bottom: 1px solid var(--border); }
|
||||||
|
.dedup-table .dedup-row-group:last-of-type .dedup-detail-row td { border-bottom: none; }
|
||||||
|
|
||||||
|
.dedup-title-row td { padding: 0.75rem 1.1rem 0.2rem; border-bottom: none; }
|
||||||
|
.dedup-detail-row td { padding-top: 0.15rem; }
|
||||||
|
|
||||||
|
.dedup-title {
|
||||||
|
font-family: var(--font-display);
|
||||||
|
font-size: 1.02rem;
|
||||||
|
font-weight: 700;
|
||||||
|
color: var(--fg);
|
||||||
|
line-height: 1.35;
|
||||||
|
overflow-wrap: anywhere;
|
||||||
|
}
|
||||||
|
.dedup-title-alt { font-weight: 400; color: var(--muted); }
|
||||||
|
|
||||||
|
.dedup-context { font-size: 0.8rem; margin-top: 0.15rem; }
|
||||||
|
|
||||||
|
/* A thin colored rail on each side echoes the keep/delete verdict itself --
|
||||||
|
green for the copy that survives, muted-pink for the one on the chopping
|
||||||
|
block -- so which side is which reads before you've even read the words. */
|
||||||
|
.dedup-side {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 0.55rem;
|
||||||
|
padding-left: 0.65rem;
|
||||||
|
border-left: 2px solid transparent;
|
||||||
|
min-height: 1.6rem;
|
||||||
|
}
|
||||||
|
.dedup-side-keep { border-left-color: var(--status-success); }
|
||||||
|
.dedup-side-delete { border-left-color: var(--status-danger); }
|
||||||
|
|
||||||
|
.dedup-side-size { font-size: 0.78rem; white-space: nowrap; flex-shrink: 0; }
|
||||||
|
|
||||||
|
/* The actual filename/path -- always visible, never truncated. Tags can be
|
||||||
|
wrong; this is the ground truth for judging whether two files are really
|
||||||
|
the same recording, so it can't be hover-only the way it was in the first
|
||||||
|
pass of this redesign. */
|
||||||
|
.dedup-side-path {
|
||||||
|
padding-left: 0.65rem;
|
||||||
|
margin-top: 0.2rem;
|
||||||
|
font-size: 0.78rem;
|
||||||
|
color: var(--muted-2);
|
||||||
|
overflow-wrap: anywhere;
|
||||||
|
line-height: 1.4;
|
||||||
}
|
}
|
||||||
|
|
||||||
.empty-row td { padding: 2.5rem 1rem; text-align: center; }
|
.empty-row td { padding: 2.5rem 1rem; text-align: center; }
|
||||||
@@ -509,6 +558,13 @@ tbody td.actions-cell { display: flex; gap: 0.5rem; flex-wrap: wrap; }
|
|||||||
.badge-pulse::before { animation: pulse-dot 1.4s ease-in-out infinite; }
|
.badge-pulse::before { animation: pulse-dot 1.4s ease-in-out infinite; }
|
||||||
@keyframes pulse-dot { 0%, 100% { opacity: 1; } 50% { opacity: 0.35; } }
|
@keyframes pulse-dot { 0%, 100% { opacity: 1; } 50% { opacity: 0.35; } }
|
||||||
|
|
||||||
|
/* The dedup "Caught by" column is only 10.75rem wide -- "Acoustically
|
||||||
|
Similar" at the default badge size overflowed past it and rendered on
|
||||||
|
top of the keep-side format pill. Smaller size/padding/tracking keeps it
|
||||||
|
inside the column instead of shrinking the column itself, which would
|
||||||
|
just crowd the Keep/Delete path columns next to it. */
|
||||||
|
.badge-caughtby { font-size: 0.6rem; padding: 0.24rem 0.55rem; letter-spacing: 0.02em; }
|
||||||
|
|
||||||
/* ---- stacked status bar (playlist track reconciliation) ---- */
|
/* ---- stacked status bar (playlist track reconciliation) ---- */
|
||||||
|
|
||||||
.status-bar {
|
.status-bar {
|
||||||
|
|||||||
@@ -1,5 +1,65 @@
|
|||||||
{% extends "base.html" %}
|
{% extends "base.html" %}
|
||||||
{% block title %}Dedup — alembic{% endblock %}
|
{% block title %}Dedup — alembic{% endblock %}
|
||||||
|
|
||||||
|
{% macro candidate_rows(c, mode) %}
|
||||||
|
<tbody class="dedup-row-group">
|
||||||
|
<tr class="dedup-title-row">
|
||||||
|
<td colspan="{{ 5 if mode == 'pending' else 4 }}">
|
||||||
|
{% if c.same_title %}
|
||||||
|
<div class="dedup-title">{{ c.keep.title }}</div>
|
||||||
|
{% else %}
|
||||||
|
<div class="dedup-title">{{ c.keep.title }} <span class="dedup-title-alt">/ {{ c.delete.title }}</span></div>
|
||||||
|
{% endif %}
|
||||||
|
{% if c.same_artist and c.same_album and c.keep.album %}
|
||||||
|
<div class="dedup-context muted">{{ c.keep.artist }} · {{ c.keep.album }}</div>
|
||||||
|
{% elif c.same_artist %}
|
||||||
|
<div class="dedup-context muted">{{ c.keep.artist }}</div>
|
||||||
|
{% else %}
|
||||||
|
<div class="dedup-context muted">{{ c.keep.artist }} / {{ c.delete.artist }}</div>
|
||||||
|
{% endif %}
|
||||||
|
</td>
|
||||||
|
</tr>
|
||||||
|
<tr class="dedup-detail-row">
|
||||||
|
{% if mode == 'pending' %}
|
||||||
|
<td><input type="checkbox" name="candidate_id" value="{{ c.id }}" form="bulk-delete-form"></td>
|
||||||
|
{% endif %}
|
||||||
|
<td><span class="badge badge-caughtby {{ 'badge-warning' if c.pass_label == 'Acoustically Similar' else 'badge-info' }}" title="{{ c.pass_name }}">{{ c.pass_label }}</span></td>
|
||||||
|
<td>
|
||||||
|
<div class="dedup-side dedup-side-keep">
|
||||||
|
<span class="badge {{ c.keep.badge }}">{{ c.keep.ext or '?' }}</span>
|
||||||
|
{% if c.keep.size_bytes %}<span class="muted dedup-side-size">{{ (c.keep.size_bytes / 1024 / 1024) | round(1) }} MB</span>{% endif %}
|
||||||
|
</div>
|
||||||
|
<div class="dedup-side-path mono muted">{{ c.keep.display }}</div>
|
||||||
|
</td>
|
||||||
|
<td>
|
||||||
|
<div class="dedup-side dedup-side-delete">
|
||||||
|
<span class="badge {{ c.delete.badge }}">{{ c.delete.ext or '?' }}</span>
|
||||||
|
{% if c.delete.size_bytes %}<span class="muted dedup-side-size">{{ (c.delete.size_bytes / 1024 / 1024) | round(1) }} MB</span>{% endif %}
|
||||||
|
</div>
|
||||||
|
<div class="dedup-side-path mono muted">{{ c.delete.display }}</div>
|
||||||
|
</td>
|
||||||
|
<td class="actions-cell">
|
||||||
|
{% if mode == 'pending' %}
|
||||||
|
<form method="post" action="/dedup/confirm" class="inline"
|
||||||
|
onsubmit="return confirm('Permanently delete this file from disk?\n{{ c.delete.display }}')">
|
||||||
|
<input type="hidden" name="candidate_id" value="{{ c.id }}">
|
||||||
|
<button type="submit" class="btn btn-sm btn-danger">Delete</button>
|
||||||
|
</form>
|
||||||
|
<form method="post" action="/dedup/ignore" class="inline">
|
||||||
|
<input type="hidden" name="candidate_id" value="{{ c.id }}">
|
||||||
|
<button type="submit" class="btn btn-sm btn-ghost" title="Keep both copies and never flag this pair again">Keep both</button>
|
||||||
|
</form>
|
||||||
|
{% else %}
|
||||||
|
<form method="post" action="/dedup/unignore" class="inline">
|
||||||
|
<input type="hidden" name="candidate_id" value="{{ c.id }}">
|
||||||
|
<button type="submit" class="btn btn-sm btn-ghost">Un-ignore</button>
|
||||||
|
</form>
|
||||||
|
{% endif %}
|
||||||
|
</td>
|
||||||
|
</tr>
|
||||||
|
</tbody>
|
||||||
|
{% endmacro %}
|
||||||
|
|
||||||
{% block content %}
|
{% block content %}
|
||||||
<div class="page-header">
|
<div class="page-header">
|
||||||
<div>
|
<div>
|
||||||
@@ -18,48 +78,30 @@
|
|||||||
{% endif %}
|
{% endif %}
|
||||||
|
|
||||||
<h2>Pending candidates</h2>
|
<h2>Pending candidates</h2>
|
||||||
<p class="muted" style="margin-top:-0.5rem;">Paths are relative to the library root. Hover a name for the full path.</p>
|
<p class="muted" style="margin-top:-0.5rem;">Song title up top so you can scan straight down the list; the actual file path is always shown underneath each side so you can check they're really the same file before confirming.</p>
|
||||||
<form id="bulk-delete-form" method="post" action="/dedup/confirm"
|
<form id="bulk-delete-form" method="post" action="/dedup/confirm"
|
||||||
onsubmit="return confirm('Permanently delete every selected file from disk? This cannot be undone.')"></form>
|
onsubmit="return confirm('Permanently delete every selected file from disk? This cannot be undone.')"></form>
|
||||||
<div class="table-wrap scroll">
|
<div class="table-wrap scroll">
|
||||||
<table class="dedup-table">
|
<table class="dedup-table">
|
||||||
<colgroup>
|
<colgroup>
|
||||||
<col style="width:2.2rem"><col style="width:9.5rem"><col style="width:28%">
|
<col style="width:2.2rem"><col style="width:10.75rem"><col style="width:37.5%">
|
||||||
<col style="width:28%"><col style="width:5rem"><col style="width:9rem">
|
<col style="width:37.5%"><col style="width:10rem">
|
||||||
</colgroup>
|
</colgroup>
|
||||||
<thead>
|
<thead>
|
||||||
<tr><th></th><th>Caught by</th><th>Keep</th><th>Delete</th><th>Size</th><th>Action</th></tr>
|
<tr><th></th><th>Caught by</th><th>Keep</th><th>Delete</th><th>Action</th></tr>
|
||||||
</thead>
|
</thead>
|
||||||
|
{% for c in candidates %}
|
||||||
|
{{ candidate_rows(c, 'pending') }}
|
||||||
|
{% else %}
|
||||||
<tbody>
|
<tbody>
|
||||||
{% set pass_badge = {"File Naming": "badge-info", "Acoustically Similar": "badge-warning"} %}
|
<tr class="empty-row"><td colspan="5">
|
||||||
{% for c in candidates %}
|
|
||||||
<tr>
|
|
||||||
<td><input type="checkbox" name="candidate_id" value="{{ c.id }}" form="bulk-delete-form"></td>
|
|
||||||
<td><span class="badge {{ pass_badge.get(c.pass_label, 'badge-info') }}" title="{{ c.pass_name }}">{{ c.pass_label }}</span></td>
|
|
||||||
<td class="muted mono path-cell" title="{{ c.keep_path }}">{{ c.keep_display }}</td>
|
|
||||||
<td class="mono path-cell" title="{{ c.delete_path }}">{{ c.delete_display }}</td>
|
|
||||||
<td class="muted">{{ (c.delete_size_bytes / 1024 / 1024) | round(1) if c.delete_size_bytes else '?' }} MB</td>
|
|
||||||
<td class="actions-cell">
|
|
||||||
<form method="post" action="/dedup/confirm" class="inline"
|
|
||||||
onsubmit="return confirm('Permanently delete this file from disk?\n{{ c.delete_display }}')">
|
|
||||||
<input type="hidden" name="candidate_id" value="{{ c.id }}">
|
|
||||||
<button type="submit" class="btn btn-sm btn-danger">Delete</button>
|
|
||||||
</form>
|
|
||||||
<form method="post" action="/dedup/ignore" class="inline">
|
|
||||||
<input type="hidden" name="candidate_id" value="{{ c.id }}">
|
|
||||||
<button type="submit" class="btn btn-sm btn-ghost" title="Keep both copies and never flag this pair again">Keep both</button>
|
|
||||||
</form>
|
|
||||||
</td>
|
|
||||||
</tr>
|
|
||||||
{% else %}
|
|
||||||
<tr class="empty-row"><td colspan="6">
|
|
||||||
<div class="empty-state">
|
<div class="empty-state">
|
||||||
<span class="sparkle" style="position:relative; display:inline-block; margin-bottom:0.5rem;"></span>
|
<span class="sparkle" style="position:relative; display:inline-block; margin-bottom:0.5rem;"></span>
|
||||||
<p class="muted" style="margin:0;">No pending candidates. Run a scan.</p>
|
<p class="muted" style="margin:0;">No pending candidates. Run a scan.</p>
|
||||||
</div>
|
</div>
|
||||||
</td></tr>
|
</td></tr>
|
||||||
{% endfor %}
|
|
||||||
</tbody>
|
</tbody>
|
||||||
|
{% endfor %}
|
||||||
</table>
|
</table>
|
||||||
</div>
|
</div>
|
||||||
{% if candidates %}
|
{% if candidates %}
|
||||||
@@ -72,28 +114,15 @@
|
|||||||
<div class="table-wrap scroll">
|
<div class="table-wrap scroll">
|
||||||
<table class="dedup-table">
|
<table class="dedup-table">
|
||||||
<colgroup>
|
<colgroup>
|
||||||
<col style="width:9.5rem"><col style="width:33%">
|
<col style="width:10.75rem"><col style="width:39.25%">
|
||||||
<col style="width:33%"><col style="width:5rem"><col style="width:7rem">
|
<col style="width:39.25%"><col style="width:7rem">
|
||||||
</colgroup>
|
</colgroup>
|
||||||
<thead>
|
<thead>
|
||||||
<tr><th>Caught by</th><th>Keep</th><th>Delete</th><th>Size</th><th></th></tr>
|
<tr><th>Caught by</th><th>Keep</th><th>Delete</th><th></th></tr>
|
||||||
</thead>
|
</thead>
|
||||||
<tbody>
|
{% for c in ignored %}
|
||||||
{% for c in ignored %}
|
{{ candidate_rows(c, 'ignored') }}
|
||||||
<tr>
|
{% endfor %}
|
||||||
<td><span class="badge badge-muted" title="{{ c.pass_name }}">{{ c.pass_label }}</span></td>
|
|
||||||
<td class="muted mono path-cell" title="{{ c.keep_path }}">{{ c.keep_display }}</td>
|
|
||||||
<td class="muted mono path-cell" title="{{ c.delete_path }}">{{ c.delete_display }}</td>
|
|
||||||
<td class="muted">{{ (c.delete_size_bytes / 1024 / 1024) | round(1) if c.delete_size_bytes else '?' }} MB</td>
|
|
||||||
<td>
|
|
||||||
<form method="post" action="/dedup/unignore" class="inline">
|
|
||||||
<input type="hidden" name="candidate_id" value="{{ c.id }}">
|
|
||||||
<button type="submit" class="btn btn-sm btn-ghost">Un-ignore</button>
|
|
||||||
</form>
|
|
||||||
</td>
|
|
||||||
</tr>
|
|
||||||
{% endfor %}
|
|
||||||
</tbody>
|
|
||||||
</table>
|
</table>
|
||||||
</div>
|
</div>
|
||||||
{% endif %}
|
{% endif %}
|
||||||
|
|||||||
@@ -34,11 +34,26 @@ set -euo pipefail
|
|||||||
APPLY=0
|
APPLY=0
|
||||||
JSON_MODE=0
|
JSON_MODE=0
|
||||||
ONLY_PATHS_FILE=""
|
ONLY_PATHS_FILE=""
|
||||||
|
AUTO_APPLY_NUMBERED=0
|
||||||
while [[ $# -gt 0 ]]; do
|
while [[ $# -gt 0 ]]; do
|
||||||
case "$1" in
|
case "$1" in
|
||||||
--apply) APPLY=1; shift ;;
|
--apply) APPLY=1; shift ;;
|
||||||
--json) JSON_MODE=1; shift ;;
|
--json) JSON_MODE=1; shift ;;
|
||||||
--only-paths) ONLY_PATHS_FILE="$2"; shift 2 ;;
|
--only-paths) ONLY_PATHS_FILE="$2"; shift 2 ;;
|
||||||
|
# Delete two kinds of match unconditionally, independent of
|
||||||
|
# --apply/--only-paths, while everything else stays dry-run/log-only:
|
||||||
|
# - numbered-sibling twins (see is_numbered_twin): same dir, same name,
|
||||||
|
# same extension, just the ".N" collision suffix -- the same download
|
||||||
|
# landing twice.
|
||||||
|
# - format upgrades within an exact tag match (see is_format_upgrade):
|
||||||
|
# Pass 2/3 already proved same song via identical tags; a different
|
||||||
|
# extension there just means "better format vs. worse", never a
|
||||||
|
# different version.
|
||||||
|
# Excludes Pass 4 (cross-album fuzzy) entirely -- matching across
|
||||||
|
# different albums/directories is exactly where a different master or
|
||||||
|
# DJ-mix edit can share tags without being interchangeable, so it always
|
||||||
|
# needs a human. See dedup_review_service.scan().
|
||||||
|
--auto-apply-numbered) AUTO_APPLY_NUMBERED=1; shift ;;
|
||||||
*) shift ;;
|
*) shift ;;
|
||||||
esac
|
esac
|
||||||
done
|
done
|
||||||
@@ -97,6 +112,44 @@ host_path() {
|
|||||||
|
|
||||||
SEP=$'\x1c' # ASCII file separator — safe with any music metadata
|
SEP=$'\x1c' # ASCII file separator — safe with any music metadata
|
||||||
|
|
||||||
|
# True if two paths are the SAME directory, SAME extension, and identical
|
||||||
|
# basenames except for a numeric ".N" infix sldl/beets inserts on a filename
|
||||||
|
# collision (e.g. "Title.flac" vs "Title.1.flac"). This is deliberately
|
||||||
|
# narrower than any tag-based pass: no fuzzy matching, no cross-album
|
||||||
|
# ambiguity, just the literal same download landing twice. Used to
|
||||||
|
# auto-delete regardless of which pass found the pair -- Pass 1 only fires
|
||||||
|
# when the canonical name is a beets ghost; the very common case where BOTH
|
||||||
|
# copies are already tracked in beets (so Pass 1 defers) surfaces instead via
|
||||||
|
# Pass 2/3's tag-based grouping, and still deserves the same free pass.
|
||||||
|
is_numbered_twin() {
|
||||||
|
local a="$1" b="$2"
|
||||||
|
[[ "${a%/*}" == "${b%/*}" ]] || return 1
|
||||||
|
local ba="${a##*/}" bb="${b##*/}"
|
||||||
|
local ext="${ba##*.}"
|
||||||
|
[[ "${ext,,}" == "${bb##*.}" ]] || return 1
|
||||||
|
[[ "$(echo "$ba" | sed -E "s/\.[0-9]+\.${ext}\$/.${ext}/")" \
|
||||||
|
== "$(echo "$bb" | sed -E "s/\.[0-9]+\.${ext}\$/.${ext}/")" ]]
|
||||||
|
}
|
||||||
|
|
||||||
|
# True if this pair is a same-song format upgrade found by an EXACT tag match
|
||||||
|
# (Pass 2 case-insensitive or Pass 3 normalized -- both require identical
|
||||||
|
# albumartist+album+title after case/punctuation folding, so "same song" is
|
||||||
|
# already proven by the pass itself) where the two files simply have
|
||||||
|
# different extensions. rank_file() always ranks FLAC ahead of any other
|
||||||
|
# format regardless of size/dirty-name/etc, so within a tag-exact group a
|
||||||
|
# cross-extension pair unconditionally means "the format-preferred copy vs.
|
||||||
|
# a lesser one" -- no judgment call. Deliberately excludes Pass 4
|
||||||
|
# (cross_album_fuzzy): that pass matches by artist+title-base ACROSS
|
||||||
|
# different albums/directories, which is exactly where a different master,
|
||||||
|
# DJ-mix edit, or reissue can share tags without being an interchangeable
|
||||||
|
# copy -- those still need a human to look at album/context before deleting
|
||||||
|
# either side.
|
||||||
|
is_format_upgrade() {
|
||||||
|
local pass="$1" keep="$2" delete="$3"
|
||||||
|
[[ "$pass" == "case_insensitive" || "$pass" == "normalized" ]] || return 1
|
||||||
|
[[ "${keep##*.}" != "${delete##*.}" ]]
|
||||||
|
}
|
||||||
|
|
||||||
# Rank a file: lower = better. FLAC > MP3 > other (WAV/etc.); clean filename
|
# Rank a file: lower = better. FLAC > MP3 > other (WAV/etc.); clean filename
|
||||||
# > sync-conflict artifact (-DESKTOP-, (1), .1.); extended/club mix > radio
|
# > sync-conflict artifact (-DESKTOP-, (1), .1.); extended/club mix > radio
|
||||||
# edit at same format; then larger wins.
|
# edit at same format; then larger wins.
|
||||||
@@ -177,7 +230,7 @@ process_group() {
|
|||||||
esac
|
esac
|
||||||
|
|
||||||
local first=1
|
local first=1
|
||||||
local keep_path=""
|
local keep_path="" keep_id="" any_auto=0
|
||||||
while IFS="$SEP" read -r _sk id hp; do
|
while IFS="$SEP" read -r _sk id hp; do
|
||||||
[[ -z "$hp" ]] && continue
|
[[ -z "$hp" ]] && continue
|
||||||
local size_bytes; size_bytes=$(stat -c '%s' "$hp" 2>/dev/null || echo 0)
|
local size_bytes; size_bytes=$(stat -c '%s' "$hp" 2>/dev/null || echo 0)
|
||||||
@@ -185,14 +238,31 @@ process_group() {
|
|||||||
if [[ $first -eq 1 ]]; then
|
if [[ $first -eq 1 ]]; then
|
||||||
log " KEEP ($size_h) $hp"
|
log " KEEP ($size_h) $hp"
|
||||||
keep_path="$hp"
|
keep_path="$hp"
|
||||||
|
keep_id="$id"
|
||||||
first=0
|
first=0
|
||||||
else
|
else
|
||||||
log " DELETE ($size_h) $hp"
|
log " DELETE ($size_h) $hp"
|
||||||
json_emit "{\"pass\":\"${pass_name}\",\"keep_path\":\"$(json_escape "$keep_path")\",\"delete_path\":\"$(json_escape "$hp")\",\"delete_size_bytes\":${size_bytes}}"
|
|
||||||
|
# Auto-apply-numbered deletes numbered_sibling matches unconditionally,
|
||||||
|
# bypassing --only-paths (that gate exists only for the manual confirm
|
||||||
|
# flow, which never runs alongside this flag). All other passes stay
|
||||||
|
# dry-run/log-only here regardless.
|
||||||
|
local do_delete=0 auto_applied=0
|
||||||
if [[ $APPLY -eq 1 ]]; then
|
if [[ $APPLY -eq 1 ]]; then
|
||||||
|
do_delete=1
|
||||||
|
elif [[ $AUTO_APPLY_NUMBERED -eq 1 ]] \
|
||||||
|
&& { is_numbered_twin "$keep_path" "$hp" || is_format_upgrade "$pass_name" "$keep_path" "$hp"; }; then
|
||||||
|
do_delete=1
|
||||||
|
auto_applied=1
|
||||||
|
any_auto=1
|
||||||
|
fi
|
||||||
|
local auto_json="false"; [[ $auto_applied -eq 1 ]] && auto_json="true"
|
||||||
|
json_emit "{\"pass\":\"${pass_name}\",\"keep_path\":\"$(json_escape "$keep_path")\",\"delete_path\":\"$(json_escape "$hp")\",\"delete_size_bytes\":${size_bytes},\"auto_applied\":${auto_json}}"
|
||||||
|
|
||||||
|
if [[ $do_delete -eq 1 ]]; then
|
||||||
# With --only-paths given, only delete entries the caller explicitly
|
# With --only-paths given, only delete entries the caller explicitly
|
||||||
# confirmed; without it, delete everything (original behavior).
|
# confirmed; without it, delete everything (original behavior).
|
||||||
if [[ -n "$ONLY_PATHS_FILE" && -z "${ONLY_PATHS[$hp]:-}" ]]; then
|
if [[ $auto_applied -eq 0 && -n "$ONLY_PATHS_FILE" && -z "${ONLY_PATHS[$hp]:-}" ]]; then
|
||||||
log " (skipped — not in --only-paths confirm list)"
|
log " (skipped — not in --only-paths confirm list)"
|
||||||
elif [[ -n "$id" ]]; then
|
elif [[ -n "$id" ]]; then
|
||||||
# In beets — beet remove -d removes from DB + disk. Best-effort: a
|
# In beets — beet remove -d removes from DB + disk. Best-effort: a
|
||||||
@@ -206,6 +276,18 @@ process_group() {
|
|||||||
fi
|
fi
|
||||||
fi
|
fi
|
||||||
done < <(echo "$ranked" | grep -v '^$' | sort -n)
|
done < <(echo "$ranked" | grep -v '^$' | sort -n)
|
||||||
|
|
||||||
|
# An auto-applied numbered-twin deletion can leave the numbered-named file
|
||||||
|
# as the sole survivor (it won on size despite the dirty-filename penalty).
|
||||||
|
# Rename it back to the canonical name so the library doesn't accumulate
|
||||||
|
# ".1."/".2." filenames for tracks that no longer have a duplicate. Only
|
||||||
|
# auto_applied triggers this (never a manual --only-paths confirm, which
|
||||||
|
# may leave the sibling undeleted if the user didn't confirm it -- renaming
|
||||||
|
# then would collide); only when it's actually in beets (a ghost survivor
|
||||||
|
# has nothing to move); only when the name still has the dirty infix.
|
||||||
|
if [[ $any_auto -eq 1 && -n "$keep_id" && "$keep_path" =~ \.[0-9]+\.[A-Za-z0-9]+$ ]]; then
|
||||||
|
beet move "id:${keep_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${keep_id} failed (rename survivor)"
|
||||||
|
fi
|
||||||
return 0 # explicit: the loop's last command status must not leak out under set -e
|
return 0 # explicit: the loop's last command status must not leak out under set -e
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -248,16 +330,20 @@ while IFS="$SEP" read -r numbered_id numbered_cp; do
|
|||||||
# Canonical wins: delete numbered (in beets), keep canonical (ghost)
|
# Canonical wins: delete numbered (in beets), keep canonical (ghost)
|
||||||
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
|
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
|
||||||
"$host_canonical" "${numbered_id}${SEP}${numbered_cp}"
|
"$host_canonical" "${numbered_id}${SEP}${numbered_cp}"
|
||||||
if [[ $APPLY -eq 1 ]]; then
|
if [[ $APPLY -eq 1 || $AUTO_APPLY_NUMBERED -eq 1 ]]; then
|
||||||
# canonical is now a ghost file; import it into beets
|
# canonical is now a ghost file; import it into beets
|
||||||
beet import -q -s "${canonical_cp}" >> "$LOG" 2>&1 || log " WARN: beet import failed for ${canonical_cp}"
|
beet import -q -s "${canonical_cp}" >> "$LOG" 2>&1 || log " WARN: beet import failed for ${canonical_cp}"
|
||||||
fi
|
fi
|
||||||
else
|
else
|
||||||
# Numbered wins (canonical is ghost): rm the ghost, then beet move to rename
|
# Numbered wins (canonical is ghost): rm the ghost, then beet move to rename
|
||||||
# the numbered file to the canonical path (id query — see process_group).
|
# the numbered file to the canonical path (id query — see process_group).
|
||||||
|
# Safe to reach with just AUTO_APPLY_NUMBERED: this loop only ever builds
|
||||||
|
# genuine numbered-sibling pairs (see is_numbered_twin), and process_group
|
||||||
|
# above already deleted the ghost canonical before this runs, so the move
|
||||||
|
# target is guaranteed clear -- no rename collision.
|
||||||
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
|
process_group "P1 numbered-sibling: ${numbered_cp##*/Library/}" \
|
||||||
"${numbered_id}${SEP}${numbered_cp}" "$host_canonical"
|
"${numbered_id}${SEP}${numbered_cp}" "$host_canonical"
|
||||||
if [[ $APPLY -eq 1 ]]; then
|
if [[ $APPLY -eq 1 || $AUTO_APPLY_NUMBERED -eq 1 ]]; then
|
||||||
beet move "id:${numbered_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${numbered_id} failed"
|
beet move "id:${numbered_id}" >> "$LOG" 2>&1 || log " WARN: beet move id:${numbered_id} failed"
|
||||||
fi
|
fi
|
||||||
fi
|
fi
|
||||||
|
|||||||
@@ -458,7 +458,7 @@ def main():
|
|||||||
+ (f" upgrade-from={sorted(upgrade_from)}" if upgrade_from else ""))
|
+ (f" upgrade-from={sorted(upgrade_from)}" if upgrade_from else ""))
|
||||||
print(f"[enrich] {len(flacs)} FLAC files in library\n")
|
print(f"[enrich] {len(flacs)} FLAC files in library\n")
|
||||||
|
|
||||||
looked = found = upgraded = 0
|
looked = found = upgraded = write_failed = 0
|
||||||
by_source = {s: 0 for s in cascade}
|
by_source = {s: 0 for s in cascade}
|
||||||
touched = []
|
touched = []
|
||||||
for p in flacs:
|
for p in flacs:
|
||||||
@@ -519,17 +519,25 @@ def main():
|
|||||||
if set_flac_tag(sp, BUY_URL_TAG, url):
|
if set_flac_tag(sp, BUY_URL_TAG, url):
|
||||||
touched.append(sp)
|
touched.append(sp)
|
||||||
else:
|
else:
|
||||||
print(" ! metaflac write failed")
|
write_failed += 1
|
||||||
|
print(" ! metaflac write failed (permissions? disk full?) -- NOT tagged")
|
||||||
else:
|
else:
|
||||||
touched.append(sp)
|
touched.append(sp)
|
||||||
|
|
||||||
if found % 50 == 0:
|
if found % 50 == 0:
|
||||||
print(f" ...{found} links from {looked} lookups so far", flush=True)
|
print(f" ...{found} links from {looked} lookups so far", flush=True)
|
||||||
|
|
||||||
print(f"\n[enrich] {looked} lookups, {found} {'tagged' if args.apply else 'would-tag'}"
|
# touched is only appended to on an actual successful write (apply mode)
|
||||||
|
# or a would-tag match (dry run) -- len(touched) is what really landed on
|
||||||
|
# disk. `found` counts matches regardless of write outcome, so reporting
|
||||||
|
# `found` here as "tagged" lied about full success on 2026-07-22, when
|
||||||
|
# every write in the run failed (root-owned files under explo/) but the
|
||||||
|
# summary line still read "51 tagged".
|
||||||
|
print(f"\n[enrich] {looked} lookups, {len(touched)} {'tagged' if args.apply else 'would-tag'}"
|
||||||
f" ({', '.join(f'{s}={by_source[s]}' for s in cascade)})"
|
f" ({', '.join(f'{s}={by_source[s]}' for s in cascade)})"
|
||||||
f" no-match={looked - found}"
|
f" no-match={looked - found}"
|
||||||
+ (f" upgraded={upgraded}" if upgrade_from else ""))
|
+ (f" upgraded={upgraded}" if upgrade_from else "")
|
||||||
|
+ (f" WRITE-FAILED={write_failed}" if write_failed else ""))
|
||||||
|
|
||||||
if args.apply and touched and az_key and args.azuracast_base:
|
if args.apply and touched and az_key and args.azuracast_base:
|
||||||
print("[enrich] telling AzuraCast to reprocess touched files...")
|
print("[enrich] telling AzuraCast to reprocess touched files...")
|
||||||
|
|||||||
@@ -57,11 +57,12 @@ OUT_DIR = f"{_MUSIC_DATA_DIR}/playlists-laptop"
|
|||||||
# Absolute Windows path to the laptop's library root.
|
# Absolute Windows path to the laptop's library root.
|
||||||
LIBRARY_LAPTOP_ROOT = r"C:\Music\Library"
|
LIBRARY_LAPTOP_ROOT = r"C:\Music\Library"
|
||||||
|
|
||||||
# Beets stores paths with this prefix (the container-side mount of the
|
# Prefixes beets may display library paths under, tried in order: the real
|
||||||
# library); strip it to get the path relative to the library root. This is
|
# library dir (everything since the 2026-07-08 path rewrite) and the legacy
|
||||||
# still "/music/" through the migration's Stage 0-3 transitional mount;
|
# transitional mount (any stale pre-rewrite entry). Strip whichever matches
|
||||||
# revisit at Stage 4 once beets' directory: becomes MUSIC_DATA_DIR/Library.
|
# to get the path relative to the library root; when neither matches the
|
||||||
BEETS_LIBRARY_PREFIX = "/music/"
|
# track is outside the library and is skipped.
|
||||||
|
BEETS_LIBRARY_PREFIXES = (f"{_MUSIC_DATA_DIR}/Library/", "/music/")
|
||||||
|
|
||||||
M3U_EXT = ".m3u"
|
M3U_EXT = ".m3u"
|
||||||
# Older formats from earlier iterations of this script — swept by reconcile.
|
# Older formats from earlier iterations of this script — swept by reconcile.
|
||||||
@@ -157,9 +158,12 @@ def build_beets_index() -> dict[tuple[str, str], list[tuple[str, str, str, str]]
|
|||||||
if len(parts) != 4:
|
if len(parts) != 4:
|
||||||
continue
|
continue
|
||||||
aa, alb, ti, path = parts
|
aa, alb, ti, path = parts
|
||||||
if not path.startswith(BEETS_LIBRARY_PREFIX):
|
rel = next(
|
||||||
|
(path[len(p):] for p in BEETS_LIBRARY_PREFIXES if path.startswith(p)),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
if rel is None:
|
||||||
continue
|
continue
|
||||||
rel = path[len(BEETS_LIBRARY_PREFIX):]
|
|
||||||
idx.setdefault((_normkey(alb), _normkey(ti)), []).append((aa, alb, ti, rel))
|
idx.setdefault((_normkey(alb), _normkey(ti)), []).append((aa, alb, ti, rel))
|
||||||
return idx
|
return idx
|
||||||
|
|
||||||
|
|||||||
@@ -49,8 +49,10 @@ _ALEMBIC_CONFIG_DIR = os.environ.get("ALEMBIC_CONFIG_DIR", "/config")
|
|||||||
_MUSIC_DATA_DIR = os.environ.get("MUSIC_DATA_DIR", "/data/music")
|
_MUSIC_DATA_DIR = os.environ.get("MUSIC_DATA_DIR", "/data/music")
|
||||||
|
|
||||||
LIBRARY = f"{_MUSIC_DATA_DIR}/Library"
|
LIBRARY = f"{_MUSIC_DATA_DIR}/Library"
|
||||||
# Still "/music" through the migration's Stage 0-3 transitional beets mount;
|
# Legacy prefix from the migration's transitional beets mount. Since the
|
||||||
# revisit at Stage 4 once beets' directory: becomes MUSIC_DATA_DIR/Library.
|
# 2026-07-08 path rewrite beets displays real LIBRARY paths, so this only
|
||||||
|
# matters for accepting pasted /music/... paths as input and as a fallback
|
||||||
|
# beets query form for any stale pre-rewrite DB entry.
|
||||||
CONTAINER_PREFIX = "/music"
|
CONTAINER_PREFIX = "/music"
|
||||||
SPOTIFY_ENV = f"{_ALEMBIC_CONFIG_DIR}/pipeline/_spotify.env"
|
SPOTIFY_ENV = f"{_ALEMBIC_CONFIG_DIR}/pipeline/_spotify.env"
|
||||||
SPOTIFY_GENRE = f"{os.environ.get('PIPELINE_DIR', '/app/pipeline')}/lib/spotify-genre.py"
|
SPOTIFY_GENRE = f"{os.environ.get('PIPELINE_DIR', '/app/pipeline')}/lib/spotify-genre.py"
|
||||||
@@ -316,15 +318,19 @@ def resolve_input_path(arg):
|
|||||||
|
|
||||||
|
|
||||||
def beets_id_for(host_path):
|
def beets_id_for(host_path):
|
||||||
cp = host_to_container(host_path)
|
# Real path first (how beets displays everything since the 2026-07-08
|
||||||
r = subprocess.run(
|
# path rewrite), legacy /music/... form second for any stale entry.
|
||||||
["beet", "ls", "-f", "$id|||$path", f"path:{cp}"],
|
# Querying only the legacy form (the old behavior) matched nothing after
|
||||||
capture_output=True, text=True, check=True)
|
# the rewrite, so retag-from-url silently skipped beet update/move.
|
||||||
for line in r.stdout.splitlines():
|
for cp in (host_path, host_to_container(host_path)):
|
||||||
if "|||" in line:
|
r = subprocess.run(
|
||||||
id_str, p = line.split("|||", 1)
|
["beet", "ls", "-f", "$id|||$path", f"path:{cp}"],
|
||||||
if p == cp:
|
capture_output=True, text=True, check=True)
|
||||||
return int(id_str)
|
for line in r.stdout.splitlines():
|
||||||
|
if "|||" in line:
|
||||||
|
id_str, p = line.split("|||", 1)
|
||||||
|
if p == cp:
|
||||||
|
return int(id_str)
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -425,6 +425,14 @@ PYEOF
|
|||||||
case "$q_code" in
|
case "$q_code" in
|
||||||
200) mark_ok "Qobuz token" "valid (buy-link lookup live)" ;;
|
200) mark_ok "Qobuz token" "valid (buy-link lookup live)" ;;
|
||||||
401) mark_warn "Qobuz token" "EXPIRED — re-export X-User-Auth-Token to ${ALEMBIC_CONFIG_DIR:-/config}/pipeline/qobuz/token" ;;
|
401) mark_warn "Qobuz token" "EXPIRED — re-export X-User-Auth-Token to ${ALEMBIC_CONFIG_DIR:-/config}/pipeline/qobuz/token" ;;
|
||||||
|
# A genuinely expired/bad token gets a JSON 401 from Qobuz's own API.
|
||||||
|
# 403 instead means the request never reached that code at all -- Qobuz's
|
||||||
|
# Akamai edge is rejecting the request outright (seen 2026-07-22: the
|
||||||
|
# exact same request got this "Access Denied"/edgesuite.net block from
|
||||||
|
# alembic's VPN egress IP, but a clean 401 from a non-VPN IP). Re-
|
||||||
|
# exporting the token does nothing for that -- it's the egress IP's
|
||||||
|
# reputation, not the credential. Say so, so it isn't mistaken for 401.
|
||||||
|
403) mark_warn "Qobuz token" "blocked (HTTP 403, likely Akamai/CDN, not the token) — VPN egress IP may be flagged; re-exporting the token won't fix this" ;;
|
||||||
*) mark_warn "Qobuz token" "check failed (HTTP ${q_code:-none})" ;;
|
*) mark_warn "Qobuz token" "check failed (HTTP ${q_code:-none})" ;;
|
||||||
esac
|
esac
|
||||||
else
|
else
|
||||||
|
|||||||
@@ -195,7 +195,18 @@ while IFS= read -r -d '' new_file; do
|
|||||||
existing_id="${match%%$'\x1f'*}"
|
existing_id="${match%%$'\x1f'*}"
|
||||||
existing_container="${match#*$'\x1f'}"
|
existing_container="${match#*$'\x1f'}"
|
||||||
|
|
||||||
existing="${MUSIC_DATA_DIR:-/data/music}/Library${existing_container#/music}"
|
# beets displays real ${MUSIC_DATA_DIR}/Library/... paths since the
|
||||||
|
# 2026-07-08 path rewrite -- use them as-is; only a legacy /music/...
|
||||||
|
# display path still needs the prefix swap. Blindly prepending (the old
|
||||||
|
# behavior) doubled the prefix, so every match logged [stale-db], the MP3
|
||||||
|
# never got replaced, and the leftover-import step at the bottom imported
|
||||||
|
# the staged FLAC as a NEW track -- producing exactly the flac+mp3 twins
|
||||||
|
# this script exists to prevent.
|
||||||
|
if [[ "$existing_container" == /music/* ]]; then
|
||||||
|
existing="${MUSIC_DATA_DIR:-/data/music}/Library${existing_container#/music}"
|
||||||
|
else
|
||||||
|
existing="$existing_container"
|
||||||
|
fi
|
||||||
if [[ ! -f "$existing" ]]; then
|
if [[ ! -f "$existing" ]]; then
|
||||||
echo "[stale-db] beets has $existing_container but file missing" | tee -a "$LOG"
|
echo "[stale-db] beets has $existing_container but file missing" | tee -a "$LOG"
|
||||||
continue
|
continue
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ Usage:
|
|||||||
scrub-watermark-text.py # dry run
|
scrub-watermark-text.py # dry run
|
||||||
scrub-watermark-text.py --apply # actually strip
|
scrub-watermark-text.py --apply # actually strip
|
||||||
"""
|
"""
|
||||||
import sys, re, argparse
|
import os, sys, re, argparse
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from mutagen import File as MFile
|
from mutagen import File as MFile
|
||||||
from mutagen.id3 import ID3, ID3NoHeaderError
|
from mutagen.id3 import ID3, ID3NoHeaderError
|
||||||
|
|||||||
Reference in New Issue
Block a user