Fix dedup apply lock-hang, mawk-broken tag passes; add track delete, fix button/select CSS

confirm_and_apply no longer marks candidates confirmed before their apply
job actually succeeds (a stuck/skipped_lock run was silently vanishing
confirmed deletes without deleting anything), switches the fuzzy pass to
--apply-pairs to avoid a full rescan on every confirm, and adds a job
timeout so a hung subprocess can't hold the global pipeline lock forever.

dedup-library.sh's Pass 2-4 use gawk-only features (gensub, POSIX interval
regex) but were running under mawk (Debian's default awk) in the deployed
image, throwing silent syntax errors on every run -- switched those
invocations to gawk explicitly and added it to the Dockerfile.

Also: per-track delete button on the library list/detail pages, and two
CSS fixes (the primary button's hover gradient was getting clobbered by
the base .btn:hover rule's plain background, and the select dropdown
arrow had no background-size so it rendered oversized).
This commit is contained in:
andrew
2026-07-09 09:12:02 -06:00
parent 9d1eee3337
commit 92e5326437
8 changed files with 182 additions and 41 deletions
+76 -35
View File
@@ -13,6 +13,14 @@ _SCRIPT = "dedup-library.sh"
_FUZZY_SCRIPT = "find-fuzzy-dupes.py"
_FUZZY_PASS = "fuzzy_audio"
# Generous ceiling for any pipeline script invoked here. Without one, a
# stuck subprocess (e.g. a stalled stat() on the NAS mount) holds the single
# global pipeline lock forever -- 2026-07-08 incident: one stuck dedup:apply
# call held the lock for 16+ hours, silently skipping every later confirm
# click, each of which still got marked confirmed=True (see confirm_and_apply)
# and vanished from the pending list without anything actually being deleted.
_JOB_TIMEOUT_SECONDS = 1800
def _script_for_pass(pass_name: str) -> str:
return _FUZZY_SCRIPT if pass_name == _FUZZY_PASS else _SCRIPT
@@ -49,7 +57,7 @@ def _parse_json_lines(output: str) -> list[dict]:
async def _run_scan(job_key: str, script_name: str, triggered_by: str) -> DedupRun:
script = str(settings.pipeline_dir / "lib" / script_name)
job_run, output = await pipeline_runner.run_job_capture(
job_key, [script, "--json"], triggered_by=triggered_by
job_key, [script, "--json"], triggered_by=triggered_by, timeout=_JOB_TIMEOUT_SECONDS
)
candidates = _parse_json_lines(output)
@@ -111,68 +119,101 @@ async def scan_fuzzy(triggered_by: str = "manual") -> DedupRun:
return await _run_scan("dedup:scan_fuzzy", _FUZZY_SCRIPT, triggered_by)
def _write_apply_args(script_name: str, script: str, group: list[DedupCandidate], stamp: int) -> list[str]:
"""Build the --apply invocation for one pass's confirmed group.
find-fuzzy-dupes.py gets --apply-pairs: it applies the exact keep/delete
pairs the user already reviewed without re-scanning the whole library
for acoustic matches (that rescan is what hung for 16+ hours on
2026-07-08, holding the global pipeline lock). dedup-library.sh has no
equivalent fast path -- its tag-based passes are cheap to rescan, so
--apply --only-paths (full rescan, filtered to the confirmed list) is
fine there.
"""
if script_name == _FUZZY_SCRIPT:
pairs_file = settings.logs_dir / f"dedup-confirm-{stamp}-{script_name}.jsonl"
pairs_file.parent.mkdir(parents=True, exist_ok=True)
pairs_file.write_text(
"\n".join(
json.dumps({"keep_path": c.keep_path, "delete_path": c.delete_path}) for c in group
)
+ "\n"
)
return [script, "--apply-pairs", str(pairs_file), "--json"]
confirm_file = settings.logs_dir / f"dedup-confirm-{stamp}-{script_name}.txt"
confirm_file.parent.mkdir(parents=True, exist_ok=True)
confirm_file.write_text("\n".join(c.delete_path for c in group) + "\n")
return [script, "--apply", "--only-paths", str(confirm_file), "--json"]
async def confirm_and_apply(candidate_ids: list[int], confirmed_by: str) -> DedupRun | None:
"""Apply only the confirmed candidate deletions.
Candidates are only marked confirmed=True once we know their apply job
actually ran (status == "success") -- if the shared pipeline lock is
held by something else, or the job times out/crashes, those candidates
are left untouched so they stay visible in the pending list instead of
silently vanishing without being deleted (see 2026-07-08 incident).
Cheap pre-check here: skip anything where delete_path or keep_path no
longer exists (something already changed it since the scan). The real
longer exists (something already changed it since the scan). Further
re-verification of ranking happens for free inside the underlying
script itself: --apply --only-paths re-runs its passes from scratch and
recomputes keep/delete for every group before consulting the only-paths
allowlist, so if a group's ranking flipped since the scan (e.g. the old
keep_path is gone and delete_path is now the last copy), the script's
fresh pass will assign delete_path the KEEP role instead -- it never
reaches a DELETE branch for it, so --only-paths naming it is simply
never consulted. No duplicate ranking logic needed here.
script itself -- see _write_apply_args().
dedup-library.sh (tag-based passes) and find-fuzzy-dupes.py (acoustic
fingerprint pass) are separate scripts, each with its own --apply
--only-paths invocation -- candidates are grouped by pass_name and each
group is applied through the script that actually produced it.
fingerprint pass) are separate scripts -- candidates are grouped by
pass_name and each group is applied through the script that actually
produced it.
"""
db = SessionLocal()
try:
candidates = [db.get(DedupCandidate, cid) for cid in candidate_ids]
candidates = [c for c in candidates if c is not None and not c.applied]
now = time.time()
confirmed_candidates = []
for c in candidates:
if not Path(c.delete_path).exists() or not Path(c.keep_path).exists():
continue
c.confirmed = True
c.confirmed_by = confirmed_by
c.confirmed_at = now
confirmed_candidates.append(c)
db.commit()
if not confirmed_candidates:
candidates = [c for c in candidates if Path(c.delete_path).exists() and Path(c.keep_path).exists()]
if not candidates:
return None
by_script: dict[str, list[DedupCandidate]] = {}
for c in confirmed_candidates:
for c in candidates:
by_script.setdefault(_script_for_pass(c.pass_name), []).append(c)
now = time.time()
finished_at = now
log_paths = []
ran_candidates: list[DedupCandidate] = []
for script_name, group in by_script.items():
confirm_file = settings.logs_dir / f"dedup-confirm-{int(now * 1000)}-{script_name}.txt"
confirm_file.parent.mkdir(parents=True, exist_ok=True)
confirm_file.write_text("\n".join(c.delete_path for c in group) + "\n")
script = str(settings.pipeline_dir / "lib" / script_name)
argv = _write_apply_args(script_name, script, group, int(now * 1000))
job_run, _output = await pipeline_runner.run_job_capture(
"dedup:apply",
[script, "--apply", "--only-paths", str(confirm_file), "--json"],
argv,
triggered_by=f"manual:{confirmed_by}",
timeout=_JOB_TIMEOUT_SECONDS,
)
finished_at = job_run.finished_at
finished_at = job_run.finished_at or finished_at
if job_run.log_path:
log_paths.append(job_run.log_path)
still_there = {c.delete_path for c in confirmed_candidates if Path(c.delete_path).exists()}
if job_run.status != "success":
# Lock contention, crash, or timeout -- nothing in this
# group was actually run. Leave confirmed=False so these
# stay in the pending list for a retry.
continue
for c in group:
c.confirmed = True
c.confirmed_by = confirmed_by
c.confirmed_at = now
ran_candidates.extend(group)
db.commit()
if not ran_candidates:
return None
still_there = {c.delete_path for c in ran_candidates if Path(c.delete_path).exists()}
actually_deleted = 0
for c in confirmed_candidates:
for c in ran_candidates:
if c.delete_path not in still_there:
c.applied = True
actually_deleted += 1
@@ -183,7 +224,7 @@ async def confirm_and_apply(candidate_ids: list[int], confirmed_by: str) -> Dedu
finished_at=finished_at,
mode="apply",
deleted=actually_deleted,
kept=len(confirmed_candidates) - actually_deleted,
kept=len(ran_candidates) - actually_deleted,
log_path=";".join(log_paths) or None,
)
db.add(apply_run)