commit 60fcf2a714547f6760223da0d74de80311c4770f parent 63e8742bf090dedfb49da99cffad941edb9b7fe7 Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st> Date: Sun, 9 Aug 2026 16:34:45 -0400 Give the unattended runner the disk rule it never had Every download a person starts by clicking has had a preflight for a long time. autoRunner -- the one path that dispatches downloads for days with nobody watching -- contained no reference to disk at all. It is the process most likely to fill a disk and the least likely to be observed while it does. The disk is at 98%. The gate is one shared helper because there is one disk. It adds the two things checkDiskSpace was missing: hysteresis, so resuming at the number that stopped us cannot flap on every deleted temp file, and a reason string, so no caller has to invent its own wording. Three modes, and the distinction is the point -- "enforce" for unattended loops (reads and writes the latch), "observe" for UI polls (reads it, so a dashboard cannot show green while the pipeline is held), "manual" for a one-shot someone just clicked (floor only; the hysteresis exists to stop a loop flapping, not to argue with a person who is standing right there). Four more byte-writing holes closed: downloadVideoPipelineAction, whose near-identical sibling already called the lowDiskError defined in the same file; persistKept, per item rather than once, since it writes full containers in a loop; the truncated-audio re-fetch, checked BEFORE it deletes the stub rather than after; and backfillBatch, where diskFloorHit was set and then ignored while the batch kept pulling candidates. Derived sidecars stay ungated on purpose and the code says so -- they are kilobytes, holding them frees nothing and costs days. The dashboard's red "downloads paused" could only ever mean the manual toggle, so a real disk stop said nothing and both-at-once was unreadable. They are two instruments now. The auto-runner spec was checked against a disabled gate before being believed: it goes red with 2 picks dispatched. common 634/634, disk-space.spec 5/5, tsc clean both packages, editor build clean. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Diffstat:
20 files changed, 742 insertions(+), 33 deletions(-)
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -3,6 +3,8 @@ import type { Paths } from "../lib/paths"; import { mapConcurrent } from "../lib/concurrency"; import { getPaths } from "../lib/paths"; import { getSettings } from "../lib/settings"; +import { diskGate } from "../lib/diskSpace"; +import { formatBytes } from "../lib/format"; import { detectPlatform } from "../lib/platform"; import { getWorkerPool } from "../jobs/workerPool"; import { getRegistry } from "../jobs/registry"; @@ -288,6 +290,12 @@ async function runLoop( let metaCache: ChannelMeta[] = []; let metaAt = 0; + // Whether the last next() saw the disk gate closed. next() runs on every + // scheduling tick, so without this the log fills with one identical line per + // tick for as long as the disk is full — which is precisely the situation in + // which the log needs to stay readable. Logged on each transition instead. + let diskIdle = false; + // A graceful stop (the job record is gone after an e2e reset, or the policy was // disabled) is modeled as a soft drain: stop picking, let in-flight finish. // Combined with the job's real drain signal so either one ends the run. @@ -351,6 +359,32 @@ async function runLoop( if (kind === "download" && settings.downloadsPaused) { return null; } + // THE UNATTENDED DISK GATE. This runner is the one path that dispatches + // downloads for days with nobody watching, and until now it was the only + // byte-writing path with no disk check at all — every manual action had a + // preflight, the thing that runs by itself did not. + // + // Idle (return null), don't stop: the gate is a refusal to START more work, + // and it self-heals, so the runner must still be here to notice. Download + // only — transcription writes a transcript.json next to audio it already + // has, so stopping it frees nothing. + if (kind === "download") { + const gate = await diskGate(paths, settings); + if (!gate.ok) { + if (!diskIdle) { + diskIdle = true; + onLog(`Auto-download idle: ${gate.message}.`); + } + return null; + } + if (diskIdle) { + diskIdle = false; + onLog( + `Auto-download resuming: ${formatBytes(gate.freeBytes)} free, ` + + `above the ${formatBytes(gate.resumeBytes)} resume mark.`, + ); + } + } // Refresh the (rarely-changing) channel list/platforms on a TTL. const now = Date.now(); diff --git a/common/controller/backfillBatch.ts b/common/controller/backfillBatch.ts @@ -278,7 +278,18 @@ export async function runBackfillBatch( const action = candidateAction(state, { force: opts.force === true, - allowRedownload, + // MAKE THE FLAG TRUE. `diskFloorHit` used to be set and then ignored — + // the batch kept pulling candidates and kept asking reacquireMediaFor to + // refuse them, one statfs and one media re-check per video, for the rest + // of a run over tens of thousands of videos. + // + // Once the floor is hit, re-acquisition is off for the remainder of the + // run: exactly the behaviour allowRedownload:false already describes, so + // those videos land in `missingInput` and stay visible as work the + // corpus still owes. Note what this does NOT stop — a video whose media + // is already on disk still gets diarized/attributed, because those write + // kilobyte sidecars and holding them frees nothing while losing days. + allowRedownload: allowRedownload && !result.diskFloorHit, }); if (action === "skip") continue; if (action === "fresh") { @@ -340,6 +351,14 @@ export async function runBackfillBatch( signal: runSignal, }); if (reacquired.status === "disk-floor") { + // Latches for the rest of the run — next() reads this and stops + // offering videos that would need a fetch. + if (!result.diskFloorHit) { + log( + `Disk floor reached — no more media will be re-acquired this run. ` + + `Videos needing it are counted as missing-input.`, + ); + } result.diskFloorHit = true; result.missingInput++; return; diff --git a/common/controller/backfillReacquire.ts b/common/controller/backfillReacquire.ts @@ -10,7 +10,7 @@ // // 1. OPT-IN. The caller only reaches here when settings.backfill.allowRedownload // (or an explicit per-run flag) is set. Default off. -// 2. DISK FLOOR, per item. checkDiskSpace against settings.minFreeDiskGB before +// 2. DISK FLOOR, per item. The shared diskGate (floor + hysteresis) before // each fetch, not once at the start: a long run's twentieth video must not // inherit the first one's headroom. Under the floor is a refusal to START, // reported as its own outcome so the batch can stop rather than fail 76,000 @@ -32,8 +32,7 @@ import path from "node:path"; import { readdir, rm } from "node:fs/promises"; import type { Paths } from "../lib/paths"; import { getSettings } from "../lib/settings"; -import { checkDiskSpace } from "../lib/diskSpace"; -import { formatBytes } from "../lib/format"; +import { diskGate } from "../lib/diskSpace"; import { isPermanentlyGone } from "../lib/availability"; import { resolveEffectiveAvailability } from "../lib/availability-server"; import { isDoNotClean } from "../lib/doNotClean-server"; @@ -77,12 +76,9 @@ export async function reacquireMediaFor(opts: { } const settings = getSettings(); - const disk = await checkDiskSpace(opts.paths, settings); - if (!disk.ok) { - log( - `Not re-acquiring ${opts.videoId}: only ${formatBytes(disk.freeBytes)} free, ` + - `below the ${formatBytes(disk.thresholdBytes)} floor.`, - ); + const gate = await diskGate(opts.paths, settings); + if (!gate.ok) { + log(`Not re-acquiring ${opts.videoId}: ${gate.message}.`); return { status: "disk-floor", cleanup: NOTHING_TO_CLEAN }; } diff --git a/common/controller/persistKept.test.ts b/common/controller/persistKept.test.ts @@ -59,6 +59,7 @@ test("persistKept is a no-op when keep-latest is off", async () => { persisted: 0, failed: 0, skippedNoUrl: 0, + skippedLowDisk: 0, }); }); }); diff --git a/common/controller/persistKept.ts b/common/controller/persistKept.ts @@ -2,6 +2,7 @@ import path from "node:path"; import type { Paths } from "../lib/paths"; import type { ChannelConfig } from "../lib/channelConfig"; import { getSettings } from "../lib/settings"; +import { diskGate } from "../lib/diskSpace"; import { resolveCookiePolicy } from "../lib/cookiePolicy"; import { isSavedVideo } from "../lib/savedVideo-server"; import { computeKeptVideoIds } from "./keptVideos"; @@ -31,6 +32,10 @@ export type PersistKeptResult = { failed: number; // No resolvable source URL (no metadata + not in the playlist). skippedNoUrl: number; + // Not attempted because free disk fell under the floor part-way through. A + // refusal to start, not a failure — these videos are still kept, still + // unpersisted, and the next pass picks them up. + skippedLowDisk: number; }; export async function persistKept({ @@ -63,6 +68,7 @@ export async function persistKept({ persisted: 0, failed: 0, skippedNoUrl: 0, + skippedLowDisk: 0, }; if (keptIds.size === 0) { log( @@ -78,13 +84,32 @@ export async function persistKept({ // partway through. computeKeptVideoIds returns an unordered set; sort by id desc // as a stable proxy (YYYYMMDD_-prefixed ids sort by recency, like the window). const ordered = [...keptIds].sort((a, b) => b.localeCompare(a)); + // Set once the disk gate closes mid-pass. Everything after it is counted as + // skipped rather than attempted, so the log says how much was left. + let lowDisk = false; for (const videoId of ordered) { if (signal?.aborted) break; + if (lowDisk) { + result.skippedLowDisk += 1; + continue; + } const videoDir = path.join(dataDir, videoId); if (await isSavedVideo(videoDir)) { result.alreadySaved += 1; continue; } + // PER ITEM, not once at the start. This pass writes full source containers — + // the largest files the app produces — so the twentieth video must not + // inherit the first one's headroom. Same shape as backfillReacquire's + // per-item floor. Checked only for videos that will actually be fetched, so + // an already-saved window costs no syscalls. + const gate = await diskGate(paths, settings); + if (!gate.ok) { + lowDisk = true; + result.skippedLowDisk += 1; + log(` ${videoId}: ${gate.message} — stopping; remaining videos skipped.`); + continue; + } const url = await findVideoSourceUrl(paths, channelSlug, videoId, channelConfig); if (!url) { result.skippedNoUrl += 1; @@ -114,7 +139,11 @@ export async function persistKept({ } log( `Persist kept: ${result.persisted} persisted, ${result.alreadySaved} already saved, ` + - `${result.failed} failed, ${result.skippedNoUrl} skipped (no URL) of ${result.kept} kept.`, + `${result.failed} failed, ${result.skippedNoUrl} skipped (no URL)` + + (result.skippedLowDisk > 0 + ? `, ${result.skippedLowDisk} skipped (low disk)` + : "") + + ` of ${result.kept} kept.`, ); return result; } diff --git a/common/lib/diskSpace.test.ts b/common/lib/diskSpace.test.ts @@ -0,0 +1,108 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { evaluateDiskGate } from "./diskSpace"; + +// The gate's whole job is the pair (floor, hysteresis). evaluateDiskGate is the +// pure core precisely so both can be pinned without a filesystem: the flapping +// bug these tests exist to prevent only shows up as a SEQUENCE of decisions, and +// a test that could only take one measurement could never see it. + +const GB = 1024 ** 3; + +const gate = (freeGB: number, latched: boolean, min = 5, margin = 2) => + evaluateDiskGate({ + freeBytes: freeGB * GB, + minFreeDiskGB: min, + resumeMarginGB: margin, + latched, + }); + +test("above the floor, an open gate stays open", () => { + const g = gate(10, false); + assert.equal(g.ok, true); + assert.equal(g.reason, "ok"); + assert.equal(g.latched, false); + assert.equal(g.message, ""); +}); + +test("below the floor closes the gate and latches it", () => { + const g = gate(3, false); + assert.equal(g.ok, false); + assert.equal(g.reason, "below-floor"); + assert.equal(g.latched, true); + assert.match(g.message, /below the/); +}); + +test("a latched gate does NOT reopen at the floor — this is the flap", () => { + // 6 GB free is above the 5 GB floor, so the un-latched rule would say go. The + // latched rule holds out for 7 GB. Without this, the first resumed download + // drops free space back under 5 and the pipeline oscillates forever. + const open = gate(6, false); + assert.equal(open.ok, true, "an open gate is happy at 6 GB"); + + const held = gate(6, true); + assert.equal(held.ok, false); + assert.equal(held.reason, "below-resume-margin"); + assert.equal(held.latched, true, "stays latched"); + assert.match(held.message, /waiting for/); +}); + +test("a latched gate reopens once the resume mark is cleared", () => { + const g = gate(7, true); + assert.equal(g.ok, true); + assert.equal(g.reason, "ok"); + assert.equal(g.latched, false, "the latch clears itself"); +}); + +test("the full stop-hold-resume sequence does not flap", () => { + let latched = false; + const decisions: boolean[] = []; + // Free space drops under the floor, recovers slowly past it, then past the + // resume mark. The gate must open exactly once, at the end. + for (const freeGB of [4, 4.5, 5, 5.5, 6, 6.9, 7]) { + const g = gate(freeGB, latched); + latched = g.latched; + decisions.push(g.ok); + } + assert.deepEqual(decisions, [ + false, + false, + false, + false, + false, + false, + true, + ]); +}); + +test("a zero margin resumes at the floor (hysteresis opted out)", () => { + const g = gate(5, true, 5, 0); + assert.equal(g.ok, true); + assert.equal(g.resumeBytes, 5 * GB); +}); + +test("minFreeDiskGB 0 disables the gate and clears any latch", () => { + const g = gate(0, true, 0, 2); + assert.equal(g.enabled, false); + assert.equal(g.ok, true); + assert.equal(g.latched, false); +}); + +test("exactly at the floor is enough for an open gate", () => { + // The floor is a minimum to HAVE, not to exceed — checkDiskSpace has always + // used >=, and the gate must not silently tighten it. + assert.equal(gate(5, false).ok, true); +}); + +test("a statfs failure fails open", () => { + // getFreeBytes reports Infinity when statfs throws. A measurement glitch must + // never be the thing that stops a multi-week pipeline. + const g = evaluateDiskGate({ + freeBytes: Number.POSITIVE_INFINITY, + minFreeDiskGB: 5, + resumeMarginGB: 2, + latched: true, + }); + assert.equal(g.ok, true); + assert.equal(g.latched, false); +}); diff --git a/common/lib/diskSpace.ts b/common/lib/diskSpace.ts @@ -1,6 +1,7 @@ import { statfs } from "node:fs/promises"; import type { Paths } from "./paths"; import type { SiteSettings } from "./settings"; +import { formatBytes } from "./format"; const BYTES_PER_GB = 1024 ** 3; @@ -59,3 +60,187 @@ export async function checkDiskSpace( ): Promise<DiskSpaceStatus> { return checkDiskSpaceFor(paths.transcriptsDir, settings); } + +// --------------------------------------------------------------------------- +// THE GATE: one answer to "may byte-writing work start right now?", shared by +// every path that writes media. +// +// checkDiskSpace above answers the instantaneous question. Two things it does +// not have are exactly what an unattended runner asking every few seconds +// needs: +// +// 1. HYSTERESIS. Resuming at the same number we stopped at flaps. The first +// video to restart writing pushes free space back under the floor, we +// stop, a temp file gets cleaned up, we start again — a runner that +// oscillates on every deleted file, logging a stop each time. So a stop +// LATCHES, and clears only once free space has climbed back to +// minFreeDiskGB + resumeMarginGB. The margin is what makes "resumed" mean +// the operator freed something, not that a scratch file went away. +// +// 2. A REASON. "downloads paused" has meant exactly one thing for as long as +// it has existed — the manual toggle. A disk stop that renders identically +// gets read as manual and left alone, which is how a full disk goes +// unnoticed for a week. Every result here carries a reason and a +// preformatted message so no caller has to invent its own wording. +// +// The latch is module-level on purpose: there is one disk, so there is one +// rule, and every caller in the process shares it. It is self-healing — any +// check that sees enough headroom (or a disabled gate) clears it — so no caller +// has to remember to reset it, and a crashed or cancelled job cannot leave the +// pipeline wedged. +// +// DELIBERATELY NOT GATED: derived sidecars (digest.json, diarization.json, +// attribution.json). Those are kilobytes. Stopping them frees nothing and costs +// days of a multi-week sweep, so they run regardless of this gate. Only paths +// that write MEDIA — containers, audio, re-fetched source — ask. + +export type DiskGateReason = + // Enough headroom, or the gate is switched off (minFreeDiskGB === 0). + | "ok" + // Free space is under the floor. A fresh stop. + | "below-floor" + // Latched: back above the floor, but not yet above floor + resume margin, so + // we keep holding rather than flap. + | "below-resume-margin"; + +export type DiskGateStatus = { + // True when byte-writing work may start. Callers idle or refuse when false — + // this is a refusal to START, never a reason to kill work already running. + ok: boolean; + // Whether the gate is configured at all. When false the UI hides it. + enabled: boolean; + freeBytes: number; + // The floor: minFreeDiskGB. + thresholdBytes: number; + // The higher bar a latched gate must clear to reopen: floor + resume margin. + // Equal to thresholdBytes when the margin is 0. + resumeBytes: number; + reason: DiskGateReason; + // Preformatted, log-ready, and empty when ok. Callers append their own + // subject ("Not re-acquiring X: " + message). + message: string; +}; + +// The pure core, so hysteresis is testable without a filesystem or a process. +// `latched` is the previous decision; the returned `latched` is the next one. +export function evaluateDiskGate(opts: { + freeBytes: number; + minFreeDiskGB: number; + resumeMarginGB: number; + latched: boolean; +}): DiskGateStatus & { latched: boolean } { + const { freeBytes, minFreeDiskGB, resumeMarginGB } = opts; + const enabled = minFreeDiskGB > 0; + const thresholdBytes = minFreeDiskGB * BYTES_PER_GB; + const resumeBytes = (minFreeDiskGB + Math.max(0, resumeMarginGB)) * BYTES_PER_GB; + if (!enabled) { + return { + ok: true, + enabled, + freeBytes, + thresholdBytes, + resumeBytes, + reason: "ok", + latched: false, + message: "", + }; + } + // A latched gate is held to the HIGHER bar; an open one only has to clear the + // floor. That asymmetry is the whole of the hysteresis. + const bar = opts.latched ? resumeBytes : thresholdBytes; + if (freeBytes >= bar) { + return { + ok: true, + enabled, + freeBytes, + thresholdBytes, + resumeBytes, + reason: "ok", + latched: false, + message: "", + }; + } + const belowFloor = freeBytes < thresholdBytes; + return { + ok: false, + enabled, + freeBytes, + thresholdBytes, + resumeBytes, + reason: belowFloor ? "below-floor" : "below-resume-margin", + latched: true, + message: belowFloor + ? `only ${formatBytes(freeBytes)} free, below the ` + + `${formatBytes(thresholdBytes)} floor` + : `only ${formatBytes(freeBytes)} free — above the ` + + `${formatBytes(thresholdBytes)} floor but waiting for ` + + `${formatBytes(resumeBytes)} before resuming`, + }; +} + +let gateLatched = false; + +// Which of the three kinds of caller is asking. The distinctions are the whole +// reason this is one shared function rather than three private checks: +// +// "enforce" (default) — an unattended loop deciding whether to take more +// work. Reads AND writes the shared latch, so once it has stopped it is +// held to the higher resume bar. This is what the hysteresis is for. +// +// "observe" — a UI poll. Reads the latch but never writes it, so a dashboard +// refreshing every few seconds reports exactly the state the runners are +// in — INCLUDING "stopped, waiting for the resume mark" — without being the +// thing that changes it. Reporting only the floor here would put the +// dashboard back to showing green while the pipeline sits idle, which is +// the invisible-stop defect this whole gate exists to remove. +// +// "manual" — a one-shot the operator just triggered by clicking something. +// Only the floor applies, and the latch is neither read nor written. The +// hysteresis exists to stop an automatic loop flapping, not to argue with a +// person who is standing right there; equally, one manual download must not +// silently reopen the gate for the unattended runner. +export type DiskGateMode = "enforce" | "observe" | "manual"; + +// Measure the transcripts filesystem and apply the gate. Skips the statfs +// syscall entirely when the gate is disabled, like checkDiskSpaceFor — this +// runs before every item of every byte-writing batch. +export async function diskGate( + paths: Paths, + settings: SiteSettings, + opts?: { mode?: DiskGateMode }, +): Promise<DiskGateStatus> { + const mode = opts?.mode ?? "enforce"; + if (settings.minFreeDiskGB <= 0) { + if (mode === "enforce") gateLatched = false; + return { + ok: true, + enabled: false, + freeBytes: Number.POSITIVE_INFINITY, + thresholdBytes: 0, + resumeBytes: 0, + reason: "ok", + message: "", + }; + } + const freeBytes = await getFreeBytes(paths.transcriptsDir); + const { latched, ...status } = evaluateDiskGate({ + freeBytes, + minFreeDiskGB: settings.minFreeDiskGB, + resumeMarginGB: settings.resumeMarginGB, + latched: mode === "manual" ? false : gateLatched, + }); + if (mode === "enforce") gateLatched = latched; + return status; +} + +// Whether the gate is currently holding. Read-only; for surfaces that want to +// say "stopped by disk" without taking another measurement. +export function isDiskGateLatched(): boolean { + return gateLatched; +} + +// Drop the latch. For tests and for an operator action that means "try again +// now" — nothing in normal operation needs it, since the gate self-heals. +export function resetDiskGate(): void { + gateLatched = false; +} diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -111,6 +111,13 @@ export type SiteSettings = { // prevented from starting and a running batch stops launching new videos // (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts. minFreeDiskGB: number; + // Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see + // before it resumes. Resuming at the same number we stopped at flaps — the + // first restarted download pushes free space back under the floor. This is + // the hysteresis margin, so "resumed" means the operator actually freed + // something rather than a scratch file being cleaned up. 0 disables the + // hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts. + resumeMarginGB: number; // Default number of videos transcribed in parallel when a "Transcribe // missing" / bucket run doesn't specify its own concurrency. The per-run // Concurrency input in the channel UI overrides this for a single run. @@ -566,6 +573,12 @@ export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10; export const MIN_FREE_DISK_GB_DEFAULT = 5; export const MIN_FREE_DISK_GB_MAX = 100000; +// Hysteresis margin for the low-disk gate. 2 GB is deliberately larger than any +// single scratch file the pipeline writes, so cleaning one up cannot by itself +// reopen the gate. +export const RESUME_MARGIN_GB_DEFAULT = 2; +export const RESUME_MARGIN_GB_MAX = 1000; + export const PARALLEL_TRANSCRIPTIONS_MAX = 16; export const PARALLEL_TRANSCRIPTIONS_DEFAULT = 2; @@ -975,6 +988,7 @@ function defaults(): SiteSettings { sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS, downloadFormat: "auto", minFreeDiskGB: MIN_FREE_DISK_GB_DEFAULT, + resumeMarginGB: RESUME_MARGIN_GB_DEFAULT, parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, inlineTranscribeOnFallback: false, skipLiveDownloads: true, @@ -1240,6 +1254,16 @@ export function clampMinFreeDiskGB(value: unknown): number { return n; } +export function clampResumeMarginGB(value: unknown): number { + const n = + typeof value === "number" && Number.isFinite(value) + ? Math.floor(value) + : RESUME_MARGIN_GB_DEFAULT; + if (n < 0) return 0; + if (n > RESUME_MARGIN_GB_MAX) return RESUME_MARGIN_GB_MAX; + return n; +} + export function clampParallelTranscriptions(value: unknown): number { const n = typeof value === "number" && Number.isFinite(value) @@ -1315,6 +1339,7 @@ export function getSettings(): SiteSettings { merged.downloadFormat = "auto"; } merged.minFreeDiskGB = clampMinFreeDiskGB(merged.minFreeDiskGB); + merged.resumeMarginGB = clampResumeMarginGB(merged.resumeMarginGB); merged.parallelTranscriptions = clampParallelTranscriptions( merged.parallelTranscriptions, ); @@ -1522,6 +1547,7 @@ export async function writeSettings(next: SiteSettings): Promise<void> { ? next.downloadFormat : "auto", minFreeDiskGB: clampMinFreeDiskGB(next.minFreeDiskGB), + resumeMarginGB: clampResumeMarginGB(next.resumeMarginGB), parallelTranscriptions: clampParallelTranscriptions( next.parallelTranscriptions, ), diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -10,8 +10,7 @@ import { type ChannelHandling, } from "../lib/channelConfig"; import { getSettings } from "../lib/settings"; -import { checkDiskSpace } from "../lib/diskSpace"; -import { formatBytes } from "../lib/format"; +import { diskGate } from "../lib/diskSpace"; import { detectPlatform } from "../lib/platform"; import { isRealAudioFile } from "../lib/videoStatus"; import { readVttProvenance } from "../lib/subtitleProvenance"; @@ -874,12 +873,11 @@ async function runManagedDownloads( if (opts.drainSignal?.aborted) return; if (firstFailure && abortOnError) return; if (lowDiskStopped) return; - const disk = await checkDiskSpace(opts.paths, settings); - if (!disk.ok) { + const gate = await diskGate(opts.paths, settings); + if (!gate.ok) { lowDiskStopped = true; opts.onLog( - `Stopping batch: ${formatBytes(disk.freeBytes)} free is below ` + - `the ${settings.minFreeDiskGB} GB disk floor. Remaining videos skipped.\n`, + `Stopping batch: ${gate.message}. Remaining videos skipped.\n`, ); return; } diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,9 @@ # Changelog ## [Unreleased] +- **The automatic downloader now stops when the disk is nearly full — it never did before.** Every download you start by clicking something has checked free space for a long time. The one thing that runs unattended, for days, choosing downloads by itself, did not check at all: there was no mention of disk anywhere in it. That is the process most likely to fill a disk and the least likely to have anyone watching while it does. It now consults the same floor as everything else and simply **goes idle** rather than stopping, so it picks up again on its own once space is free — no restart, nothing to remember. Four more paths that write large files were checked and gated the same way: **re-downloading a single video** (its near-identical sibling, "re-download to archive", already checked — this one had been missed), **persisting kept videos**, which now checks *before each video* instead of once at the start, since it writes full video containers in a loop and the twentieth should not be relying on the first one's headroom, **re-downloading truncated audio**, checked before the old file is deleted rather than after, and the **backfill batch**, whose "disk floor reached" flag was being set and then ignored while it kept asking for more work. Derived files — digests, speaker diarization, attribution — are deliberately **not** gated: they are kilobytes, holding them back frees nothing, and it would throw away days of a multi-week run for no gain. +- **Resuming now needs a little more free space than stopping did, so the pipeline can't flap.** If work restarted at exactly the number that stopped it, the first resumed download would push free space back under the floor, stop again, and oscillate — logging a stop each time, which is precisely when the log needs to stay readable. There is a new **Resume margin** setting (default **2 GB**) that a stopped pipeline has to clear before it starts writing again, so "resumed" means you actually freed something rather than a temporary file being tidied away. Set it to 0 for the old behaviour. +- **A disk stop now says it is a disk stop.** The dashboard had one red "downloads paused" marker, and it could only ever mean the manual pause button — during a real disk stop it said nothing at all, and if both were true you would un-pause and watch nothing happen. Manual pause and disk stop are now **two separate readouts** with their own wording, and the disk one shows how much is free, what the floor is, and — when it is holding for the resume margin — the number it is waiting for. A live free-space readout is now always visible whenever the floor is switched on, not only once something has gone wrong. - **Speaker diarization now works on very long recordings instead of dying on them.** The engine compares every detected speaker turn against every other one, so its memory grows with the *square* of how many turns there are — on a dense six-hour stream that reached 10.5 GB and the kernel killed it, after about forty minutes of work, having produced nothing. Six of ten videos over six hours died this way. A second, quieter problem sat underneath it: the audio was decoded into memory **all at once**, which is another 1.8 GB for an eight-hour file before any analysis starts. **Long recordings are now processed in 45-minute windows**, which fixes both at once rather than trading one off against the other — each window has a fraction of the turns, so the comparison shrinks by the *square* of that fraction, and only one window's audio is ever in memory. Measured on the six-hour stream that was being killed: **peak memory 2.0 GB against 10.5 GB, and it completes** — 3,433 speaker turns out of a file that previously produced none. The hard part of windowing is not the windows, it is that the same person appears in several of them and has to be recognized as one speaker rather than nine; each window's speakers are therefore reduced to a voice fingerprint and those are matched across the whole recording, with turns rejoined across the seams. **Short recordings are untouched — the same code path, the same results, byte for byte** — which is checked against an existing file and is what keeps everything already produced reproducible. Windowing is recorded on the file it produces but is deliberately not part of what makes a result look out-of-date, so this does not mark a single existing diarization stale. With this in place the **Max audio hours** limit added above now defaults to off. - **A new scan checks the media already on disk for corruption, and it found some.** The integrity probes have existed since audio-checked downloads landed, but they only ever ran on files as they arrived — nothing had ever asked whether the files sitting here are still readable. They are not all readable. On this corpus the first run took **83 seconds** across 953 media files and turned up: a **4.0 GiB** unreadable video container left behind by a download that never finished (on a disk at 99% full); **three zero-byte audio files** whose videos consequently look "already downloaded" to the pipeline and so are never fetched again, one of them sitting next to a perfectly good file the app would have passed over in its favour; **39 audio files under a name the app does not recognize**, so 1.2 GiB of audio that is present but invisible; and **7 files whose length disagrees with their metadata** by anything up to five hours, which is what a truncated download looks like. The scan is **tiered on purpose**: the default pass only reads each file's header, which costs milliseconds, and the thorough pass — a full decode — is opt-in and runs only on files the cheap pass found reason to doubt, because running it over everything would be hours of processor time. Anything it cannot resolve is reported as *"could not tell"* and never as a fault. **It deletes nothing.** Findings appear on Actionable and per-channel under Diagnostics, each with the evidence for the verdict, and can be marked reviewed to dismiss them; deleting a file remains the separate, deliberate click it already was. - **Very long recordings are now set aside instead of being attempted and killed.** Speaker diarization on a multi-hour stream can exhaust this machine's memory and be killed by the kernel partway through — six of ten videos over six hours died that way, each after about forty minutes of work, producing nothing. The cause is not length as such: the engine holds a comparison of every detected speaker turn against every other, so its memory grows with the *square* of how many turns there are, and turn density varies fortyfold across this corpus (a sparse seven-hour reaction video finished; a dense six-hour stream did not). Length is simply the only predictor available for free, from information already on disk, before spending the forty minutes. So there is now a **Max audio hours** setting (default **4**, `0` turns it off), and a video over it is reported as **deferred**: a third number, shown on the channel's Backfill card and in its summary line, and **never added to the work still to do**. That last part is the point — a limit that quietly shrank the backlog would let a capped corpus report itself as finished. Deferred is not a failure and not "no input"; it is work deliberately not attempted, and raising the limit is all it takes to ask for it. **This is a stopgap and it is meant to be removed:** windowing the engine's work is the real fix, and when that lands the default goes to 0. Note that a backfill already running when this shipped keeps the old behaviour until it is stopped and started again. diff --git a/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts b/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts @@ -13,6 +13,8 @@ import path from "node:path"; import { readdir, rm } from "node:fs/promises"; import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; import { getPaths, type Paths } from "yt-dlp-transcript-common/lib/paths"; +import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { diskGate } from "yt-dlp-transcript-common/lib/diskSpace"; import { isRealAudioFile, isTranscriptVtt, @@ -68,6 +70,18 @@ export async function fixIncompleteTranscriptOne(opts: { const { slug, videoId, config, paths, onLog, signal, tracker } = opts; const videoDir = videoDirOf(paths, slug, videoId); const audioFormat = config.audioFormat ?? "mp3"; + // BEFORE the delete below, not after. This function removes the truncated + // audio so the re-download doesn't see it as already present — refusing on a + // full disk once that audio is gone would leave the video with neither the + // stub nor a replacement. Latching, because the channel-level batch calls this + // in a loop unattended. + const gate = await diskGate(paths, getSettings()); + if (!gate.ok) { + throw new Error( + `Cannot re-download audio for ${videoId}: ${gate.message}. ` + + `Free up space or lower the floor in Settings.`, + ); + } const url = await findVideoSourceUrl(paths, slug, videoId, config); if (!url) { throw new Error( diff --git a/editor/app/channels/[slug]/persistActions.ts b/editor/app/channels/[slug]/persistActions.ts @@ -3,7 +3,7 @@ import { revalidatePath } from "next/cache"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; -import { checkDiskSpace } from "yt-dlp-transcript-common/lib/diskSpace"; +import { diskGate } from "yt-dlp-transcript-common/lib/diskSpace"; import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { downloadQueueKey, @@ -26,7 +26,10 @@ export async function persistKeptAction( const paths = getPaths(); const config = await readChannelConfig(paths, slug); if (!config) return { ok: false, error: `Channel "${slug}" not found` }; - const disk = await checkDiskSpace(paths, getSettings()); + // The operator clicked this: only the floor applies and the shared hysteresis + // latch is left alone (see diskGate). This is a preflight only — persistKept + // re-checks per video, because it writes full containers in a loop. + const disk = await diskGate(paths, getSettings(), { mode: "manual" }); if (!disk.ok) { return { ok: false, diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -26,7 +26,7 @@ import { mergeRosterFile } from "yt-dlp-transcript-common/controller/rosterStore import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy"; -import { checkDiskSpace } from "yt-dlp-transcript-common/lib/diskSpace"; +import { diskGate } from "yt-dlp-transcript-common/lib/diskSpace"; import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { runManagedFunction, @@ -43,7 +43,9 @@ import type { Paths } from "yt-dlp-transcript-common/lib/paths"; async function lowDiskError( paths: Paths, ): Promise<{ ok: false; error: string } | null> { - const disk = await checkDiskSpace(paths, getSettings()); + // The operator clicked this: only the floor applies and the shared hysteresis + // latch is left alone (see diskGate). runYtdlp re-checks per video mid-batch. + const disk = await diskGate(paths, getSettings(), { mode: "manual" }); if (disk.ok) return null; return { ok: false, diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -15,7 +15,7 @@ import { type DownloadFormatPreset, } from "yt-dlp-transcript-common/ytdlp/downloadFormat"; import { getPaths, type Paths } from "yt-dlp-transcript-common/lib/paths"; -import { checkDiskSpace } from "yt-dlp-transcript-common/lib/diskSpace"; +import { diskGate } from "yt-dlp-transcript-common/lib/diskSpace"; import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { audioFilesToRemove, @@ -140,6 +140,10 @@ export async function downloadVideoPipelineAction( const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); + // This action fetches a full container. Its sibling redownloadToArchiveAction + // has always preflighted the disk; this one never did. + const lowDisk = await lowDiskError(paths); + if (lowDisk) return lowDisk; const url = await findVideoSourceUrl(paths, slug, videoId, r.config); if (!url) { return { @@ -253,17 +257,19 @@ export async function redownloadToArchiveAction( }); } -// Preflight disk-space gate shared with the per-video download actions. +// Preflight disk-space gate shared by every per-video action that can pull +// bytes down. `latch: false` — the operator clicked this, so only the floor +// applies and the shared hysteresis latch is left alone (see diskGate). async function lowDiskError( paths: Paths, ): Promise<{ ok: false; error: string } | null> { - const disk = await checkDiskSpace(paths, getSettings()); - if (disk.ok) return null; + const gate = await diskGate(paths, getSettings(), { mode: "manual" }); + if (gate.ok) return null; return { ok: false, error: - `Low disk space: ${formatBytes(disk.freeBytes)} free, ` + - `${formatBytes(disk.thresholdBytes)} required. Free up space or ` + + `Low disk space: ${formatBytes(gate.freeBytes)} free, ` + + `${formatBytes(gate.thresholdBytes)} required. Free up space or ` + `lower the floor in Settings.`, }; } @@ -288,6 +294,17 @@ export async function whisperVideoAction( const entries = await readdir(videoDir).catch(() => [] as string[]); const hasAudio = entries.some(isRealAudioFile); if (!hasAudio) { + // Only this branch writes bytes. Transcribing audio that is already on + // disk produces a transcript.json measured in kilobytes, so a low disk + // is no reason to refuse it — the gate belongs on the fetch, not on the + // whole action. + const gate = await diskGate(paths, getSettings(), { mode: "manual" }); + if (!gate.ok) { + throw new Error( + `Cannot download audio for ${videoId}: ${gate.message}. ` + + `Free up space or lower the floor in Settings.`, + ); + } const url = await findVideoSourceUrl(paths, slug, videoId, r.config); if (!url) { throw new Error( diff --git a/editor/app/components/dashboard/PipelineBand.tsx b/editor/app/components/dashboard/PipelineBand.tsx @@ -12,6 +12,7 @@ import { DigestSweepControls } from "../../jobs/components/DigestSweepControls"; import { BackfillSweepControls } from "../../jobs/components/BackfillSweepControls"; import { syncAllChannelsAction, type SyncAllResult } from "../../channels/actions"; import { fmtTime } from "../../widget/lib/relativeTime"; +import { formatBytes } from "yt-dlp-transcript-common/lib/format"; // The hero: a single live instrument readout of the whole pipeline — running/ // queued jobs, worker-pool occupancy + pause state, sync heartbeat + scheduler — @@ -39,6 +40,14 @@ export function PipelineBand({ const busy = workerList.filter((w) => w.busy).length; const paused = workers?.paused ?? false; const downloadsPaused = workers?.downloadsPaused ?? false; + // THE TWO REASONS DOWNLOADS STOP, AND THEY ARE NOT THE SAME THING. + // `downloadsPaused` is the manual toggle; `disk.low` is the gate deciding on + // its own. This band used to render one red "downloads paused" that could + // only ever mean the toggle — so a real disk stop said nothing at all, and if + // both were true the operator would un-pause and watch nothing happen. They + // now get separate instruments with their own wording. + const disk = jobs?.disk ?? null; + const diskLow = disk?.low ?? false; const digest = sync?.digest ?? null; // Deliberately not rounded up. At 0.13% a "1%" would be a lie of the kind // that makes an 80-day backfill look nearly begun. @@ -119,7 +128,49 @@ export function PipelineBand({ {downloadsPaused && ( <Instrument dotClass="bg-destructive" - label={<span className="text-destructive">downloads paused</span>} + label={ + <span + aria-label="downloads paused manually" + className="text-destructive" + > + downloads paused + <span className="text-muted-foreground"> · manual</span> + </span> + } + /> + )} + {disk?.enabled && ( + <Instrument + dotClass={diskLow ? "bg-destructive" : "bg-success/40"} + label={ + diskLow ? ( + <span + aria-label="disk low" + className="text-destructive" + title={disk.message} + > + downloads stopped · disk{" "} + <span className="font-medium"> + {formatBytes(disk.freeBytes)} + </span>{" "} + free, floor {formatBytes(disk.thresholdBytes)} + {disk.reason === "below-resume-margin" && ( + <span className="text-muted-foreground"> + {" "} + · resumes at {formatBytes(disk.resumeBytes)} + </span> + )} + </span> + ) : ( + <> + disk{" "} + <span className="font-medium"> + {formatBytes(disk.freeBytes)} + </span>{" "} + free + </> + ) + } /> )} {backfill?.anyKind && ( diff --git a/editor/app/jobs/active/buildActiveJobs.ts b/editor/app/jobs/active/buildActiveJobs.ts @@ -10,7 +10,10 @@ import { } from "yt-dlp-transcript-common/controller/channels"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; -import { checkDiskSpace } from "yt-dlp-transcript-common/lib/diskSpace"; +import { + diskGate, + type DiskGateReason, +} from "yt-dlp-transcript-common/lib/diskSpace"; import type { RunningJobsListItem } from "../components/RunningJobsList"; export type DiskStatusView = { @@ -22,6 +25,14 @@ export type DiskStatusView = { // True when the gate is enabled and free space is at/below the floor — i.e. // downloads are currently being blocked. low: boolean; + // The bar a stopped pipeline has to clear to resume (floor + resume margin). + // Equal to thresholdBytes when the margin is 0. + resumeBytes: number; + // Why, in one word, so a surface can say "disk" rather than leaving the + // operator to read a red "downloads paused" as the manual toggle. + reason: DiskGateReason; + // Log-ready explanation; empty when not low. + message: string; }; export type ActiveJobsPayload = { @@ -206,12 +217,19 @@ export async function buildActiveJobsPayload(): Promise<ActiveJobsPayload> { displayName: channelStats.get(slug)?.config.name ?? slug, })); - const diskStatus = await checkDiskSpace(paths, getSettings()); + // "observe" — a UI poll, several times a minute. It reports the state the + // runners are actually in (latch included, so a pipeline held for the resume + // margin reads as stopped rather than green) without being the thing that + // moves that latch. + const diskStatus = await diskGate(paths, getSettings(), { mode: "observe" }); const disk: DiskStatusView = { enabled: diskStatus.enabled, freeBytes: diskStatus.freeBytes, thresholdBytes: diskStatus.thresholdBytes, + resumeBytes: diskStatus.resumeBytes, low: diskStatus.enabled && !diskStatus.ok, + reason: diskStatus.reason, + message: diskStatus.message, }; return { jobs, channels, disk }; diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -10,6 +10,8 @@ import { isReportDebouncePreset, getSettings, MIN_FREE_DISK_GB_MAX, + RESUME_MARGIN_GB_DEFAULT, + RESUME_MARGIN_GB_MAX, normalizeSocialSvg, parseSocialLinks, PARALLEL_TRANSCRIPTIONS_DEFAULT, @@ -61,6 +63,7 @@ export async function saveSettingsAction( formData.get("autoRefreshIntervalSeconds") ?? "", ).trim(); const minFreeDiskRaw = String(formData.get("minFreeDiskGB") ?? "").trim(); + const resumeMarginRaw = String(formData.get("resumeMarginGB") ?? "").trim(); const inlineTranscribeOnFallback = formData.get("inlineTranscribeOnFallback") === "on"; const skipLiveDownloads = formData.get("skipLiveDownloads") === "on"; @@ -153,6 +156,23 @@ export async function saveSettingsAction( }; } + // A settings.json (or a form) predating the field sends nothing; fall back to + // the default rather than rejecting the whole save over a field the operator + // never saw. + const resumeMarginParsed = + resumeMarginRaw === "" + ? RESUME_MARGIN_GB_DEFAULT + : Number.parseInt(resumeMarginRaw, 10); + if (!Number.isFinite(resumeMarginParsed)) { + return { ok: false, error: "resumeMarginGB must be a number" }; + } + if (resumeMarginParsed < 0 || resumeMarginParsed > RESUME_MARGIN_GB_MAX) { + return { + ok: false, + error: `resumeMarginGB must be between 0 and ${RESUME_MARGIN_GB_MAX}`, + }; + } + // Sync-scheduler block. Values are clamped by sanitizeSyncScheduler inside // writeSettings, so we only coerce here (NaN/blank fall back to defaults). const intOrNaN = (key: string): number => @@ -422,6 +442,7 @@ export async function saveSettingsAction( sleepBetweenDownloadsSeconds: sleepParsed, downloadFormat, minFreeDiskGB: minFreeDiskParsed, + resumeMarginGB: resumeMarginParsed, parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, inlineTranscribeOnFallback, skipLiveDownloads, diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -165,6 +165,13 @@ export function SettingsForm({ initial, apps, digestApps }: Props) { hint="Downloads are prevented from starting, and a running batch stops launching new videos, when free space on the transcripts directory falls below this floor. Default 5 GB. Set to 0 to disable the gate." /> <Field + label="Resume margin (GB)" + name="resumeMarginGB" + defaultValue={String(initial.resumeMarginGB)} + type="number" + hint="Extra headroom above the floor that a disk-stopped pipeline must see before it starts writing again. Without it the first resumed download drops free space back under the floor and the pipeline flaps. Default 2 GB. Set to 0 to resume at the floor." + /> + <Field label="Auto-refresh interval (seconds)" name="autoRefreshIntervalSeconds" defaultValue={String(initial.autoRefreshIntervalSeconds)} diff --git a/editor/app/widget/components/MonitorWidget.tsx b/editor/app/widget/components/MonitorWidget.tsx @@ -374,6 +374,11 @@ function DiskStrip({ disk }: { disk: DiskStatusView }) { <> {" "} — below {formatBytes(disk.thresholdBytes)} floor; downloads paused + {/* Says which of the two bars is being waited on. Without it a + stop that is already above the floor reads as a contradiction. */} + {disk.reason === "below-resume-margin" && ( + <> (resumes at {formatBytes(disk.resumeBytes)})</> + )} </> )} </span> diff --git a/editor/e2e/disk-space.spec.ts b/editor/e2e/disk-space.spec.ts @@ -1,15 +1,29 @@ +import { mkdir, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; -import { pathExists, resetData, writeSettings, generateReport } from "./helpers"; +import { + pathExists, + resetData, + resolvePath, + writeSettings, + generateReport, +} from "./helpers"; +import { baseUrl } from "./baseUrl"; // The low-disk gate (common/lib/diskSpace.ts) is driven by the // minFreeDiskGB setting. e2e can't force the filesystem to fill up, so these // tests pin the threshold absurdly high (so the real test disk always reads as // "below floor") to exercise the blocked path, and 0 to exercise the disabled // path. The per-video stop gate in runManagedDownloads shares the same -// checkDiskSpace as the preflight — with a static threshold the preflight +// diskGate as the preflight — with a static threshold the preflight // always intercepts first, so a mid-batch stop only happens on a real-time // drop, which isn't reproducible here; the preflight + disabled paths cover the // shared logic. +// +// The hysteresis rule itself (stop at the floor, resume only at floor + margin) +// is pinned as a SEQUENCE of decisions in common/lib/diskSpace.test.ts, where a +// fake free-space number can be fed in. What can only be checked here is that +// the unattended auto-download runner consults the gate at all — it did not, +// which was the single largest hole in Phase A, and it is asserted below. const HUGE_FLOOR_GB = 100000; // == MIN_FREE_DISK_GB_MAX; far above any real disk @@ -62,6 +76,149 @@ test("lets downloads proceed when the gate is disabled (0)", async ({ ).toBe(true); }); +// A YouTube channel with undownloaded videos, enough for the auto-download +// runner to have something to pick. Mirrors auto-queue.spec's fixture. +async function makeDownloadChannel(slug: string, ids: string[]) { + const root = resolvePath(`test-transcripts/channels/${slug}`); + await mkdir(`${root}/data`, { recursive: true }); + await writeFile( + `${root}/config.json`, + JSON.stringify({ + handling: "youtube", + name: slug, + url: `https://www.youtube.com/@${slug}/videos`, + }), + ); + await writeFile( + `${root}/playlist`, + ids.map((id) => `https://www.youtube.com/watch?v=${id}`).join("\n") + "\n", + ); + await writeFile( + `${root}/snapshot.json`, + JSON.stringify({ + generatedAt: "2026-06-01T00:00:00.000Z", + totals: { videos: ids.length, transcribed: 0, downloaded: 0 }, + buckets: { + noTranscript: [], + downloadedNoTranscript: [], + untranscoded: [], + multipleAudioFormats: [], + transcribedWithAudio: [], + untranscribable: [], + noMetadata: [], + failedListed: [], + missingFromArchive: [], + duplicateDirs: [], + partialDownloads: [], + corruptSource: [], + nonStandardVtt: [], + skippedByFilter: [], + }, + undownloadedIds: ids, + }), + ); +} + +test("the unattended auto-download runner idles below the floor", async ({ + request, +}) => { + // THE REGRESSION. autoRunner.ts is the one path that dispatches downloads for + // days with nobody watching, and before Phase A it contained no reference to + // disk at all — every manual action had a preflight, the thing that runs by + // itself did not. Without the gate this test dispatches and yt-dlp starts. + await resetData(null); + await makeDownloadChannel("alpha", ["d1", "d2"]); + await writeSettings({ + adminTitle: "Test Admin", + maxTranscriptPageBytes: 8388608, + sleepBetweenDownloadsSeconds: 0, + minFreeDiskGB: HUGE_FLOOR_GB, + autoQueue: { + transcription: {}, + download: { + enabled: true, + maxWorkers: null, + root: { + id: "root", + mode: "strict", + children: [ + { id: "leaf-alpha", match: { type: "channel", value: "alpha" } }, + ], + }, + }, + }, + }); + + const start = await request.post(`${baseUrl}/api/auto-queue/control`, { + data: { kind: "download", action: "start" }, + }); + expect(start.ok()).toBeTruthy(); + + try { + // Give the runner real time to tick. It must stay running (the gate IDLES, + // it does not stop the runner — otherwise freeing space would need a manual + // restart) while never handing out a single video. + await page_waitMs(6000); + const status = await request.get(`${baseUrl}/api/auto-queue/status`); + const body = (await status.json()) as { + download: { runner: { running: boolean }; picks: unknown[] }; + }; + expect(body.download.runner.running).toBe(true); + expect(body.download.picks).toHaveLength(0); + // And nothing landed on disk. + expect( + await pathExists("test-transcripts/channels/alpha/data/d1"), + ).toBe(false); + } finally { + await request.post(`${baseUrl}/api/auto-queue/control`, { + data: { kind: "download", action: "stop" }, + }); + } +}); + +// A plain sleep. The assertion here is an ABSENCE (no picks), so there is +// nothing to poll toward — the test has to give the runner room to misbehave. +function page_waitMs(ms: number): Promise<void> { + return new Promise((r) => setTimeout(r, ms)); +} + +test("the dashboard tells a disk stop apart from the manual pause", async ({ + page, +}) => { + // The defect: PipelineBand rendered one red "downloads paused" that read + // settings.downloadsPaused — the MANUAL toggle. A real disk stop said + // nothing, and when both were true the operator would un-pause and watch + // nothing happen. + await resetData("test-pipeline"); + + // Two distinct labels, because they are two distinct facts. (getByLabel + // matches by substring, so both are exact-matched — "Pipeline" alone pulls in + // five unrelated nodes.) + const diskStop = page.getByLabel("disk low", { exact: true }); + const manualPause = page.getByLabel("downloads paused manually", { + exact: true, + }); + + // Disk stop only. The band must name the disk, and must NOT claim a manual + // pause — reading a disk stop as the manual toggle is the original defect. + await writeSettings({ minFreeDiskGB: HUGE_FLOOR_GB, downloadsPaused: false }); + await page.goto("/"); + await expect(diskStop).toContainText("disk", { timeout: 15_000 }); + await expect(manualPause).toHaveCount(0); + + // Both at once: two instruments, so un-pausing is visibly not enough. + await writeSettings({ minFreeDiskGB: HUGE_FLOOR_GB, downloadsPaused: true }); + await page.reload(); + await expect(manualPause).toBeVisible({ timeout: 15_000 }); + await expect(diskStop).toBeVisible(); + + // Gate off, still manually paused: only the manual instrument remains. + await writeSettings({ minFreeDiskGB: 0, downloadsPaused: true }); + await page.reload(); + await expect(manualPause).toBeVisible({ timeout: 15_000 }); + await expect(diskStop).toHaveCount(0); +}); + test("active-jobs API and monitor widget report low disk", async ({ page }) => { await resetData("test-pipeline"); @@ -70,10 +227,25 @@ test("active-jobs API and monitor widget report low disk", async ({ page }) => { const lowRes = await page.request.get("/api/jobs/active"); expect(lowRes.ok()).toBe(true); const lowPayload = (await lowRes.json()) as { - disk: { enabled: boolean; low: boolean }; + disk: { + enabled: boolean; + low: boolean; + reason: string; + message: string; + resumeBytes: number; + thresholdBytes: number; + }; }; expect(lowPayload.disk.enabled).toBe(true); expect(lowPayload.disk.low).toBe(true); + // The payload says WHY, so a surface never has to guess between the manual + // toggle and the gate. + expect(lowPayload.disk.reason).toBe("below-floor"); + expect(lowPayload.disk.message).toContain("below the"); + // Resume needs strictly more headroom than the floor (default margin 2 GB). + expect(lowPayload.disk.resumeBytes).toBeGreaterThan( + lowPayload.disk.thresholdBytes, + ); // The monitor widget surfaces the warning. await page.goto("/widget");