commit 74050956c077b8c02c3a1841de168790cf3f82e9
parent 30ff28db2d184133fafa743bbeb35f0bb11af533
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 29 Sep 2026 23:36:45 -0400
common, editor: review M2, L4, L9 — the health pass is armed above the idle gate; an auto-pause says whether the drive is not there or not answering
- M2: `startStorageHealthWatch` arms the 15 s health pass on its own, and
instrumentation arms it ABOVE the idle gate, beside the storage boot probe:
it is in memory and writes nothing, and without it an idle boot has no
registered locations and a stall the watchdog marks is never cleared. The
five-minute pass (`startStorageWatch`), which may write an auto-pause, stays
below the gate and arms nothing else.
- L9: the five-minute pass counts a channel whose inspect says `stalled` as
down, on a configured location or not.
- L4: the pause record carries `cause: "not-there" | "not-answering"`
(optional; the sanitizer keeps it; a record without one reads as not there,
which is what every earlier record meant). autoPauseReasonOf says "on a
drive that is not answering … when the drive answers again" for the second.
SETTINGS.md regenerated for the new field.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
5 files changed, 178 insertions(+), 50 deletions(-)
diff --git a/SETTINGS.md b/SETTINGS.md
@@ -449,6 +449,7 @@ Per entry — each entry spells its own values.
| `reason` | One reason today. A union so a second one has somewhere to go, and so a surface can say WHICH machine decided rather than "automatic". |
| `since` | ISO, for "auto-paused — media unreachable since <date>". |
| `previousTier` | The base tier the channel had before the machine paused it; what a restore puts back. Never `paused` (that would restore to paused — a no-op dressed as a restore). |
+| `cause` | `not-there` (the drive is unmounted or unplugged) or `not-answering` (it is there and does not answer: a stalled disk). Only what the words on /review, the rack and the channel page say. Optional: a record written before it existed reads as `not-there`. |
Default:
diff --git a/common/controller/storageWatch.test.ts b/common/controller/storageWatch.test.ts
@@ -6,6 +6,7 @@ import path from "node:path";
import type { Paths } from "../lib/paths";
import type { SiteSettings } from "../lib/settings";
import {
+ autoPauseReasonOf,
compileLanes,
sanitizeChannelPriority,
} from "../lib/channelPriority";
@@ -15,7 +16,9 @@ import {
resetStorageWatchSuspicion,
runStorageHealthPass,
runStorageWatchPass,
+ startStorageHealthWatch,
startStorageWatch,
+ stopStorageHealthWatch,
stopStorageWatch,
} from "./storageWatch";
import { inspectChannelMedia } from "../lib/channelMedia";
@@ -470,15 +473,12 @@ test("a probe that throws is 'could not ask': ok, never stalled", async () => {
});
});
-test("arming the watch runs a health pass at once, and stopping it stops both timers", async () => {
+test("the health pass is armed on its own, runs once at once, and stops; the five-minute watch arms no health pass", async () => {
await withTmp(async (h) => {
let asked = 0;
- const armed = startStorageWatch({
- paths: h.paths,
+ const armed = startStorageHealthWatch({
io: h.io,
- bins: h.paths,
- write: false,
- healthProbe: async () => {
+ probe: async () => {
asked += 1;
return "stalled";
},
@@ -487,17 +487,22 @@ test("arming the watch runs a health pass at once, and stopping it stops both ti
try {
assert.equal(armed, true);
// Armed once per process.
- assert.equal(startStorageWatch({ io: h.io }), false);
+ assert.equal(startStorageHealthWatch({ io: h.io }), false);
for (let i = 0; i < 50 && locationHealth("cold")?.state !== "stalled"; i++) {
await new Promise((r) => setTimeout(r, 10));
}
assert.equal(asked, 1);
assert.equal(locationHealth("cold")?.state, "stalled");
} finally {
- stopStorageWatch();
+ stopStorageHealthWatch();
}
- assert.equal(startStorageWatch({ io: h.io, healthProbe: async () => "ok" }), true);
+ assert.equal(startStorageHealthWatch({ io: h.io, probe: async () => "ok" }), true);
+ stopStorageHealthWatch();
+ // The watch (below the idle gate) runs nothing at arm time and asks no drive.
+ resetStorageHealth();
+ assert.equal(startStorageWatch({ paths: h.paths, io: h.io, bins: h.paths, write: false }), true);
stopStorageWatch();
+ assert.equal(locationHealth("cold"), undefined);
});
});
@@ -545,3 +550,42 @@ test("the pass registers every location, records a verdict's detector, and a ver
assert.match(lines.join("\n"), /"cold": drive not answering — its disk \(sdz1\)/);
});
});
+
+test("a counters verdict records its device (for the watchdog); a stat verdict forgets it", async () => {
+ await withTmp(async (h) => {
+ const verdicts = [
+ { answer: null, detector: "counters" as const, device: "sdz1" },
+ { answer: "ok" as const, detector: "stat" as const },
+ ];
+ let i = 0;
+ const probe = async () => verdicts[i++];
+ await runStorageHealthPass({ io: h.io, probe });
+ assert.equal(locationHealth("cold")?.device, "sdz1");
+ await runStorageHealthPass({ io: h.io, probe });
+ assert.equal(locationHealth("cold")?.device, undefined);
+ assert.equal(locationHealth("cold")?.detector, "stat");
+ });
+});
+
+test("a stall auto-pauses after two passes, and says the drive is not answering (not that it is not there)", async () => {
+ await withTmp(async (h) => {
+ await seedRelocated(h, "slow", { targetExists: true });
+ await runStorageHealthPass({ io: h.io, probe: async () => "stalled" });
+ await twoPasses(h);
+ const entry = h.io.read().channelPriority.channels.slow;
+ assert.equal(entry?.tier, "paused");
+ assert.equal(entry?.autoPaused?.cause, "not-answering");
+ const reason = autoPauseReasonOf(h.io.read().channelPriority, "slow");
+ assert.match(String(reason), /drive that is not answering/);
+ assert.match(String(reason), /when the drive answers again/);
+ // The sanitizer keeps the cause; a record without one reads as not there.
+ const kept = sanitizeChannelPriority(h.io.read().channelPriority);
+ assert.equal(kept.channels.slow?.autoPaused?.cause, "not-answering");
+ const old = sanitizeChannelPriority({
+ channels: {
+ a: { tier: "paused", autoPaused: { reason: "storage", since: "", previousTier: "low" } },
+ },
+ });
+ assert.match(String(autoPauseReasonOf(old, "a")), /drive that is not there/);
+ });
+});
diff --git a/common/controller/storageWatch.ts b/common/controller/storageWatch.ts
@@ -14,6 +14,7 @@ import {
resolveFocusSlugs,
restoreAfterMedia,
sanitizeChannelPriority,
+ type AutoPauseCause,
type ChannelPriority,
} from "../lib/channelPriority";
import { LANES } from "../lib/autoQueueTypes";
@@ -204,6 +205,7 @@ export async function runStorageWatchPass(
const configs = await listChannelConfigs(paths);
let model: ChannelPriority = settings.channelPriority;
+ const pauseCauses = new Map<string, AutoPauseCause>();
for (const { slug, config } of configs) {
const wasAutoPaused = Boolean(model.channels[slug]?.autoPaused);
@@ -236,10 +238,19 @@ export async function runStorageWatchPass(
// `in-transition` is NEVER a reason to pause: a marker means a move is
// running or was interrupted, and the relocate job is precisely the thing
// that would then be refused by the state it created.
+ // A stall is down too, on a location or not (a root typed by hand gets its
+ // `stalled` from the watchdog alone).
const down =
media.status === "in-transition"
? false
- : locationDown || media.status === "unreachable";
+ : locationDown ||
+ media.status === "unreachable" ||
+ media.status === "stalled";
+ // WHICH down it is, for the pause record's words (autoPauseReasonOf).
+ const cause: AutoPauseCause =
+ media.status === "stalled" || probe?.status === "stalled"
+ ? "not-answering"
+ : "not-there";
if (down && !wasAutoPaused) {
// ONE BAD READ IS A SUSPICION, TWO IN A ROW IS A FACT. See rule 3.
@@ -254,7 +265,8 @@ export async function runStorageWatchPass(
continue;
}
const before = model;
- model = autoPauseForMedia(model, slug);
+ model = autoPauseForMedia(model, slug, new Date(), cause);
+ pauseCauses.set(slug, cause);
// autoPauseForMedia no-ops on a channel the OPERATOR already paused —
// which is right, and means "nothing changed" is a normal outcome here.
if (model !== before) {
@@ -310,7 +322,9 @@ export async function runStorageWatchPass(
)
: stored;
let merged: ChannelPriority = base;
- for (const slug of out.paused) merged = autoPauseForMedia(merged, slug);
+ for (const slug of out.paused) {
+ merged = autoPauseForMedia(merged, slug, new Date(), pauseCauses.get(slug));
+ }
for (const slug of out.restored) merged = restoreAfterMedia(merged, slug);
merged = sanitizeChannelPriority(merged);
// AND THE TREES, in the same write. Two writes to one settings file race each
@@ -347,16 +361,19 @@ export const STORAGE_WATCH_INTERVAL_MS = 5 * 60_000;
// cable pulled out. It is no help for a drive that is here and stalled — an
// SMR disk in a USB enclosure resetting under a long write — because every
// in-process call on that drive waits for it, and four waits stop the editor
-// answering at all. So every 15 s this asks each location's root, out of
-// process (`probeLocationHealth`), and records the answer in
-// `lib/storageHealth.ts`, which every page and poll consults before it touches
-// a drive. One miss marks a location `stalled` at once; two clean answers in a
-// row clear it (the rules are that module's).
+// answering at all. So every 15 s this reads each location's block device
+// counters in /sys (`detectLocationHealth`; a child `stat` of the root only
+// where no device can be named) and records the answer in `lib/storageHealth.ts`,
+// which every page and poll consults before it touches a drive. One `stalled`
+// answer marks a location at once; two clean answers in a row clear it (the
+// rules are that module's).
//
// READ-ONLY AND IN MEMORY. It writes no settings and pauses nothing: the pass
// above sees a stalled location as down (its probe answers `stalled` without
-// asking) and pauses on its own cadence. Armed and stopped with the watch, so an
-// idle boot (which does not arm the watch) has no health probe either.
+// asking) and pauses on its own cadence. So, like the boot probe, it is armed
+// ABOVE the idle gate (`startStorageHealthWatch`, from instrumentation): an idle
+// boot has no five-minute pass, but it has the health pass — without it nothing
+// registers the locations, and a stall the watchdog marks is never cleared.
export type StorageHealthPassOpts = {
// Default: the configured locations, read from settings.
@@ -393,14 +410,23 @@ function recordVerdict(
verdict: HealthVerdict,
now: number,
): HealthTransition | null {
+ // The counters' device, for the watchdog's slow-or-stalled check; a stat
+ // verdict forgets it.
+ const device =
+ verdict.detector === "counters"
+ ? verdict.device
+ : verdict.detector === "stat"
+ ? null
+ : undefined;
if (verdict.answer === null) {
- if (verdict.detector) noteLocationDetector(loc.id, verdict.detector);
+ if (verdict.detector) noteLocationDetector(loc.id, verdict.detector, device);
return null;
}
return recordLocationHealth(loc, verdict.answer, {
now,
cause: verdict.cause ?? STAT_CAUSE,
...(verdict.detector ? { detector: verdict.detector } : {}),
+ ...(device !== undefined ? { device } : {}),
});
}
@@ -474,9 +500,10 @@ export async function refreshLocationHealth(
// The cadence
// ---------------------------------------------------------------------------
-// Fifteen seconds (lib/storageHealth.ts says why). Each pass is one short-lived
-// findmnt per location and a read of its device's counters in /sys (or, with no
-// device, one short-lived `stat`).
+// Fifteen seconds (lib/storageHealth.ts says why). Each pass reads each
+// location's device counters in /sys (a findmnt only when the device is not
+// known yet or its /sys entry stopped reading; with no device, one short-lived
+// `stat`).
export const STORAGE_HEALTH_INTERVAL_MS = HEALTH_PROBE_INTERVAL_MS;
// A per-module-copy singleton, deliberately left so: it is not a temp-file
@@ -488,16 +515,11 @@ let timer: ReturnType<typeof setInterval> | null = null;
let healthTimer: ReturnType<typeof setInterval> | null = null;
let healthInFlight = false;
-// ARMED ONCE PER PROCESS. `unref()` so it never holds the event loop open — a
-// CLI that imports a controller must still exit. Two timers: the five-minute
-// watch pass and the fifteen-second health pass, which also runs once at once
-// so a drive that is already stalled at boot is known before the first page.
+// THE FIVE-MINUTE PASS, ARMED ONCE PER PROCESS, below the idle gate: it writes
+// settings (an auto-pause). `unref()` so it never holds the event loop open — a
+// CLI that imports a controller must still exit.
export function startStorageWatch(
- opts: StorageWatchOpts & {
- intervalMs?: number;
- healthIntervalMs?: number;
- healthProbe?: LocationHealthProbe;
- } = {},
+ opts: StorageWatchOpts & { intervalMs?: number } = {},
): boolean {
if (timer) return false;
const every = opts.intervalMs ?? STORAGE_WATCH_INTERVAL_MS;
@@ -509,15 +531,37 @@ export function startStorageWatch(
});
}, every);
timer.unref?.();
+ return true;
+}
+
+export function stopStorageWatch(): void {
+ if (timer) clearInterval(timer);
+ timer = null;
+}
+
+// THE FIFTEEN-SECOND HEALTH PASS, ARMED ONCE PER PROCESS, above the idle gate:
+// it is in memory and writes nothing (see the section header). It also runs
+// once at once, so the locations are registered and a drive that is already
+// stalled is on its way to being known before the first page.
+export function startStorageHealthWatch(
+ opts: {
+ io?: { read: () => SiteSettings };
+ bins?: Pick<VolumeBins, "findmntBin">;
+ probe?: LocationHealthProbe;
+ intervalMs?: number;
+ log?: (line: string) => void;
+ } = {},
+): boolean {
+ if (healthTimer) return false;
const health = () => {
- // One at a time: a pass is bounded by the probe's timer, but a pass that
- // overran the interval must not stack a second one on top of it.
+ // One at a time: a pass is bounded by its timers, but a pass that overran
+ // the interval must not stack a second one on top of it.
if (healthInFlight) return;
healthInFlight = true;
void runStorageHealthPass({
io: opts.io,
- bins: opts.bins ?? opts.paths,
- probe: opts.healthProbe,
+ bins: opts.bins,
+ probe: opts.probe,
log: opts.log,
})
.catch((err) => {
@@ -529,18 +573,13 @@ export function startStorageWatch(
healthInFlight = false;
});
};
- healthTimer = setInterval(
- health,
- opts.healthIntervalMs ?? STORAGE_HEALTH_INTERVAL_MS,
- );
+ healthTimer = setInterval(health, opts.intervalMs ?? STORAGE_HEALTH_INTERVAL_MS);
healthTimer.unref?.();
health();
return true;
}
-export function stopStorageWatch(): void {
- if (timer) clearInterval(timer);
+export function stopStorageHealthWatch(): void {
if (healthTimer) clearInterval(healthTimer);
- timer = null;
healthTimer = null;
}
diff --git a/common/lib/channelPriority.ts b/common/lib/channelPriority.ts
@@ -183,8 +183,15 @@ export type ChannelAutoPause = {
reason: "storage";
since: string;
previousTier: StoredChannelTier;
+ cause?: AutoPauseCause;
};
+// Which storage trouble paused it: the drive is not there (unmounted,
+// unplugged), or it is there and not answering (lib/storageHealth.ts). A record
+// with none was written before the second existed, when the first was the only
+// one.
+export type AutoPauseCause = "not-there" | "not-answering";
+
export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = {
reason:
"One reason today. A union so a second one has somewhere to go, and so " +
@@ -195,6 +202,11 @@ export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = {
"The base tier the channel had before the machine paused it; what a " +
"restore puts back. Never `paused` (that would restore to paused — a " +
"no-op dressed as a restore).",
+ cause:
+ "`not-there` (the drive is unmounted or unplugged) or `not-answering` (it " +
+ "is there and does not answer: a stalled disk). Only what the words on " +
+ "/review, the rack and the channel page say. Optional: a record written " +
+ "before it existed reads as `not-there`.",
};
// Each field is documented in CHANNEL_PRIORITY_FIELD_DOCS below (rendered into SETTINGS.md).
@@ -387,12 +399,16 @@ function sanitizeAutoPause(
reason: "storage",
since: typeof r.since === "string" ? r.since : "",
previousTier: previousTier === "paused" ? DEFAULT_CHANNEL_TIER : previousTier,
+ ...(r.cause === "not-there" || r.cause === "not-answering"
+ ? { cause: r.cause }
+ : {}),
};
}
// --- Auto-pause: the machine's own pause, and its undo ----------------------
-// PAUSE A CHANNEL BECAUSE ITS DRIVE IS NOT THERE, recording what to put back.
+// PAUSE A CHANNEL BECAUSE ITS DRIVE IS NOT THERE (OR NOT ANSWERING), recording
+// what to put back, and which of the two it was.
//
// A NO-OP IN TWO CASES, and both matter. Already auto-paused: a flapping drive
// must not overwrite `previousTier` with the `paused` it wrote last time, which
@@ -403,6 +419,7 @@ export function autoPauseForMedia(
model: ChannelPriority,
slug: string,
now: Date = new Date(),
+ cause?: AutoPauseCause,
): ChannelPriority {
const key = slug.trim();
if (!key) return model;
@@ -421,6 +438,7 @@ export function autoPauseForMedia(
reason: "storage",
since: now.toISOString(),
previousTier,
+ ...(cause ? { cause } : {}),
},
},
},
@@ -466,6 +484,12 @@ export function autoPauseReasonOf(
const auto = model.channels[slug]?.autoPaused;
if (!auto) return null;
const since = auto.since ? ` since ${auto.since.slice(0, 10)}` : "";
+ if (auto.cause === "not-answering") {
+ return (
+ `Auto-paused — its media is on a drive that is not answering${since}. ` +
+ `It returns to ${auto.previousTier} on its own when the drive answers again.`
+ );
+ }
return (
`Auto-paused — its media is on a drive that is not there${since}. ` +
`It returns to ${auto.previousTier} on its own when the drive is back.`
diff --git a/editor/instrumentation.ts b/editor/instrumentation.ts
@@ -4,8 +4,10 @@
//
// See editor/app/scheduler/heartbeat.ts and SCHEDULED_SYNC.md.
//
-// Everything armed here except the shutdown reaper is skipped when the process
-// boots idle (ARCHILYZER_IDLE_BOOT) — see isIdleBoot below.
+// Everything armed here that starts or writes work is skipped when the process
+// boots idle (ARCHILYZER_IDLE_BOOT) — see isIdleBoot below. What stays armed
+// only stops work or only reads: the shutdown reaper, the persisted-pause
+// restore, the storage boot probe and the drive health pass.
//
// The one STATIC import in this file, and safe as one because idleBoot.ts
// imports nothing and touches no Node API: the Edge bundle's static Node-API
@@ -98,6 +100,23 @@ export async function register() {
// Lazy-imported and `void`ed like everything else here: the controller pulls
// in execa transitively (Node-only), and a disk that cannot be probed must
// never block server readiness.
+ // THE DRIVE HEALTH PASS, every 15 s — and ON AN IDLE BOOT TOO, like the
+ // probe below. It reads each location's block device counters (never the
+ // drive) and keeps, in memory only, which drives are not answering; every
+ // page and poll asks it before touching a drive, and its watchdog marks a
+ // drive a page reached and got no answer from. It writes nothing and starts
+ // no work, and without it nothing would ever clear such a mark. The
+ // five-minute pass that may auto-pause channels is armed below the idle gate.
+ // See common/controller/storageWatch.ts and common/lib/storageHealth.ts.
+ try {
+ const { startStorageHealthWatch } = await import(
+ "yt-dlp-transcript-common/controller/storageWatch"
+ );
+ startStorageHealthWatch({ log: (line) => console.log(line) });
+ } catch {
+ /* a health pass that fails to arm must not block server readiness */
+ }
+
// Kept so the queued-meta pass below can wait for it: a re-queue for a
// channel whose location is mid-autoRepoint would be refused as unreachable.
let storagePass: Promise<unknown> = Promise.resolve();
@@ -165,10 +184,11 @@ export async function register() {
// is the thing that looks, on a five-minute cadence, and auto-pauses (and
// later restores) the channels on a location that is not there.
//
- // BELOW THE IDLE GATE, deliberately, and unlike the boot probe above: this
- // one WRITES settings.channelPriority, and a container pointed at somebody
- // else's corpus for the first time has no business rewriting that corpus's
- // priority document. See common/controller/storageWatch.ts.
+ // BELOW THE IDLE GATE, deliberately, and unlike the boot probe and the
+ // health pass above: this one WRITES settings.channelPriority, and a
+ // container pointed at somebody else's corpus for the first time has no
+ // business rewriting that corpus's priority document. See
+ // common/controller/storageWatch.ts.
try {
const { startStorageWatch } = await import(
"yt-dlp-transcript-common/controller/storageWatch"