Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 74050956c077b8c02c3a1841de168790cf3f82e9
parent 30ff28db2d184133fafa743bbeb35f0bb11af533
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue, 29 Sep 2026 23:36:45 -0400

common, editor: review M2, L4, L9 — the health pass is armed above the idle gate; an auto-pause says whether the drive is not there or not answering

- M2: `startStorageHealthWatch` arms the 15 s health pass on its own, and
  instrumentation arms it ABOVE the idle gate, beside the storage boot probe:
  it is in memory and writes nothing, and without it an idle boot has no
  registered locations and a stall the watchdog marks is never cleared. The
  five-minute pass (`startStorageWatch`), which may write an auto-pause, stays
  below the gate and arms nothing else.
- L9: the five-minute pass counts a channel whose inspect says `stalled` as
  down, on a configured location or not.
- L4: the pause record carries `cause: "not-there" | "not-answering"`
  (optional; the sanitizer keeps it; a record without one reads as not there,
  which is what every earlier record meant). autoPauseReasonOf says "on a
  drive that is not answering … when the drive answers again" for the second.
  SETTINGS.md regenerated for the new field.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
MSETTINGS.md | 1+
Mcommon/controller/storageWatch.test.ts | 62+++++++++++++++++++++++++++++++++++++++++++++++++++++---------
Mcommon/controller/storageWatch.ts | 107++++++++++++++++++++++++++++++++++++++++++++++++++++++-------------------------
Mcommon/lib/channelPriority.ts | 26+++++++++++++++++++++++++-
Meditor/instrumentation.ts | 32++++++++++++++++++++++++++------
5 files changed, 178 insertions(+), 50 deletions(-)

diff --git a/SETTINGS.md b/SETTINGS.md @@ -449,6 +449,7 @@ Per entry — each entry spells its own values. | `reason` | One reason today. A union so a second one has somewhere to go, and so a surface can say WHICH machine decided rather than "automatic". | | `since` | ISO, for "auto-paused — media unreachable since <date>". | | `previousTier` | The base tier the channel had before the machine paused it; what a restore puts back. Never `paused` (that would restore to paused — a no-op dressed as a restore). | +| `cause` | `not-there` (the drive is unmounted or unplugged) or `not-answering` (it is there and does not answer: a stalled disk). Only what the words on /review, the rack and the channel page say. Optional: a record written before it existed reads as `not-there`. | Default: diff --git a/common/controller/storageWatch.test.ts b/common/controller/storageWatch.test.ts @@ -6,6 +6,7 @@ import path from "node:path"; import type { Paths } from "../lib/paths"; import type { SiteSettings } from "../lib/settings"; import { + autoPauseReasonOf, compileLanes, sanitizeChannelPriority, } from "../lib/channelPriority"; @@ -15,7 +16,9 @@ import { resetStorageWatchSuspicion, runStorageHealthPass, runStorageWatchPass, + startStorageHealthWatch, startStorageWatch, + stopStorageHealthWatch, stopStorageWatch, } from "./storageWatch"; import { inspectChannelMedia } from "../lib/channelMedia"; @@ -470,15 +473,12 @@ test("a probe that throws is 'could not ask': ok, never stalled", async () => { }); }); -test("arming the watch runs a health pass at once, and stopping it stops both timers", async () => { +test("the health pass is armed on its own, runs once at once, and stops; the five-minute watch arms no health pass", async () => { await withTmp(async (h) => { let asked = 0; - const armed = startStorageWatch({ - paths: h.paths, + const armed = startStorageHealthWatch({ io: h.io, - bins: h.paths, - write: false, - healthProbe: async () => { + probe: async () => { asked += 1; return "stalled"; }, @@ -487,17 +487,22 @@ test("arming the watch runs a health pass at once, and stopping it stops both ti try { assert.equal(armed, true); // Armed once per process. - assert.equal(startStorageWatch({ io: h.io }), false); + assert.equal(startStorageHealthWatch({ io: h.io }), false); for (let i = 0; i < 50 && locationHealth("cold")?.state !== "stalled"; i++) { await new Promise((r) => setTimeout(r, 10)); } assert.equal(asked, 1); assert.equal(locationHealth("cold")?.state, "stalled"); } finally { - stopStorageWatch(); + stopStorageHealthWatch(); } - assert.equal(startStorageWatch({ io: h.io, healthProbe: async () => "ok" }), true); + assert.equal(startStorageHealthWatch({ io: h.io, probe: async () => "ok" }), true); + stopStorageHealthWatch(); + // The watch (below the idle gate) runs nothing at arm time and asks no drive. + resetStorageHealth(); + assert.equal(startStorageWatch({ paths: h.paths, io: h.io, bins: h.paths, write: false }), true); stopStorageWatch(); + assert.equal(locationHealth("cold"), undefined); }); }); @@ -545,3 +550,42 @@ test("the pass registers every location, records a verdict's detector, and a ver assert.match(lines.join("\n"), /"cold": drive not answering — its disk \(sdz1\)/); }); }); + +test("a counters verdict records its device (for the watchdog); a stat verdict forgets it", async () => { + await withTmp(async (h) => { + const verdicts = [ + { answer: null, detector: "counters" as const, device: "sdz1" }, + { answer: "ok" as const, detector: "stat" as const }, + ]; + let i = 0; + const probe = async () => verdicts[i++]; + await runStorageHealthPass({ io: h.io, probe }); + assert.equal(locationHealth("cold")?.device, "sdz1"); + await runStorageHealthPass({ io: h.io, probe }); + assert.equal(locationHealth("cold")?.device, undefined); + assert.equal(locationHealth("cold")?.detector, "stat"); + }); +}); + +test("a stall auto-pauses after two passes, and says the drive is not answering (not that it is not there)", async () => { + await withTmp(async (h) => { + await seedRelocated(h, "slow", { targetExists: true }); + await runStorageHealthPass({ io: h.io, probe: async () => "stalled" }); + await twoPasses(h); + const entry = h.io.read().channelPriority.channels.slow; + assert.equal(entry?.tier, "paused"); + assert.equal(entry?.autoPaused?.cause, "not-answering"); + const reason = autoPauseReasonOf(h.io.read().channelPriority, "slow"); + assert.match(String(reason), /drive that is not answering/); + assert.match(String(reason), /when the drive answers again/); + // The sanitizer keeps the cause; a record without one reads as not there. + const kept = sanitizeChannelPriority(h.io.read().channelPriority); + assert.equal(kept.channels.slow?.autoPaused?.cause, "not-answering"); + const old = sanitizeChannelPriority({ + channels: { + a: { tier: "paused", autoPaused: { reason: "storage", since: "", previousTier: "low" } }, + }, + }); + assert.match(String(autoPauseReasonOf(old, "a")), /drive that is not there/); + }); +}); diff --git a/common/controller/storageWatch.ts b/common/controller/storageWatch.ts @@ -14,6 +14,7 @@ import { resolveFocusSlugs, restoreAfterMedia, sanitizeChannelPriority, + type AutoPauseCause, type ChannelPriority, } from "../lib/channelPriority"; import { LANES } from "../lib/autoQueueTypes"; @@ -204,6 +205,7 @@ export async function runStorageWatchPass( const configs = await listChannelConfigs(paths); let model: ChannelPriority = settings.channelPriority; + const pauseCauses = new Map<string, AutoPauseCause>(); for (const { slug, config } of configs) { const wasAutoPaused = Boolean(model.channels[slug]?.autoPaused); @@ -236,10 +238,19 @@ export async function runStorageWatchPass( // `in-transition` is NEVER a reason to pause: a marker means a move is // running or was interrupted, and the relocate job is precisely the thing // that would then be refused by the state it created. + // A stall is down too, on a location or not (a root typed by hand gets its + // `stalled` from the watchdog alone). const down = media.status === "in-transition" ? false - : locationDown || media.status === "unreachable"; + : locationDown || + media.status === "unreachable" || + media.status === "stalled"; + // WHICH down it is, for the pause record's words (autoPauseReasonOf). + const cause: AutoPauseCause = + media.status === "stalled" || probe?.status === "stalled" + ? "not-answering" + : "not-there"; if (down && !wasAutoPaused) { // ONE BAD READ IS A SUSPICION, TWO IN A ROW IS A FACT. See rule 3. @@ -254,7 +265,8 @@ export async function runStorageWatchPass( continue; } const before = model; - model = autoPauseForMedia(model, slug); + model = autoPauseForMedia(model, slug, new Date(), cause); + pauseCauses.set(slug, cause); // autoPauseForMedia no-ops on a channel the OPERATOR already paused — // which is right, and means "nothing changed" is a normal outcome here. if (model !== before) { @@ -310,7 +322,9 @@ export async function runStorageWatchPass( ) : stored; let merged: ChannelPriority = base; - for (const slug of out.paused) merged = autoPauseForMedia(merged, slug); + for (const slug of out.paused) { + merged = autoPauseForMedia(merged, slug, new Date(), pauseCauses.get(slug)); + } for (const slug of out.restored) merged = restoreAfterMedia(merged, slug); merged = sanitizeChannelPriority(merged); // AND THE TREES, in the same write. Two writes to one settings file race each @@ -347,16 +361,19 @@ export const STORAGE_WATCH_INTERVAL_MS = 5 * 60_000; // cable pulled out. It is no help for a drive that is here and stalled — an // SMR disk in a USB enclosure resetting under a long write — because every // in-process call on that drive waits for it, and four waits stop the editor -// answering at all. So every 15 s this asks each location's root, out of -// process (`probeLocationHealth`), and records the answer in -// `lib/storageHealth.ts`, which every page and poll consults before it touches -// a drive. One miss marks a location `stalled` at once; two clean answers in a -// row clear it (the rules are that module's). +// answering at all. So every 15 s this reads each location's block device +// counters in /sys (`detectLocationHealth`; a child `stat` of the root only +// where no device can be named) and records the answer in `lib/storageHealth.ts`, +// which every page and poll consults before it touches a drive. One `stalled` +// answer marks a location at once; two clean answers in a row clear it (the +// rules are that module's). // // READ-ONLY AND IN MEMORY. It writes no settings and pauses nothing: the pass // above sees a stalled location as down (its probe answers `stalled` without -// asking) and pauses on its own cadence. Armed and stopped with the watch, so an -// idle boot (which does not arm the watch) has no health probe either. +// asking) and pauses on its own cadence. So, like the boot probe, it is armed +// ABOVE the idle gate (`startStorageHealthWatch`, from instrumentation): an idle +// boot has no five-minute pass, but it has the health pass — without it nothing +// registers the locations, and a stall the watchdog marks is never cleared. export type StorageHealthPassOpts = { // Default: the configured locations, read from settings. @@ -393,14 +410,23 @@ function recordVerdict( verdict: HealthVerdict, now: number, ): HealthTransition | null { + // The counters' device, for the watchdog's slow-or-stalled check; a stat + // verdict forgets it. + const device = + verdict.detector === "counters" + ? verdict.device + : verdict.detector === "stat" + ? null + : undefined; if (verdict.answer === null) { - if (verdict.detector) noteLocationDetector(loc.id, verdict.detector); + if (verdict.detector) noteLocationDetector(loc.id, verdict.detector, device); return null; } return recordLocationHealth(loc, verdict.answer, { now, cause: verdict.cause ?? STAT_CAUSE, ...(verdict.detector ? { detector: verdict.detector } : {}), + ...(device !== undefined ? { device } : {}), }); } @@ -474,9 +500,10 @@ export async function refreshLocationHealth( // The cadence // --------------------------------------------------------------------------- -// Fifteen seconds (lib/storageHealth.ts says why). Each pass is one short-lived -// findmnt per location and a read of its device's counters in /sys (or, with no -// device, one short-lived `stat`). +// Fifteen seconds (lib/storageHealth.ts says why). Each pass reads each +// location's device counters in /sys (a findmnt only when the device is not +// known yet or its /sys entry stopped reading; with no device, one short-lived +// `stat`). export const STORAGE_HEALTH_INTERVAL_MS = HEALTH_PROBE_INTERVAL_MS; // A per-module-copy singleton, deliberately left so: it is not a temp-file @@ -488,16 +515,11 @@ let timer: ReturnType<typeof setInterval> | null = null; let healthTimer: ReturnType<typeof setInterval> | null = null; let healthInFlight = false; -// ARMED ONCE PER PROCESS. `unref()` so it never holds the event loop open — a -// CLI that imports a controller must still exit. Two timers: the five-minute -// watch pass and the fifteen-second health pass, which also runs once at once -// so a drive that is already stalled at boot is known before the first page. +// THE FIVE-MINUTE PASS, ARMED ONCE PER PROCESS, below the idle gate: it writes +// settings (an auto-pause). `unref()` so it never holds the event loop open — a +// CLI that imports a controller must still exit. export function startStorageWatch( - opts: StorageWatchOpts & { - intervalMs?: number; - healthIntervalMs?: number; - healthProbe?: LocationHealthProbe; - } = {}, + opts: StorageWatchOpts & { intervalMs?: number } = {}, ): boolean { if (timer) return false; const every = opts.intervalMs ?? STORAGE_WATCH_INTERVAL_MS; @@ -509,15 +531,37 @@ export function startStorageWatch( }); }, every); timer.unref?.(); + return true; +} + +export function stopStorageWatch(): void { + if (timer) clearInterval(timer); + timer = null; +} + +// THE FIFTEEN-SECOND HEALTH PASS, ARMED ONCE PER PROCESS, above the idle gate: +// it is in memory and writes nothing (see the section header). It also runs +// once at once, so the locations are registered and a drive that is already +// stalled is on its way to being known before the first page. +export function startStorageHealthWatch( + opts: { + io?: { read: () => SiteSettings }; + bins?: Pick<VolumeBins, "findmntBin">; + probe?: LocationHealthProbe; + intervalMs?: number; + log?: (line: string) => void; + } = {}, +): boolean { + if (healthTimer) return false; const health = () => { - // One at a time: a pass is bounded by the probe's timer, but a pass that - // overran the interval must not stack a second one on top of it. + // One at a time: a pass is bounded by its timers, but a pass that overran + // the interval must not stack a second one on top of it. if (healthInFlight) return; healthInFlight = true; void runStorageHealthPass({ io: opts.io, - bins: opts.bins ?? opts.paths, - probe: opts.healthProbe, + bins: opts.bins, + probe: opts.probe, log: opts.log, }) .catch((err) => { @@ -529,18 +573,13 @@ export function startStorageWatch( healthInFlight = false; }); }; - healthTimer = setInterval( - health, - opts.healthIntervalMs ?? STORAGE_HEALTH_INTERVAL_MS, - ); + healthTimer = setInterval(health, opts.intervalMs ?? STORAGE_HEALTH_INTERVAL_MS); healthTimer.unref?.(); health(); return true; } -export function stopStorageWatch(): void { - if (timer) clearInterval(timer); +export function stopStorageHealthWatch(): void { if (healthTimer) clearInterval(healthTimer); - timer = null; healthTimer = null; } diff --git a/common/lib/channelPriority.ts b/common/lib/channelPriority.ts @@ -183,8 +183,15 @@ export type ChannelAutoPause = { reason: "storage"; since: string; previousTier: StoredChannelTier; + cause?: AutoPauseCause; }; +// Which storage trouble paused it: the drive is not there (unmounted, +// unplugged), or it is there and not answering (lib/storageHealth.ts). A record +// with none was written before the second existed, when the first was the only +// one. +export type AutoPauseCause = "not-there" | "not-answering"; + export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = { reason: "One reason today. A union so a second one has somewhere to go, and so " + @@ -195,6 +202,11 @@ export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = { "The base tier the channel had before the machine paused it; what a " + "restore puts back. Never `paused` (that would restore to paused — a " + "no-op dressed as a restore).", + cause: + "`not-there` (the drive is unmounted or unplugged) or `not-answering` (it " + + "is there and does not answer: a stalled disk). Only what the words on " + + "/review, the rack and the channel page say. Optional: a record written " + + "before it existed reads as `not-there`.", }; // Each field is documented in CHANNEL_PRIORITY_FIELD_DOCS below (rendered into SETTINGS.md). @@ -387,12 +399,16 @@ function sanitizeAutoPause( reason: "storage", since: typeof r.since === "string" ? r.since : "", previousTier: previousTier === "paused" ? DEFAULT_CHANNEL_TIER : previousTier, + ...(r.cause === "not-there" || r.cause === "not-answering" + ? { cause: r.cause } + : {}), }; } // --- Auto-pause: the machine's own pause, and its undo ---------------------- -// PAUSE A CHANNEL BECAUSE ITS DRIVE IS NOT THERE, recording what to put back. +// PAUSE A CHANNEL BECAUSE ITS DRIVE IS NOT THERE (OR NOT ANSWERING), recording +// what to put back, and which of the two it was. // // A NO-OP IN TWO CASES, and both matter. Already auto-paused: a flapping drive // must not overwrite `previousTier` with the `paused` it wrote last time, which @@ -403,6 +419,7 @@ export function autoPauseForMedia( model: ChannelPriority, slug: string, now: Date = new Date(), + cause?: AutoPauseCause, ): ChannelPriority { const key = slug.trim(); if (!key) return model; @@ -421,6 +438,7 @@ export function autoPauseForMedia( reason: "storage", since: now.toISOString(), previousTier, + ...(cause ? { cause } : {}), }, }, }, @@ -466,6 +484,12 @@ export function autoPauseReasonOf( const auto = model.channels[slug]?.autoPaused; if (!auto) return null; const since = auto.since ? ` since ${auto.since.slice(0, 10)}` : ""; + if (auto.cause === "not-answering") { + return ( + `Auto-paused — its media is on a drive that is not answering${since}. ` + + `It returns to ${auto.previousTier} on its own when the drive answers again.` + ); + } return ( `Auto-paused — its media is on a drive that is not there${since}. ` + `It returns to ${auto.previousTier} on its own when the drive is back.` diff --git a/editor/instrumentation.ts b/editor/instrumentation.ts @@ -4,8 +4,10 @@ // // See editor/app/scheduler/heartbeat.ts and SCHEDULED_SYNC.md. // -// Everything armed here except the shutdown reaper is skipped when the process -// boots idle (ARCHILYZER_IDLE_BOOT) — see isIdleBoot below. +// Everything armed here that starts or writes work is skipped when the process +// boots idle (ARCHILYZER_IDLE_BOOT) — see isIdleBoot below. What stays armed +// only stops work or only reads: the shutdown reaper, the persisted-pause +// restore, the storage boot probe and the drive health pass. // // The one STATIC import in this file, and safe as one because idleBoot.ts // imports nothing and touches no Node API: the Edge bundle's static Node-API @@ -98,6 +100,23 @@ export async function register() { // Lazy-imported and `void`ed like everything else here: the controller pulls // in execa transitively (Node-only), and a disk that cannot be probed must // never block server readiness. + // THE DRIVE HEALTH PASS, every 15 s — and ON AN IDLE BOOT TOO, like the + // probe below. It reads each location's block device counters (never the + // drive) and keeps, in memory only, which drives are not answering; every + // page and poll asks it before touching a drive, and its watchdog marks a + // drive a page reached and got no answer from. It writes nothing and starts + // no work, and without it nothing would ever clear such a mark. The + // five-minute pass that may auto-pause channels is armed below the idle gate. + // See common/controller/storageWatch.ts and common/lib/storageHealth.ts. + try { + const { startStorageHealthWatch } = await import( + "yt-dlp-transcript-common/controller/storageWatch" + ); + startStorageHealthWatch({ log: (line) => console.log(line) }); + } catch { + /* a health pass that fails to arm must not block server readiness */ + } + // Kept so the queued-meta pass below can wait for it: a re-queue for a // channel whose location is mid-autoRepoint would be refused as unreachable. let storagePass: Promise<unknown> = Promise.resolve(); @@ -165,10 +184,11 @@ export async function register() { // is the thing that looks, on a five-minute cadence, and auto-pauses (and // later restores) the channels on a location that is not there. // - // BELOW THE IDLE GATE, deliberately, and unlike the boot probe above: this - // one WRITES settings.channelPriority, and a container pointed at somebody - // else's corpus for the first time has no business rewriting that corpus's - // priority document. See common/controller/storageWatch.ts. + // BELOW THE IDLE GATE, deliberately, and unlike the boot probe and the + // health pass above: this one WRITES settings.channelPriority, and a + // container pointed at somebody else's corpus for the first time has no + // business rewriting that corpus's priority document. See + // common/controller/storageWatch.ts. try { const { startStorageWatch } = await import( "yt-dlp-transcript-common/controller/storageWatch"