commit e6bb04227e530605b1d80ef21433534f8e2ca4d4
parent 2ad584c08ea151347bf72161b6c1d4283d61686c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 30 Sep 2026 00:49:42 -0400
Merge r15/drive-stall (release 15 slice DS) — a drive that stops answering no longer stops the editor: per-location health in memory from the block device's counters (child stat as the fallback) and a 3 s watchdog on every gated call, at most four calls in flight per drive, a slow drive told apart from a stalled one, the pages and polls answer without touching a stalled drive, the index and stats builds hold it, /storage and the rack say not answering since when, auto-pause carries the cause; UV_THREADPOOL_SIZE=16; reviewed SHIP
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
50 files changed, 4587 insertions(+), 248 deletions(-)
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -77,6 +77,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| `AUDIO_CHECK_BACKOFF_FACTOR` | the built-in factor | The audio check's interval backoff factor, in (0, 1], for a one-off run. | common/ytdlp/audioCheckedDownload.ts |
| `ARCHILYZER_STATS_ALLOW_DOWNGRADE` | off | `1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions. | common/controller/buildStats.ts |
| `ARCHILYZER_INDEX_ALLOW_HELD` | off | `1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel. | common/controller/buildIndex.ts |
+| `UV_THREADPOOL_SIZE` | `16` for the editor (`4` is Node's own) | Threads in Node's pool for filesystem calls. A call on a stalled drive holds one until the drive answers, so the editor starts with 16. It buys time for calls already in flight and isolates nothing: the storage health probe and its gate keep new calls off a stalled drive. | Node's libuv (set by editor/package.json `start` and docker/entrypoint.sh) |
| `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts |
| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool |
| `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs |
diff --git a/SETTINGS.md b/SETTINGS.md
@@ -449,6 +449,7 @@ Per entry — each entry spells its own values.
| `reason` | One reason today. A union so a second one has somewhere to go, and so a surface can say WHICH machine decided rather than "automatic". |
| `since` | ISO, for "auto-paused — media unreachable since <date>". |
| `previousTier` | The base tier the channel had before the machine paused it; what a restore puts back. Never `paused` (that would restore to paused — a no-op dressed as a restore). |
+| `cause` | `not-there` (the drive is unmounted or unplugged) or `not-answering` (it is there and does not answer: a stalled disk). Only what the words on /review, the rack and the channel page say. Optional: a record written before it existed reads as `not-there`. |
Default:
diff --git a/common/bin/doctor.ts b/common/bin/doctor.ts
@@ -121,7 +121,9 @@ export async function collectDoctorReport(deps: DoctorDeps): Promise<DoctorRepor
const unreachable: string[] = [];
const { inspectChannelMedia } = await import("../lib/channelMedia");
for (const slug of channelSlugs) {
- const loc = await inspectChannelMedia(paths, slug);
+ const loc = await inspectChannelMedia(paths, slug, undefined, {
+ fresh: true,
+ });
if (loc.status !== "ok" && loc.status !== "in-place") {
unreachable.push(`${slug}: ${loc.status} — ${loc.detail ?? ""}`.trim());
}
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -345,7 +345,9 @@ async function scanSource(
// An unmounted drive is not an empty channel (lib/channelMedia.ts): the
// readdir below would fail, and every record the channel has would be
// removed as gone.
- const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg);
+ const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg, {
+ fresh: true,
+ });
if (isMediaHeld(media.status)) {
held.set(ch.name, heldReason(media, cfg.dataDir, locations));
continue;
@@ -462,7 +464,9 @@ async function scanSource(
// Asked again after the walk: a drive that went away DURING it leaves the
// videos after that point missing from this scan, which would remove them.
// Three syscalls a channel.
- const after = await inspectChannelMedia({ channelsDir }, ch.name, cfg);
+ const after = await inspectChannelMedia({ channelsDir }, ch.name, cfg, {
+ fresh: true,
+ });
if (isMediaHeld(after.status)) {
held.set(ch.name, heldReason(after, cfg.dataDir, locations));
continue;
diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts
@@ -288,7 +288,9 @@ async function scanSource(
channels.set(ch.name, cfg);
// An unmounted drive is not an empty channel (lib/channelMedia.ts): the
// readdir below would fail and every one of its stats would be removed.
- const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg);
+ const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg, {
+ fresh: true,
+ });
if (isMediaHeld(media.status)) {
held.set(ch.name, heldReason(media, cfg.dataDir, locations));
continue;
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -22,6 +22,7 @@ import {
} from "../lib/availability";
import { resolveCookiePolicy } from "../lib/cookiePolicy";
import { assertChannelMediaReachable } from "../lib/channelMedia";
+import { isDriveNotAnswering, onDrive } from "../lib/storageHealth";
import { CLIPS_DIR_NAME } from "../lib/clipWindow";
import { getSettings } from "../lib/settings";
import {
@@ -737,6 +738,19 @@ export async function generateChannelSnapshot(
const config = await readChannelConfig(paths, slug);
await assertChannelMediaReachable(paths, slug, config);
+ // A CHANNEL ON ANOTHER DRIVE IS WALKED THROUGH THE WATCHDOG
+ // (lib/storageHealth.ts `onDrive`): the data/ listing, the keep-latest keys and
+ // each video directory's unit below. At most four of them wait on that drive
+ // at once — this walk runs in the editor's own process after every download
+ // or sync of the channel, sixteen wide, which is exactly while a long write is
+ // stressing the drive — and one that does not answer in 3 s throws, so the
+ // scheduler keeps the last good snapshot.json, as on any failed refresh.
+ // The reconcile pass just below is sequential (one read at a time) and is
+ // not raced.
+ const drive = config?.dataDir?.trim() || undefined;
+ const through = <T>(read: () => Promise<T>): Promise<T> =>
+ drive ? onDrive(drive, read) : read();
+
// Heal any video dir that drifted from the canonical id layout before we read
// data/* (best-effort; never fail snapshot generation on a reconcile error).
try {
@@ -753,7 +767,11 @@ export async function generateChannelSnapshot(
maybeMissingRecord,
roster,
] = await Promise.all([
- readdir(dataDir, { withFileTypes: true }).catch(() => [] as Dirent[]),
+ through(() => readdir(dataDir, { withFileTypes: true })).catch((err) => {
+ // A drive that did not answer is not an empty channel: rethrown.
+ if (isDriveNotAnswering(err)) throw err;
+ return [] as Dirent[];
+ }),
readPlaylistUrls(playlistPath),
readArchive(archivePath),
loadFailedTranscriptions(paths, slug),
@@ -769,6 +787,7 @@ export async function generateChannelSnapshot(
paths,
channelSlug: slug,
keepLatest: config?.keepLatest ?? 0,
+ through,
});
const videoDirNames = dirEntries
.filter((d) => d.isDirectory())
@@ -821,7 +840,8 @@ export async function generateChannelSnapshot(
const limit = pLimit(SNAPSHOT_VIDEO_CONCURRENCY);
const perVideo = await Promise.all(
videoDirNames.map((id) =>
- limit(async () => {
+ // One video directory's reads are one unit through the watchdog.
+ limit(() => through(async () => {
const dir = path.join(dataDir, id);
const files = await readVideoFiles(dir, { checkUntranscribable: true });
// EVERY FILE IN THE DIR, STATTED ONCE, feeding two numbers.
@@ -952,7 +972,7 @@ export async function generateChannelSnapshot(
vttProvenance,
digest,
};
- }),
+ })),
),
);
diff --git a/common/controller/channels.ts b/common/controller/channels.ts
@@ -16,7 +16,8 @@ import {
readVideoFiles,
} from "../lib/videoStatus";
import { loadDigest } from "../lib/digest-server";
-import { readRelocationMarker } from "../lib/channelMedia";
+import { channelMediaStall, readRelocationMarker } from "../lib/channelMedia";
+import { isDriveNotAnswering, onDrive } from "../lib/storageHealth";
// TYPE-ONLY, and it must stay that way: ./channelSnapshot imports
// readChannelConfig from this module, and it drags in the snapshot generator's
// whole dependency graph (lmdb, the archive reader, the digest layer). A value
@@ -87,39 +88,56 @@ async function hasDigestWithItems(videoDir: string): Promise<boolean> {
);
}
-async function countDataFiles(dataDir: string): Promise<{
+// `drive` is the channel's configured target (`config.dataDir`) when its media
+// is on another drive. Then every read goes through `onDrive`: none while that
+// location is stalled, at most four in flight on it, and one that has not
+// answered in 3 s marks it stalled — and the walk answers null ("the drive did
+// not answer") instead of counts. The rest of the walk is refused without a
+// call. An in-place channel's walk is on the corpus disk and is not wrapped.
+async function countDataFiles(
+ dataDir: string,
+ drive?: string,
+): Promise<{
videos: number;
transcripts: number;
downloads: number;
digests: number;
-}> {
+} | null> {
+ const through = <T>(call: () => Promise<T>): Promise<T> =>
+ drive ? onDrive(drive, call) : call();
let dirs: Dirent[];
try {
- dirs = await readdir(dataDir, { withFileTypes: true });
- } catch {
+ dirs = await through(() => readdir(dataDir, { withFileTypes: true }));
+ } catch (err) {
+ if (isDriveNotAnswering(err)) return null;
return { videos: 0, transcripts: 0, downloads: 0, digests: 0 };
}
const videoDirs = dirs.filter((d) => d.isDirectory());
// Bounded: the largest channel has 11,224 video dirs and this used to open
// them all at once.
- const flags = await mapConcurrent(
- videoDirs,
- VIDEO_READ_CONCURRENCY,
- async (d) => {
- const dir = path.join(dataDir, d.name);
- const files = await readVideoFiles(dir);
- return {
- transcript: isVideoTranscribed(files),
- download: isVideoDownloaded(files),
- // Only transcribed videos can carry a digest, so the sidecar read is
- // skipped for the rest — the same conditional per-video sidecar-read
- // pattern channelSnapshot.ts uses for coverage and VTT provenance.
- digest: isVideoTranscribed(files)
- ? await hasDigestWithItems(dir)
- : false,
- };
- },
- );
+ let flags: Array<{ transcript: boolean; download: boolean; digest: boolean }>;
+ try {
+ flags = await mapConcurrent(videoDirs, VIDEO_READ_CONCURRENCY, (d) =>
+ // One video directory's few reads are one call through the watchdog.
+ through(async () => {
+ const dir = path.join(dataDir, d.name);
+ const files = await readVideoFiles(dir);
+ return {
+ transcript: isVideoTranscribed(files),
+ download: isVideoDownloaded(files),
+ // Only transcribed videos can carry a digest, so the sidecar read is
+ // skipped for the rest — the same conditional per-video sidecar-read
+ // pattern channelSnapshot.ts uses for coverage and VTT provenance.
+ digest: isVideoTranscribed(files)
+ ? await hasDigestWithItems(dir)
+ : false,
+ };
+ }),
+ );
+ } catch (err) {
+ if (isDriveNotAnswering(err)) return null;
+ throw err;
+ }
let transcripts = 0;
let downloads = 0;
let digests = 0;
@@ -189,8 +207,18 @@ export async function readChannelStat(
): Promise<ChannelStat | null> {
const config = await readChannelConfig(paths, slug);
if (!config) return null;
+ // A WALK OF `data/` ON A DRIVE THAT IS NOT ANSWERING IS NOT STARTED. The
+ // one-second job-list poll asks this for every channel with a job listed, and
+ // on a stalled drive each readdir and stat in the walk would hold an I/O
+ // thread until the drive came back. No counts is what a caller already
+ // handles (the row draws no progress bar).
+ if (channelMediaStall(config)) return null;
const channelDir = path.join(paths.channelsDir, slug);
- const counts = await countDataFiles(path.join(channelDir, "data"));
+ const counts = await countDataFiles(
+ path.join(channelDir, "data"),
+ config.dataDir?.trim() || undefined,
+ );
+ if (!counts) return null;
return {
slug,
config,
@@ -265,7 +293,14 @@ export async function listChannelStatsFromDisk(
if (!config) continue;
const channelDir = path.join(paths.channelsDir, slug);
const dataDir = path.join(channelDir, "data");
- const counts = await countDataFiles(dataDir);
+ // A batch job's ground truth: no drive passed, so nothing is raced and the
+ // answer is never null.
+ const counts = (await countDataFiles(dataDir)) ?? {
+ videos: 0,
+ transcripts: 0,
+ downloads: 0,
+ digests: 0,
+ };
out.push({
slug,
config,
diff --git a/common/controller/evictClipWindows.ts b/common/controller/evictClipWindows.ts
@@ -96,7 +96,9 @@ async function evictChannel(
// fine. (The JOB also declares `needsMedia: true`, which covers a
// single-channel run before it starts; this covers the corpus-wide one,
// where there is no slug for that guard to check.)
- const media = await inspectChannelMedia(opts.paths, slug);
+ const media = await inspectChannelMedia(opts.paths, slug, undefined, {
+ fresh: true,
+ });
if (media.status !== "ok" && media.status !== "in-place") {
out.skipped.push(
`${slug}: media ${media.status}${media.detail ? ` (${media.detail})` : ""} — nothing was touched`,
diff --git a/common/controller/keptVideos.ts b/common/controller/keptVideos.ts
@@ -2,6 +2,10 @@ import path from "node:path";
import { readdir } from "node:fs/promises";
import type { Paths } from "../lib/paths";
import { loadRawMetadataFromDir } from "../lib/transcripts-server";
+import { mapConcurrent } from "../lib/concurrency";
+
+// Metadata reads in flight while keying a channel's videos by upload date.
+const KEY_READ_CONCURRENCY = 16;
// Rolling "keep-latest" window computation. Given a channel's keepLatest config,
// returns the ids of the newest N videos (by upload date). Used by both the
@@ -13,6 +17,11 @@ export type ComputeKeptOptions = {
paths: Paths;
channelSlug: string;
keepLatest: number;
+ // Runs each video's metadata read. The snapshot passes `onDrive` for a
+ // channel on another drive (lib/storageHealth.ts), which caps the reads in
+ // flight on that drive and gives up on one that does not answer. Default:
+ // the read itself.
+ through?: <T>(read: () => Promise<T>) => Promise<T>;
};
// List a channel's data-dir video ids (directories, skipping dotfiles). Mirrors
@@ -58,16 +67,21 @@ async function uploadKey(videoDir: string, id: string): Promise<string> {
async function keyedVideosNewestFirst(
paths: Paths,
channelSlug: string,
+ through: <T>(read: () => Promise<T>) => Promise<T> = (read) => read(),
): Promise<Array<{ id: string; key: string }>> {
- const ids = await listChannelVideoIds(paths, channelSlug);
+ const ids = await through(() => listChannelVideoIds(paths, channelSlug));
if (ids.length === 0) return [];
const dataDir = path.join(paths.channelsDir, channelSlug, "data");
- const keyed = await Promise.all(
- ids.map(async (id) => ({
- id,
- key: await uploadKey(path.join(dataDir, id), id),
- })),
- );
+ // BOUNDED, like every other corpus-shaped fan-out (lib/concurrency.ts): one
+ // metadata read per video, and the largest channel has eleven thousand. With
+ // a `through` of `onDrive`, at most four of these are on the drive at once
+ // and the rest wait in its queue, which refuses a waiting read only when
+ // nothing on the drive has returned for the watchdog's budget — never for
+ // the queue's depth alone.
+ const keyed = await mapConcurrent(ids, KEY_READ_CONCURRENCY, async (id) => ({
+ id,
+ key: await through(() => uploadKey(path.join(dataDir, id), id)),
+ }));
keyed.sort((a, b) =>
a.key === b.key ? b.id.localeCompare(a.id) : b.key.localeCompare(a.key),
);
@@ -82,9 +96,10 @@ export async function computeKeptVideoIds({
paths,
channelSlug,
keepLatest,
+ through,
}: ComputeKeptOptions): Promise<Set<string>> {
if (!Number.isFinite(keepLatest) || keepLatest <= 0) return new Set();
- const keyed = await keyedVideosNewestFirst(paths, channelSlug);
+ const keyed = await keyedVideosNewestFirst(paths, channelSlug, through);
return new Set(keyed.slice(0, Math.floor(keepLatest)).map((k) => k.id));
}
diff --git a/common/controller/recencyIndex.ts b/common/controller/recencyIndex.ts
@@ -7,6 +7,12 @@ import { mapConcurrent } from "../lib/concurrency";
import { extractVideoId } from "../lib/videoId";
import { uploadKeyFor } from "./keptVideos";
import type { AutoQueueOrder } from "../jobs/autoQueuePolicy";
+import type { ChannelConfig } from "../lib/channelConfig";
+import {
+ isDriveNotAnswering,
+ onDrive,
+ stalledLocationForPath,
+} from "../lib/storageHealth";
// Upload-date lookup for the auto-queue's "newest first" ordering.
//
@@ -189,11 +195,22 @@ async function readTailUploadDate(file: string): Promise<string | null> {
// Date the ids in `wanted` from their on-disk metadata. Removes each id it
// keys from `wanted`, like interpolateFromPlaylist.
+//
+// `drives` maps a channel whose media is on another drive to its configured
+// target. Its reads go through `onDrive`: none while that drive's location is
+// stalled (lib/storageHealth.ts) — each would hold an I/O thread until the drive
+// came back, 32 at a time — at most four in flight on it, and one that has not
+// answered in 3 s marks it stalled. An id not read for that reason is NOT
+// memoized as a miss: it falls through to layers 3 and 4 for now, and a later
+// refresh with the drive answering reads it.
+const NOT_READ = Symbol("not read: the drive is not answering");
+
async function datesFromMetadata(
paths: Paths,
owner: ReadonlyMap<string, string>,
wanted: Set<string>,
out: Map<string, RecencyKey>,
+ drives: ReadonlyMap<string, string> = new Map(),
): Promise<void> {
const todo: string[] = [];
for (const id of wanted) {
@@ -205,7 +222,10 @@ async function datesFromMetadata(
}
continue;
}
- if (!owner.has(id)) continue;
+ const slug = owner.get(id);
+ if (slug === undefined) continue;
+ const drive = drives.get(slug);
+ if (drive && stalledLocationForPath(drive)) continue;
todo.push(id);
if (todo.length >= TAIL_READS_PER_BUILD) {
if (!tailCapLogged) {
@@ -219,19 +239,31 @@ async function datesFromMetadata(
}
}
if (todo.length === 0) return;
- const dates = await mapConcurrent(todo, TAIL_READ_CONCURRENCY, (id) =>
- readTailUploadDate(
- path.join(
+ const dates = await mapConcurrent(
+ todo,
+ TAIL_READ_CONCURRENCY,
+ async (id): Promise<string | null | typeof NOT_READ> => {
+ const slug = owner.get(id) as string;
+ const file = path.join(
paths.channelsDir,
- owner.get(id) as string,
+ slug,
"data",
id,
"metadata.info.json",
- ),
- ),
+ );
+ const drive = drives.get(slug);
+ if (!drive) return readTailUploadDate(file);
+ try {
+ return await onDrive(drive, () => readTailUploadDate(file));
+ } catch (err) {
+ if (isDriveNotAnswering(err)) return NOT_READ;
+ throw err;
+ }
+ },
);
for (const [i, id] of todo.entries()) {
const date = dates[i];
+ if (date === NOT_READ) continue;
if (tailMemo.size < TAIL_MEMO_CAP) tailMemo.set(id, date);
if (!date) continue;
out.set(id, { key: date, estimated: false });
@@ -312,7 +344,12 @@ export function interpolateFromPlaylist(
export type BuildRecencyKeysArgs = {
paths: Paths;
// The channels whose work is in play — the runner's own channel-meta list.
- meta: ReadonlyArray<{ slug: string }>;
+ // The config, when the caller holds it, is what tells layer 2 which channels
+ // are on another drive (their reads go through the stall watchdog).
+ meta: ReadonlyArray<{
+ slug: string;
+ config?: Pick<ChannelConfig, "dataDir"> | null;
+ }>;
// Every video id the caller might sort. Ids outside this set are not keyed.
candidateIds: ReadonlySet<string>;
// videoId -> owning channel slug. Required for the tail-read layer, which has
@@ -468,7 +505,12 @@ export async function buildRecencyKeys({
// auto-transcribe, whose entire candidate set is by definition absent from the
// transcript index.
if (owner && missing.size > 0) {
- await datesFromMetadata(paths, owner, missing, out);
+ const drives = new Map<string, string>();
+ for (const m of meta) {
+ const dir = m.config?.dataDir?.trim();
+ if (dir) drives.set(m.slug, dir);
+ }
+ await datesFromMetadata(paths, owner, missing, out, drives);
}
// Layer 3: playlist interpolation, for videos with nothing on disk at all.
diff --git a/common/controller/relocateChannelMedia.ts b/common/controller/relocateChannelMedia.ts
@@ -38,9 +38,17 @@ import {
type VolumeBins,
} from "../lib/storageVolumes";
import { getFreeBytes } from "../lib/diskSpace";
+import {
+ NOT_ANSWERING,
+ isDriveNotAnswering,
+ onDrive,
+ sinceText,
+ stalledLocation,
+} from "../lib/storageHealth";
import { getSettings, type SiteSettings } from "../lib/settings";
import { formatBytes } from "../lib/format";
import {
+ forgetChannelMedia,
inspectChannelMedia,
relocatedDataDir,
relocationMarkerPath,
@@ -282,10 +290,29 @@ export async function relocationRootPresenceProblem(
const named = locationForRoot(r, storage.locations);
const where = named ? ` (location "${named.id}")` : "";
+ // A destination whose drive is not answering is refused WITHOUT the stat
+ // below: that stat would wait on the drive, and a move onto it would too.
+ const stall = named ? stalledLocation(named) : null;
+ if (stall) {
+ return (
+ `The destination root ${r}${where}: ${NOT_ANSWERING} ` +
+ `(${sinceText(stall.since)}). Wait for it to answer, or check the drive.`
+ );
+ }
+
+ // Through the watchdog: a stat that has not answered in 3 s marks the
+ // location stalled and refuses the same way.
let isDir = false;
try {
- isDir = (await stat(r)).isDirectory();
- } catch {
+ isDir = (await onDrive(named ?? r, () => stat(r))).isDirectory();
+ } catch (err) {
+ if (isDriveNotAnswering(err)) {
+ return (
+ `The destination root ${r}${where}: ${NOT_ANSWERING} ` +
+ `(${err.health ? sinceText(err.health.since) : "a stat of it did not answer"}). ` +
+ `Wait for it to answer, or check the drive.`
+ );
+ }
isDir = false;
}
if (!isDir) {
@@ -373,16 +400,23 @@ async function sweepParked(
// The channel's marker, at `channels/<slug>/.relocating.json`. The FILE is the
// contract — see relocateDir.ts — and these three are the channel's name for it.
+//
+// Each write and the clear also drop the channel from `inspectChannelMedia`'s
+// five-second page memo. Every phase change (copy → swap → reclaim) writes the
+// marker AFTER the link and the config it changes, so a page asks the disk
+// again the moment the move has done something it would see.
async function writeMarker(
paths: Paths,
slug: string,
marker: RelocationMarker,
): Promise<void> {
await writeDirMarker(relocationMarkerPath(paths, slug), marker);
+ forgetChannelMedia(slug);
}
async function clearMarker(paths: Paths, slug: string): Promise<void> {
await clearDirMarker(relocationMarkerPath(paths, slug));
+ forgetChannelMedia(slug);
}
async function readMarkerRaw(
@@ -601,7 +635,9 @@ async function moveOut(args: {
// every guard and exactly wrong here: the rerun that finishes an interrupted
// move is the one caller allowed to see it. A marker for a DIFFERENT target
// was already refused above.
- const location = await inspectChannelMedia(paths, slug, args.config);
+ const location = await inspectChannelMedia(paths, slug, args.config, {
+ fresh: true,
+ });
if (!args.resumed && location.status !== "in-place") {
throw new Error(
`Channel "${slug}" is not in a movable state: ${
@@ -838,7 +874,9 @@ async function moveBack(args: {
//
// `in-transition` is allowed because a marker is what a resume carries, and a
// rerun is the caller this precondition must not refuse.
- const location = await inspectChannelMedia(paths, slug, args.config);
+ const location = await inspectChannelMedia(paths, slug, args.config, {
+ fresh: true,
+ });
if (
!args.resumed &&
location.status !== "ok" &&
diff --git a/common/controller/relocateSavedVideos.ts b/common/controller/relocateSavedVideos.ts
@@ -20,6 +20,13 @@ import {
} from "../lib/settings";
import type { RelocationMarker, RelocationPhase } from "../lib/channelMedia";
import {
+ NOT_ANSWERING,
+ isDriveNotAnswering,
+ onDrive,
+ sinceText,
+ stalledLocation,
+} from "../lib/storageHealth";
+import {
relocatedSavedVideosDir,
savedVideosMarkerPath,
} from "../lib/savedVideoStore";
@@ -156,7 +163,33 @@ export async function inspectSavedVideosStore(
detail: `the store links to ${target}, but no storage location is recorded for it`,
};
}
- if (!(await isDirectory(target))) {
+ // The store's drive is not answering: reported without the stat below,
+ // which would wait on it (the /storage page asks on every render). The
+ // page skips the store's size walk for an unreachable store, too.
+ const stall = loc ? stalledLocation(loc) : null;
+ if (stall) {
+ return {
+ dir,
+ locationId,
+ target,
+ status: "unreachable",
+ detail: `${NOT_ANSWERING} (location "${stall.label}", ${sinceText(stall.since)})`,
+ };
+ }
+ let reachable: boolean;
+ try {
+ reachable = await onDrive(loc ?? target, () => isDirectory(target));
+ } catch (err) {
+ if (!isDriveNotAnswering(err)) throw err;
+ return {
+ dir,
+ locationId,
+ target,
+ status: "unreachable",
+ detail: err.message,
+ };
+ }
+ if (!reachable) {
return {
dir,
locationId,
diff --git a/common/controller/savedVideoInventory.ts b/common/controller/savedVideoInventory.ts
@@ -4,8 +4,13 @@ import type { Paths } from "../lib/paths";
import { mapConcurrent } from "../lib/concurrency";
import { loadSavedVideo } from "../lib/savedVideo-server";
import { savedVideoPath, type SavedVideoPointer } from "../lib/savedVideo";
+import {
+ isDriveNotAnswering,
+ onDrive,
+ stalledLocationForPath,
+} from "../lib/storageHealth";
-const { pathExists, readdir } = fs;
+const { pathExists, readdir, readlink } = fs;
// Channels are read a few at a time so the per-video fan-out inside each one
// still dominates; the product is the real ceiling on open descriptors.
@@ -37,12 +42,23 @@ async function listChannelSlugs(paths: Paths): Promise<string[]> {
// channel when channelSlug is given), by following the saved-video.json pointers
// in each data dir. The pointer's `dir` is absolute, so per-channel store
// overrides resolve correctly without consulting channel config here.
+//
+// `notAnswering`, for a PAGE: pass an array and every channel whose `data/`
+// links onto a storage location whose drive is not answering
+// (lib/storageHealth.ts) is skipped — its pointers are one read per video dir,
+// each of which would wait on the drive — and its slug is pushed there, so the
+// page can say which channels it did not read. The link is read, not followed:
+// it is on the corpus disk. A relocated channel's reads then go through
+// `onDrive`'s watchdog, so a drive that stops answering mid-list is skipped and
+// named the same way. Omitted (the backup job), nothing is skipped or raced.
export async function listSavedVideos({
paths,
channelSlug,
+ notAnswering,
}: {
paths: Paths;
channelSlug?: string;
+ notAnswering?: string[];
}): Promise<SavedVideoEntry[]> {
const slugs = channelSlug ? [channelSlug] : await listChannelSlugs(paths);
// One pointer read per video dir — 78,350 of them across the corpus. Done one
@@ -53,25 +69,43 @@ export async function listSavedVideos({
CHANNEL_CONCURRENCY,
async (slug): Promise<SavedVideoEntry[]> => {
const dataDir = path.join(paths.channelsDir, slug, "data");
- if (!(await pathExists(dataDir))) return [];
- const ids = await readdir(dataDir).catch(() => [] as string[]);
- const entries = await mapConcurrent(
- ids,
- VIDEO_CONCURRENCY,
- async (videoId): Promise<SavedVideoEntry | null> => {
- const videoDir = path.join(dataDir, videoId);
- const pointer = await loadSavedVideo(videoDir);
- if (!pointer) return null;
- return {
- slug,
- videoId,
- videoDir,
- storedPath: savedVideoPath(pointer),
- pointer,
- };
- },
- );
- return entries.filter((e): e is SavedVideoEntry => e !== null);
+ let drive = "";
+ if (notAnswering) {
+ drive = await readlink(dataDir).catch(() => "");
+ if (drive && stalledLocationForPath(drive)) {
+ notAnswering.push(slug);
+ return [];
+ }
+ }
+ const through = <T>(call: () => Promise<T>): Promise<T> =>
+ drive ? onDrive(drive, call) : call();
+ try {
+ if (!(await through(() => pathExists(dataDir)))) return [];
+ const ids = await through(() =>
+ readdir(dataDir).catch(() => [] as string[]),
+ );
+ const entries = await mapConcurrent(
+ ids,
+ VIDEO_CONCURRENCY,
+ async (videoId): Promise<SavedVideoEntry | null> => {
+ const videoDir = path.join(dataDir, videoId);
+ const pointer = await through(() => loadSavedVideo(videoDir));
+ if (!pointer) return null;
+ return {
+ slug,
+ videoId,
+ videoDir,
+ storedPath: savedVideoPath(pointer),
+ pointer,
+ };
+ },
+ );
+ return entries.filter((e): e is SavedVideoEntry => e !== null);
+ } catch (err) {
+ if (!notAnswering || !isDriveNotAnswering(err)) throw err;
+ notAnswering.push(slug);
+ return [];
+ }
},
);
const out = perChannel.flat();
@@ -91,6 +125,7 @@ export type SavedVideoTotals = {
export async function savedVideoTotals(opts: {
paths: Paths;
channelSlug?: string;
+ notAnswering?: string[];
}): Promise<SavedVideoTotals> {
const entries = await listSavedVideos(opts);
let bytes = 0;
diff --git a/common/controller/storageLocations.ts b/common/controller/storageLocations.ts
@@ -24,7 +24,16 @@ import {
type StorageLocationProbe,
type VolumeBins,
} from "../lib/storageVolumes";
-import { inspectChannelMedia, relocatedDataDir } from "../lib/channelMedia";
+import {
+ forgetChannelMedia,
+ inspectChannelMedia,
+ relocatedDataDir,
+} from "../lib/channelMedia";
+import {
+ isDriveNotAnswering,
+ onDrive,
+ stalledLocation,
+} from "../lib/storageHealth";
import { relocationQueueKey } from "../lib/queueKeys";
import {
runManagedFunction,
@@ -247,61 +256,85 @@ export async function volumeFreeBytes(opts: {
out[INTERNAL_LOCATION_ID] = Number.isFinite(corpus) ? corpus : undefined;
await Promise.all(
opts.locations.map(async (loc) => {
- const st = await stat(loc.root).catch(() => null);
- if (!st?.isDirectory()) {
+ // NOT ASKED WHILE ITS DRIVE IS NOT ANSWERING: each stat and the statfs
+ // below would hold an I/O thread until it did. "—", like unmounted. The
+ // calls that reach the drive go through `onDrive`'s watchdog; one that
+ // has not answered in 3 s marks the location stalled, and reads "—" too.
+ if (stalledLocation(loc)) {
out[loc.id] = undefined;
return;
}
- // THE DIRECTORY EXISTING IS NOT THE DRIVE BEING THERE, and this is the
- // half the stat above could not catch. An unmounted mountpoint is a real,
- // empty directory ON ITS PARENT'S FILESYSTEM — so `getFreeBytes` succeeds
- // and reports the parent volume's free space, which on this machine is
- // the disk the operator is trying to empty. The location would then read
- // "233 GB free" about a platter that is not plugged in.
- //
- // THE BOUNDARY IS TESTED AT THE MOUNTPOINT, NEVER AT THE ROOT, and the
- // first version of this got that wrong in the one way that matters on
- // this machine. A location's root is `join(mountpoint, relPath)` — the
- // production `platter` is `/run/media/<user>/<uuid>/archilyzer-media`, a
- // SUBDIRECTORY of the mountpoint — so the root and its parent are on the
- // same filesystem BY CONSTRUCTION whenever `relPath` is non-empty, and
- // comparing those two devices reported "free space unknown" for a
- // correctly mounted drive. A bind mount reads the same way.
- //
- // Crossing a mount changes the device number, so `stat(mountpoint).dev`
- // against `stat(dirname(mountpoint)).dev` is the honest question, and it
- // is two syscalls — TABLES NEVER PROBE (see the header) rules out asking
- // `probeLocation`, which is up to three subprocesses.
- //
- // Only asked of a location that has learned a `volume.uuid`: that field
- // is the assertion that the root is supposed to be on its own volume. A
- // location on a plain directory (never probed, a container, a
- // subdirectory of the system disk by design) shares its parent's device
- // legitimately, and withholding its free space would be wrong.
- const mountpoint = loc.volume?.uuid
- ? (loc.volume.mountpoint ?? "").trim()
- : "";
- // `/` is its own parent, so a volume mounted at the root has no boundary
- // to test and is trivially there — the process is reading from it.
- if (mountpoint && mountpoint !== path.dirname(mountpoint)) {
- const [atMount, aboveMount] = await Promise.all([
- stat(mountpoint).catch(() => null),
- stat(path.dirname(mountpoint)).catch(() => null),
- ]);
- // Gone entirely, or present as an ordinary directory on the parent
- // filesystem: either way nothing is mounted there.
- if (!atMount || (aboveMount && aboveMount.dev === atMount.dev)) {
- out[loc.id] = undefined;
- return;
- }
+ try {
+ out[loc.id] = await freeOnLocation(loc);
+ } catch (err) {
+ if (!isDriveNotAnswering(err)) throw err;
+ out[loc.id] = undefined;
}
- const free = await getFreeBytes(loc.root);
- out[loc.id] = Number.isFinite(free) ? free : undefined;
}),
);
return out;
}
+// One location's free space, for volumeFreeBytes. Throws DriveNotAnsweringError
+// when its drive does not answer.
+async function freeOnLocation(
+ loc: StorageLocation,
+): Promise<number | undefined> {
+ const orNull = <T>(p: Promise<T>): Promise<T | null> =>
+ p.catch((err) => {
+ if (isDriveNotAnswering(err)) throw err;
+ return null;
+ });
+ const st = await orNull(onDrive(loc, () => stat(loc.root)));
+ if (!st?.isDirectory()) return undefined;
+ // THE DIRECTORY EXISTING IS NOT THE DRIVE BEING THERE, and this is the
+ // half the stat above could not catch. An unmounted mountpoint is a real,
+ // empty directory ON ITS PARENT'S FILESYSTEM — so `getFreeBytes` succeeds
+ // and reports the parent volume's free space, which on this machine is
+ // the disk the operator is trying to empty. The location would then read
+ // "233 GB free" about a platter that is not plugged in.
+ //
+ // THE BOUNDARY IS TESTED AT THE MOUNTPOINT, NEVER AT THE ROOT, and the
+ // first version of this got that wrong in the one way that matters on
+ // this machine. A location's root is `join(mountpoint, relPath)` — the
+ // production `platter` is `/run/media/<user>/<uuid>/archilyzer-media`, a
+ // SUBDIRECTORY of the mountpoint — so the root and its parent are on the
+ // same filesystem BY CONSTRUCTION whenever `relPath` is non-empty, and
+ // comparing those two devices reported "free space unknown" for a
+ // correctly mounted drive. A bind mount reads the same way.
+ //
+ // Crossing a mount changes the device number, so `stat(mountpoint).dev`
+ // against `stat(dirname(mountpoint)).dev` is the honest question, and it
+ // is two syscalls — TABLES NEVER PROBE (see the header) rules out asking
+ // `probeLocation`, which is up to three subprocesses.
+ //
+ // Only asked of a location that has learned a `volume.uuid`: that field
+ // is the assertion that the root is supposed to be on its own volume. A
+ // location on a plain directory (never probed, a container, a
+ // subdirectory of the system disk by design) shares its parent's device
+ // legitimately, and withholding its free space would be wrong.
+ const mountpoint = loc.volume?.uuid
+ ? (loc.volume.mountpoint ?? "").trim()
+ : "";
+ // `/` is its own parent, so a volume mounted at the root has no boundary
+ // to test and is trivially there — the process is reading from it.
+ if (mountpoint && mountpoint !== path.dirname(mountpoint)) {
+ // The mountpoint is the drive's own top directory; its parent is on the
+ // filesystem above it, so only the first goes through the watchdog.
+ const [atMount, aboveMount] = await Promise.all([
+ orNull(onDrive(loc, () => stat(mountpoint))),
+ stat(path.dirname(mountpoint)).catch(() => null),
+ ]);
+ // Gone entirely, or present as an ordinary directory on the parent
+ // filesystem: either way nothing is mounted there.
+ if (!atMount || (aboveMount && aboveMount.dev === atMount.dev)) {
+ return undefined;
+ }
+ }
+ const free = await onDrive(loc, () => getFreeBytes(loc.root));
+ return Number.isFinite(free) ? free : undefined;
+}
+
// ---------------------------------------------------------------------------
// The probe memo
// ---------------------------------------------------------------------------
@@ -528,7 +561,9 @@ export async function preflightRepoint(opts: {
const missing: string[] = [];
for (const slug of slugs) {
const config = await readChannelConfig(opts.paths, slug);
- const media = await inspectChannelMedia(opts.paths, slug, config);
+ const media = await inspectChannelMedia(opts.paths, slug, config, {
+ fresh: true,
+ });
if (media.status === "in-transition") {
base.problems.push(
`${slug}: a media relocation is in flight or was interrupted ` +
@@ -795,6 +830,10 @@ export async function repointStorageLocation(opts: {
}
resetStorageProbeMemo();
+ // Every channel on it has a new link and a new dataDir: the page memo's keys
+ // already differ, and this drops the old answers rather than letting them
+ // age out.
+ forgetChannelMedia();
log(
`Done. "${loc.label}" is at ${pre.newRoot}; ${pre.channels.length} ` +
`channel(s) re-pointed.`,
diff --git a/common/controller/storageStall.test.ts b/common/controller/storageStall.test.ts
@@ -0,0 +1,544 @@
+// A STALLED DRIVE IS NOT ASKED: every page-and-poll path that would touch a
+// storage location's drive in-process asks the health state first
+// (lib/storageHealth.ts) and, on a stalled location, answers WITHOUT the call.
+//
+// No test stalls a real drive. The drive here is an ordinary temp directory
+// that answers every call at once; the health state is TOLD it is stalled.
+// Every node:fs and node:fs/promises call this file's code makes is recorded
+// (the buildIndex.test.ts spy, reads included), so a gate that let one call
+// through shows up as a recorded path under the drive — it cannot hide behind
+// a call that happened to answer quickly.
+//
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/storageStall.test.ts
+
+import { after, beforeEach, test } from "node:test";
+import assert from "node:assert/strict";
+import { createRequire, syncBuiltinESMExports } from "node:module";
+import {
+ existsSync,
+ mkdirSync,
+ mkdtempSync,
+ rmSync,
+ symlinkSync,
+ writeFileSync,
+} from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import type { Paths } from "../lib/paths";
+import type { StorageLocation } from "../lib/storageLocations";
+import {
+ assertChannelMediaReachable,
+ ChannelMediaUnreachableError,
+ CHANNEL_MEDIA_MEMO_MS,
+ clearRelocationMarker,
+ forgetChannelMedia,
+ inspectChannelMedia,
+} from "../lib/channelMedia";
+import { HELD_REASON, isMediaHeld } from "../lib/channelMediaHold";
+import {
+ locationHealth,
+ recordLocationHealth,
+ registerLocationHealth,
+ resetStorageHealth,
+ setDriveCallBudget,
+} from "../lib/storageHealth";
+import {
+ probeLocation,
+ probeLocationMemo,
+ resetStorageProbeMemo,
+} from "../lib/storageVolumes";
+import { volumeFreeBytes } from "./storageLocations";
+import { readChannelStat } from "./channels";
+import { buildRecencyKeys, clearRecencyCache } from "./recencyIndex";
+import { relocationRootPresenceProblem } from "./relocateChannelMedia";
+import { generateChannelSnapshot } from "./channelSnapshot";
+import { inspectSavedVideosStore } from "./relocateSavedVideos";
+import { listSavedVideos } from "./savedVideoInventory";
+import type { SiteSettings } from "../lib/settings";
+
+// ── the spy ────────────────────────────────────────────────────────────────
+type Call = { fn: string; path: string };
+let calls: Call[] = [];
+// THE HANG: a promise-API call whose (function, path) matches never settles —
+// what a read blocked on a stalled drive looks like from here. Recorded like
+// any other call. No drive is involved.
+let hang: ((c: Call) => boolean) | null = null;
+{
+ const req = createRequire(import.meta.url);
+ const fsCjs = req("node:fs") as Record<string, unknown>;
+ const fspCjs = req("node:fs/promises") as Record<string, unknown>;
+ const NAMES = [
+ "access", "appendFile", "chmod", "copyFile", "cp", "lstat", "mkdir",
+ "open", "opendir", "readdir", "readFile", "readlink", "realpath", "rename",
+ "rm", "rmdir", "stat", "statfs", "symlink", "unlink", "utimes", "writeFile",
+ ];
+ const asPath = (v: unknown) =>
+ typeof v === "string"
+ ? v
+ : v instanceof URL
+ ? fileURLToPath(v)
+ : Buffer.isBuffer(v)
+ ? v.toString()
+ : null;
+ const wrap = (mod: Record<string, unknown>, name: string, promises = false) => {
+ const fn = mod[name];
+ if (typeof fn !== "function") return;
+ mod[name] = function (this: unknown, ...args: unknown[]) {
+ const p = asPath(args[0]);
+ if (p !== null) {
+ const call = { fn: name, path: path.resolve(p) };
+ calls.push(call);
+ if (promises && hang?.(call)) return new Promise(() => {});
+ }
+ return (fn as (...a: unknown[]) => unknown).apply(this, args);
+ };
+ };
+ for (const n of NAMES) {
+ wrap(fspCjs, n, true);
+ wrap(fsCjs, n);
+ wrap(fsCjs, `${n}Sync`);
+ }
+ wrap(fsCjs, "existsSync");
+ wrap(fsCjs, "createReadStream");
+ syncBuiltinESMExports();
+}
+
+// ── the corpus and the drive ───────────────────────────────────────────────
+const ROOT = mkdtempSync(path.join(tmpdir(), "ttb-stall-"));
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const CORPUS = path.join(ROOT, "corpus");
+const DRIVE = path.join(ROOT, "drive");
+const SLUG = "on-drive";
+const paths = {
+ transcriptsDir: CORPUS,
+ channelsDir: path.join(CORPUS, "channels"),
+ lmdbPath: path.join(CORPUS, "index.mdb"),
+ savedVideosDir: path.join(CORPUS, "saved-videos"),
+ jobsDir: path.join(CORPUS, "jobs"),
+ // A findmnt that leaves a mark if anything runs it.
+ findmntBin: path.join(ROOT, "findmnt-ran.sh"),
+ udisksctlBin: path.join(ROOT, "no-udisksctl"),
+} as unknown as Paths;
+const LOC: StorageLocation = {
+ id: "usb",
+ label: "USB drive",
+ root: DRIVE,
+ autoRepoint: false,
+};
+const TARGET = path.join(DRIVE, SLUG, "data");
+const LINK = path.join(paths.channelsDir, SLUG, "data");
+const CONFIG = { dataDir: TARGET };
+const FINDMNT_MARK = path.join(ROOT, "findmnt-ran");
+
+function seed(): void {
+ rmSync(CORPUS, { recursive: true, force: true });
+ rmSync(DRIVE, { recursive: true, force: true });
+ mkdirSync(path.join(TARGET, "vid1"), { recursive: true });
+ writeFileSync(
+ path.join(TARGET, "vid1", "metadata.info.json"),
+ JSON.stringify({ id: "vid1", upload_date: "20260601" }),
+ );
+ writeFileSync(path.join(TARGET, "vid1", "transcript.en.vtt"), "WEBVTT\n");
+ mkdirSync(path.join(paths.channelsDir, SLUG), { recursive: true });
+ writeFileSync(
+ path.join(paths.channelsDir, SLUG, "config.json"),
+ JSON.stringify({
+ handling: "youtube",
+ name: SLUG,
+ url: `https://www.youtube.com/@${SLUG}/videos`,
+ dataDir: TARGET,
+ }),
+ );
+ symlinkSync(TARGET, LINK);
+ writeFileSync(paths.findmntBin, `#!/bin/sh\ntouch ${FINDMNT_MARK}\nexit 1\n`, {
+ mode: 0o755,
+ });
+ rmSync(FINDMNT_MARK, { force: true });
+}
+
+// Anything that reaches the drive: a path under its root, or through the
+// channel's link (`data/` itself, followed, or anything below it).
+function onDrive(c: Call): boolean {
+ const under = (p: string, base: string) =>
+ p === base || p.startsWith(base + path.sep);
+ return under(c.path, DRIVE) || under(c.path, LINK);
+}
+
+function stall(): void {
+ recordLocationHealth(LOC, "stalled", { now: Date.now() });
+}
+
+beforeEach(() => {
+ hang = null;
+ setDriveCallBudget();
+ resetStorageHealth();
+ forgetChannelMedia();
+ resetStorageProbeMemo();
+ clearRecencyCache();
+ seed();
+ calls = [];
+});
+
+// ── inspectChannelMedia: the gate and the memo ─────────────────────────────
+
+const MARKER = path.join(paths.channelsDir, SLUG, ".relocating.json");
+
+test("inspect with the config in hand: on a stalled drive only the marker is read, on the corpus disk", async () => {
+ stall();
+ calls = [];
+ const media = await inspectChannelMedia(paths, SLUG, CONFIG);
+ assert.deepEqual(calls, [{ fn: "readFile", path: MARKER }]);
+ assert.deepEqual(calls.filter(onDrive), []);
+ assert.equal(media.status, "stalled");
+ assert.equal(media.target, TARGET);
+ assert.match(String(media.detail), /^drive not answering \(location "USB drive", since /);
+ // Held by both pool-wide builds, with a reason that names no path.
+ assert.equal(isMediaHeld(media.status), true);
+ assert.doesNotMatch(HELD_REASON.stalled, /\//);
+});
+
+test("inspect without the config reads config.json and nothing on the drive", async () => {
+ stall();
+ calls = [];
+ const media = await inspectChannelMedia(paths, SLUG);
+ assert.equal(media.status, "stalled");
+ assert.deepEqual(
+ calls.map((c) => [c.fn, path.relative(ROOT, c.path)]),
+ [
+ ["readFile", path.join("corpus", "channels", SLUG, "config.json")],
+ ["readFile", path.join("corpus", "channels", SLUG, ".relocating.json")],
+ ],
+ );
+});
+
+test("the marker first: a channel mid-move on a stalled drive reads in-transition", async () => {
+ writeFileSync(
+ MARKER,
+ JSON.stringify({ target: TARGET, direction: "back", startedAt: "", phase: "copy" }),
+ );
+ stall();
+ calls = [];
+ const media = await inspectChannelMedia(paths, SLUG, CONFIG);
+ assert.equal(media.status, "in-transition");
+ assert.deepEqual(calls.filter(onDrive), []);
+ // Remembered as a move: a second look within five seconds gives the move
+ // back, not the stall.
+ const again = await inspectChannelMedia(paths, SLUG, CONFIG);
+ assert.equal(again.status, "in-transition");
+});
+
+test("the start-of-work guard refuses a stalled channel without a call", async () => {
+ stall();
+ calls = [];
+ await assert.rejects(
+ () => assertChannelMediaReachable(paths, SLUG, CONFIG),
+ (err: unknown) =>
+ err instanceof ChannelMediaUnreachableError &&
+ err.status === "stalled" &&
+ /drive not answering/.test(err.message),
+ );
+ assert.deepEqual(calls.filter(onDrive), []);
+});
+
+test("the memo: two inspects within 5 s stat the drive once; fresh and age each ask again", async () => {
+ const t0 = 1_000_000;
+ const targetStats = () =>
+ calls.filter((c) => c.fn === "stat" && c.path === TARGET).length;
+ const a = await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 });
+ assert.equal(a.status, "ok");
+ assert.equal(targetStats(), 1);
+ const b = await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 + 4_999 });
+ assert.equal(b.status, "ok");
+ assert.equal(targetStats(), 1, "a second inspect inside five seconds is remembered");
+ await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 + 4_999, fresh: true });
+ assert.equal(targetStats(), 2, "fresh bypasses the memo");
+ await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 + CHANNEL_MEDIA_MEMO_MS });
+ assert.equal(targetStats(), 3, "an answer five seconds old is asked again");
+ // A fresh answer is not remembered: the memo still holds the one from t0+5000.
+ await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 + CHANNEL_MEDIA_MEMO_MS + 1 });
+ assert.equal(targetStats(), 3);
+});
+
+test("the memo is keyed by the configured target, and a mover's forget clears it", async () => {
+ const now = 2_000_000;
+ const targetStats = () =>
+ calls.filter((c) => c.fn === "stat" && c.path === TARGET).length;
+ await inspectChannelMedia(paths, SLUG, CONFIG, { now });
+ // Another configured target is another key.
+ const other = await inspectChannelMedia(paths, SLUG, { dataDir: path.join(DRIVE, "x", "data") }, { now });
+ assert.equal(other.status, "inconsistent");
+ forgetChannelMedia(SLUG);
+ await inspectChannelMedia(paths, SLUG, CONFIG, { now });
+ assert.equal(targetStats(), 2);
+ // clearRelocationMarker (the operator's last resort) forgets too.
+ await clearRelocationMarker(paths, SLUG);
+ await inspectChannelMedia(paths, SLUG, CONFIG, { now });
+ assert.equal(targetStats(), 3);
+});
+
+test("the gate is asked before the memo: a stall is seen with an ok remembered", async () => {
+ const now = 3_000_000;
+ assert.equal((await inspectChannelMedia(paths, SLUG, CONFIG, { now })).status, "ok");
+ stall();
+ calls = [];
+ const media = await inspectChannelMedia(paths, SLUG, CONFIG, { now: now + 1 });
+ assert.equal(media.status, "stalled");
+ assert.deepEqual(calls, []);
+});
+
+// ── the other gated callers ────────────────────────────────────────────────
+
+test("probeLocation and its memo: 'stalled', with no stat, no statfs and no findmnt", async () => {
+ // A remembered "available" first, so the memo's gate is the thing tested.
+ await probeLocationMemo(LOC, paths);
+ rmSync(FINDMNT_MARK, { force: true });
+ stall();
+ calls = [];
+ const direct = await probeLocation(LOC, paths);
+ const memo = await probeLocationMemo(LOC, paths);
+ assert.equal(direct.status, "stalled");
+ assert.equal(memo.status, "stalled");
+ assert.deepEqual(direct.identity, { known: false });
+ assert.equal(direct.freeBytes, undefined);
+ assert.deepEqual(calls.filter(onDrive), []);
+ assert.equal(existsSync(FINDMNT_MARK), false, "findmnt was not run");
+});
+
+test("volumeFreeBytes: a stalled location reads unknown, with no call on it", async () => {
+ stall();
+ calls = [];
+ const out = await volumeFreeBytes({ paths, locations: [LOC] });
+ assert.equal(out.usb, undefined);
+ assert.equal(typeof out.internal, "number");
+ assert.deepEqual(calls.filter(onDrive), []);
+});
+
+test("readChannelStat: no walk of data/ on a stalled drive", async () => {
+ const before = await readChannelStat(paths, SLUG);
+ assert.equal(before?.videoCount, 1);
+ stall();
+ calls = [];
+ assert.equal(await readChannelStat(paths, SLUG), null);
+ assert.deepEqual(calls.filter(onDrive), []);
+});
+
+test("recency: no tail read on a stalled drive, and no miss remembered for it", async () => {
+ const owner = new Map([["vid1", SLUG]]);
+ const meta = [{ slug: SLUG, config: CONFIG }];
+ const args = {
+ paths,
+ meta,
+ candidateIds: new Set(["vid1"]),
+ owner,
+ interpolate: false,
+ fresh: true,
+ };
+ stall();
+ calls = [];
+ const stalledKeys = await buildRecencyKeys(args);
+ assert.deepEqual(calls.filter(onDrive), []);
+ // Layer 4: an undatable id sorts oldest.
+ assert.deepEqual(stalledKeys.get("vid1"), { key: "", estimated: false });
+ // The drive answers again: the same id is read now, because the stall was
+ // not remembered as a miss.
+ resetStorageHealth();
+ const keys = await buildRecencyKeys(args);
+ assert.deepEqual(keys.get("vid1"), { key: "20260601", estimated: false });
+ assert.ok(calls.some((c) => c.fn === "open" && onDrive(c)));
+});
+
+test("a move onto a stalled location is refused without a stat of its root", async () => {
+ stall();
+ calls = [];
+ const problem = await relocationRootPresenceProblem(
+ DRIVE,
+ { locations: [LOC], defaultLocationId: "" },
+ paths,
+ );
+ assert.match(String(problem), /drive not answering/);
+ assert.match(String(problem), /location "usb"/);
+ assert.deepEqual(calls.filter(onDrive), []);
+});
+
+test("a snapshot refresh of a stalled channel throws before its walk", async () => {
+ stall();
+ calls = [];
+ await assert.rejects(
+ () => generateChannelSnapshot(paths, SLUG),
+ (err: unknown) =>
+ err instanceof ChannelMediaUnreachableError && err.status === "stalled",
+ );
+ assert.deepEqual(calls.filter(onDrive), []);
+});
+
+test("the saved-video store on a stalled drive reads unreachable, without a stat through its link", async () => {
+ const storeTarget = path.join(DRIVE, "saved-videos");
+ mkdirSync(storeTarget, { recursive: true });
+ symlinkSync(storeTarget, paths.savedVideosDir);
+ const settings = {
+ storage: { locations: [LOC], defaultLocationId: "", savedVideosLocationId: "usb" },
+ } as unknown as SiteSettings;
+ assert.equal((await inspectSavedVideosStore(paths, settings)).status, "ok");
+ stall();
+ calls = [];
+ const store = await inspectSavedVideosStore(paths, settings);
+ assert.equal(store.status, "unreachable");
+ assert.match(String(store.detail), /^drive not answering \(location "USB drive"/);
+ const followed = calls.filter(
+ (c) =>
+ onDrive(c) ||
+ (c.path.startsWith(paths.savedVideosDir) && !["lstat", "readlink"].includes(c.fn)),
+ );
+ assert.deepEqual(followed, []);
+});
+
+test("the saved-video inventory, for a page: a stalled channel is named, not read", async () => {
+ // A saved video on the drive, so a channel that WAS read has an entry.
+ // (savedVideoInventory reads through fs-extra, whose functions graceful-fs
+ // captured before this file's spy was installed, so this case proves the skip
+ // by what comes back, not by the spy.)
+ writeFileSync(
+ path.join(TARGET, "vid1", "saved-video.json"),
+ JSON.stringify({ storedAt: "", dir: "/store", file: "v.mp4", bytes: 7 }),
+ );
+ assert.equal((await listSavedVideos({ paths, notAnswering: [] })).length, 1);
+ stall();
+ const notAnswering: string[] = [];
+ assert.deepEqual(await listSavedVideos({ paths, notAnswering }), []);
+ assert.deepEqual(notAnswering, [SLUG]);
+ // Without the array (the backup job) nothing is skipped: it reads the drive.
+ assert.equal((await listSavedVideos({ paths })).length, 1);
+});
+
+// ── the watchdog: a call that does not answer in the budget ────────────────
+// The location is registered (the health pass does that in the editor) but
+// answering; the hang makes one call on the drive never settle. The budget is
+// shortened to 100 ms; lib/storageHealth.test.ts holds the 3 s default.
+
+function hangOnDrive(fns: string[]): void {
+ hang = (c) => fns.includes(c.fn) && onDrive(c);
+}
+
+async function watchdogCase(
+ fns: string[],
+ run: () => Promise<unknown>,
+): Promise<unknown> {
+ registerLocationHealth([LOC]);
+ setDriveCallBudget(100);
+ hangOnDrive(fns);
+ calls = [];
+ const started = Date.now();
+ const out = await run();
+ assert.ok(Date.now() - started < 2_000, "answered on the watchdog, not on the drive");
+ assert.equal(locationHealth("usb")?.state, "stalled", "the location is marked at once");
+ assert.match(String(locationHealth("usb")?.cause), /a read in the editor did not answer/);
+ return out;
+}
+
+test("watchdog: inspect's target stat never answers → stalled, marked, and nothing more is asked of the drive", async () => {
+ const media = (await watchdogCase(["stat"], () =>
+ inspectChannelMedia(paths, SLUG, CONFIG),
+ )) as { status: string; detail?: string };
+ assert.equal(media.status, "stalled");
+ assert.match(String(media.detail), /^drive not answering \(location "USB drive"/);
+ // The next inspect, the guard and the walk make no call on the drive.
+ calls = [];
+ assert.equal((await inspectChannelMedia(paths, SLUG, CONFIG)).status, "stalled");
+ await assert.rejects(() => assertChannelMediaReachable(paths, SLUG, CONFIG));
+ assert.equal(await readChannelStat(paths, SLUG), null);
+ assert.deepEqual(calls.filter(onDrive), []);
+ // Cleared (two clean answers), the drive is asked again.
+ recordLocationHealth(LOC, "ok");
+ recordLocationHealth(LOC, "ok");
+ hang = null;
+ assert.equal(
+ (await inspectChannelMedia(paths, SLUG, CONFIG, { fresh: true })).status,
+ "ok",
+ );
+});
+
+test("watchdog: a video directory's read never answers mid-walk → readChannelStat answers null", async () => {
+ for (const id of ["vid2", "vid3", "vid4", "vid5", "vid6"]) {
+ mkdirSync(path.join(TARGET, id), { recursive: true });
+ }
+ const out = await watchdogCase(["readdir"], async () => {
+ // The walk's first readdir (of data/ itself, through the link) answers;
+ // the video directories' do not.
+ hang = (c) =>
+ c.fn === "readdir" && c.path.startsWith(path.join(LINK, "vid"));
+ return readChannelStat(paths, SLUG);
+ });
+ assert.equal(out, null);
+ // At most four video directories were asked before the stall refused the rest.
+ const asked = calls.filter((c) => c.fn === "readdir" && c.path.startsWith(path.join(LINK, "vid")));
+ assert.ok(asked.length <= 4, `${asked.length} video dirs asked`);
+});
+
+test("watchdog: probeLocation's stat never answers → 'stalled'", async () => {
+ const probe = (await watchdogCase(["stat"], () => probeLocation(LOC, paths))) as {
+ status: string;
+ };
+ assert.equal(probe.status, "stalled");
+});
+
+test("watchdog: volumeFreeBytes' stat never answers → unknown", async () => {
+ const out = (await watchdogCase(["stat"], () =>
+ volumeFreeBytes({ paths, locations: [LOC] }),
+ )) as Record<string, number | undefined>;
+ assert.equal(out.usb, undefined);
+});
+
+test("watchdog: a recency tail read never answers → not dated, not remembered as a miss", async () => {
+ const args = {
+ paths,
+ meta: [{ slug: SLUG, config: CONFIG }],
+ candidateIds: new Set(["vid1"]),
+ owner: new Map([["vid1", SLUG]]),
+ interpolate: false,
+ fresh: true,
+ };
+ const keys = (await watchdogCase(["open"], () => buildRecencyKeys(args))) as Map<
+ string,
+ { key: string }
+ >;
+ assert.equal(keys.get("vid1")?.key, "");
+ resetStorageHealth();
+ hang = null;
+ assert.equal((await buildRecencyKeys(args)).get("vid1")?.key, "20260601");
+});
+
+test("watchdog: a move onto a root whose stat never answers is refused", async () => {
+ const problem = await watchdogCase(["stat"], () =>
+ relocationRootPresenceProblem(
+ DRIVE,
+ { locations: [LOC], defaultLocationId: "" },
+ paths,
+ ),
+ );
+ assert.match(String(problem), /drive not answering/);
+});
+
+test("watchdog (M3): the snapshot walk's video unit never answers → the refresh throws and writes no snapshot", async () => {
+ for (const id of ["vid2", "vid3", "vid4", "vid5", "vid6", "vid7"]) {
+ mkdirSync(path.join(TARGET, id), { recursive: true });
+ }
+ const snapshotFile = path.join(paths.channelsDir, SLUG, "snapshot.json");
+ await watchdogCase(["readdir"], async () => {
+ // data/ itself answers (the listing, the reconcile pass); the video
+ // directories do not.
+ hang = (c) => c.fn === "readdir" && c.path.startsWith(path.join(LINK, "vid"));
+ await assert.rejects(
+ () => generateChannelSnapshot(paths, SLUG),
+ (err: unknown) => err instanceof Error && err.name === "DriveNotAnsweringError",
+ );
+ });
+ assert.equal(existsSync(snapshotFile), false, "the last snapshot.json stands");
+ // At most four video directories reached the drive.
+ const asked = calls.filter(
+ (c) => c.fn === "readdir" && c.path.startsWith(path.join(LINK, "vid")),
+ );
+ assert.ok(asked.length <= 4, `${asked.length} video dirs asked`);
+});
diff --git a/common/controller/storageWatch.test.ts b/common/controller/storageWatch.test.ts
@@ -6,19 +6,36 @@ import path from "node:path";
import type { Paths } from "../lib/paths";
import type { SiteSettings } from "../lib/settings";
import {
+ autoPauseReasonOf,
compileLanes,
sanitizeChannelPriority,
} from "../lib/channelPriority";
import { LANES } from "../lib/autoQueueTypes";
import {
+ refreshLocationHealth,
resetStorageWatchSuspicion,
+ runStorageHealthPass,
runStorageWatchPass,
+ startStorageHealthWatch,
+ startStorageWatch,
+ stopStorageHealthWatch,
+ stopStorageWatch,
} from "./storageWatch";
+import { inspectChannelMedia } from "../lib/channelMedia";
+import {
+ locationHealth,
+ resetStorageHealth,
+ type LocationHealthState,
+} from "../lib/storageHealth";
// THE CONFIRMATION COUNT IS MODULE STATE (see storageWatch.ts rule 3), so each
// case starts from a clean one — otherwise the second test inherits the first
-// test's suspicions and pauses on what should be its first pass.
-beforeEach(() => resetStorageWatchSuspicion());
+// test's suspicions and pauses on what should be its first pass. The health
+// state (lib/storageHealth.ts) is process state for the same reason.
+beforeEach(() => {
+ resetStorageWatchSuspicion();
+ resetStorageHealth();
+});
// Run with:
// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/storageWatch.test.ts
@@ -375,3 +392,200 @@ test("the restore needs only one good pass", async () => {
assert.equal(h.writes, 1);
});
});
+
+// ---------------------------------------------------------------------------
+// The health pass (15 s): is the drive ANSWERING
+// ---------------------------------------------------------------------------
+//
+// The probe is injected: `probeLocationHealth` (a child `stat` raced against a
+// 3 s timer) has its own tests in lib/storageHealthProbe.test.ts. Here the
+// answers are scripted, one per pass.
+
+function scripted(answers: LocationHealthState[]) {
+ let i = 0;
+ return async () => answers[Math.min(i++, answers.length - 1)];
+}
+
+test("one missed probe stalls the location; pages then answer 'stalled' without asking", async () => {
+ await withTmp(async (h) => {
+ await seedRelocated(h, "slow", { targetExists: true });
+ const lines: string[] = [];
+ const r = await runStorageHealthPass({
+ io: h.io,
+ probe: scripted(["stalled"]),
+ log: (l) => lines.push(l),
+ });
+ assert.deepEqual(r.answers, { cold: "stalled" });
+ // Registered first (answering), so the miss is a transition from ok.
+ assert.deepEqual(r.transitions, [{ id: "cold", from: "ok", to: "stalled" }]);
+ assert.equal(locationHealth("cold")?.state, "stalled");
+ assert.match(lines.join("\n"), /"cold": drive not answering/);
+ // The target is there and would answer, but nothing asks it.
+ const media = await inspectChannelMedia(h.paths, "slow");
+ assert.equal(media.status, "stalled");
+ // The five-minute pass sees the location as down (its probe answers
+ // "stalled" without a stat) and suspects the channel, as for any outage.
+ const w = await runStorageWatchPass({ paths: h.paths, io: h.io, bins: h.paths });
+ assert.deepEqual(w.suspected, ["slow"]);
+ assert.equal(h.writes, 0);
+ });
+});
+
+test("the stall clears only after two clean probes in a row", async () => {
+ await withTmp(async (h) => {
+ await seedRelocated(h, "slow", { targetExists: true });
+ const probe = scripted(["stalled", "ok", "stalled", "ok", "ok"]);
+ const lines: string[] = [];
+ const pass = () =>
+ runStorageHealthPass({ io: h.io, probe, log: (l) => lines.push(l) });
+ await pass();
+ await pass(); // one clean answer
+ assert.equal(locationHealth("cold")?.state, "stalled");
+ await pass(); // missed again: the count starts over
+ await pass(); // one clean
+ assert.equal(locationHealth("cold")?.state, "stalled");
+ assert.equal((await inspectChannelMedia(h.paths, "slow")).status, "stalled");
+ const last = await pass(); // two clean in a row
+ assert.deepEqual(last.transitions, [{ id: "cold", from: "stalled", to: "ok" }]);
+ assert.match(lines.at(-1) ?? "", /"cold": answering again/);
+ assert.equal((await inspectChannelMedia(h.paths, "slow")).status, "ok");
+ });
+});
+
+test("a location no longer configured is forgotten", async () => {
+ await withTmp(async (h) => {
+ await runStorageHealthPass({ io: h.io, probe: scripted(["stalled"]) });
+ assert.equal(locationHealth("cold")?.state, "stalled");
+ await runStorageHealthPass({ locations: [], probe: scripted(["ok"]) });
+ assert.equal(locationHealth("cold"), undefined);
+ });
+});
+
+test("a probe that throws is 'could not ask': ok, never stalled", async () => {
+ await withTmp(async (h) => {
+ const r = await runStorageHealthPass({
+ io: h.io,
+ probe: async () => {
+ throw new Error("spawn failed");
+ },
+ });
+ assert.deepEqual(r.answers, { cold: "ok" });
+ });
+});
+
+test("the health pass is armed on its own, runs once at once, and stops; the five-minute watch arms no health pass", async () => {
+ await withTmp(async (h) => {
+ let asked = 0;
+ const armed = startStorageHealthWatch({
+ io: h.io,
+ probe: async () => {
+ asked += 1;
+ return "stalled";
+ },
+ log: () => {},
+ });
+ try {
+ assert.equal(armed, true);
+ // Armed once per process.
+ assert.equal(startStorageHealthWatch({ io: h.io }), false);
+ for (let i = 0; i < 50 && locationHealth("cold")?.state !== "stalled"; i++) {
+ await new Promise((r) => setTimeout(r, 10));
+ }
+ assert.equal(asked, 1);
+ assert.equal(locationHealth("cold")?.state, "stalled");
+ } finally {
+ stopStorageHealthWatch();
+ }
+ assert.equal(startStorageHealthWatch({ io: h.io, probe: async () => "ok" }), true);
+ stopStorageHealthWatch();
+ // The watch (below the idle gate) runs nothing at arm time and asks no drive.
+ resetStorageHealth();
+ assert.equal(startStorageWatch({ paths: h.paths, io: h.io, bins: h.paths, write: false }), true);
+ stopStorageWatch();
+ assert.equal(locationHealth("cold"), undefined);
+ });
+});
+
+test("a Refresh asks one location now: it counts as one answer, and prunes nothing", async () => {
+ await withTmp(async (h) => {
+ const other = { id: "other", label: "Other", root: "/elsewhere", autoRepoint: false };
+ await runStorageHealthPass({
+ locations: [h.io.read().storage.locations[0], other],
+ probe: scripted(["stalled"]),
+ });
+ const cold = h.io.read().storage.locations[0];
+ assert.equal(await refreshLocationHealth(cold, async () => "ok"), "ok");
+ // One clean answer is not two.
+ assert.equal(locationHealth("cold")?.state, "stalled");
+ assert.equal(locationHealth("other")?.state, "stalled");
+ await refreshLocationHealth(cold, async () => "ok");
+ assert.equal(locationHealth("cold")?.state, "ok");
+ assert.equal(locationHealth("other")?.state, "stalled");
+ });
+});
+
+test("the pass registers every location, records a verdict's detector, and a verdict with no answer changes nothing", async () => {
+ await withTmp(async (h) => {
+ const verdicts = [
+ { answer: null, detector: "counters" as const, device: "sdz1" },
+ {
+ answer: "stalled" as const,
+ detector: "counters" as const,
+ device: "sdz1",
+ cause: "its disk (sdz1) had 1 request(s) in flight and completed none in 15 s",
+ },
+ ];
+ let i = 0;
+ const probe = async () => verdicts[i++];
+ const lines: string[] = [];
+ const first = await runStorageHealthPass({ io: h.io, probe, log: (l) => lines.push(l) });
+ // Registered, answering, and the detector named — with no verdict yet.
+ assert.deepEqual(first.answers, {});
+ assert.deepEqual(first.transitions, []);
+ assert.equal(locationHealth("cold")?.state, "ok");
+ assert.equal(locationHealth("cold")?.detector, "counters");
+ const second = await runStorageHealthPass({ io: h.io, probe, log: (l) => lines.push(l) });
+ assert.deepEqual(second.transitions, [{ id: "cold", from: "ok", to: "stalled" }]);
+ assert.match(String(locationHealth("cold")?.cause), /its disk \(sdz1\)/);
+ assert.match(lines.join("\n"), /"cold": drive not answering — its disk \(sdz1\)/);
+ });
+});
+
+test("a counters verdict records its device (for the watchdog); a stat verdict forgets it", async () => {
+ await withTmp(async (h) => {
+ const verdicts = [
+ { answer: null, detector: "counters" as const, device: "sdz1" },
+ { answer: "ok" as const, detector: "stat" as const },
+ ];
+ let i = 0;
+ const probe = async () => verdicts[i++];
+ await runStorageHealthPass({ io: h.io, probe });
+ assert.equal(locationHealth("cold")?.device, "sdz1");
+ await runStorageHealthPass({ io: h.io, probe });
+ assert.equal(locationHealth("cold")?.device, undefined);
+ assert.equal(locationHealth("cold")?.detector, "stat");
+ });
+});
+
+test("a stall auto-pauses after two passes, and says the drive is not answering (not that it is not there)", async () => {
+ await withTmp(async (h) => {
+ await seedRelocated(h, "slow", { targetExists: true });
+ await runStorageHealthPass({ io: h.io, probe: async () => "stalled" });
+ await twoPasses(h);
+ const entry = h.io.read().channelPriority.channels.slow;
+ assert.equal(entry?.tier, "paused");
+ assert.equal(entry?.autoPaused?.cause, "not-answering");
+ const reason = autoPauseReasonOf(h.io.read().channelPriority, "slow");
+ assert.match(String(reason), /drive that is not answering/);
+ assert.match(String(reason), /when the drive answers again/);
+ // The sanitizer keeps the cause; a record without one reads as not there.
+ const kept = sanitizeChannelPriority(h.io.read().channelPriority);
+ assert.equal(kept.channels.slow?.autoPaused?.cause, "not-answering");
+ const old = sanitizeChannelPriority({
+ channels: {
+ a: { tier: "paused", autoPaused: { reason: "storage", since: "", previousTier: "low" } },
+ },
+ });
+ assert.match(String(autoPauseReasonOf(old, "a")), /drive that is not there/);
+ });
+});
diff --git a/common/controller/storageWatch.ts b/common/controller/storageWatch.ts
@@ -14,13 +14,33 @@ import {
resolveFocusSlugs,
restoreAfterMedia,
sanitizeChannelPriority,
+ type AutoPauseCause,
type ChannelPriority,
} from "../lib/channelPriority";
import { LANES } from "../lib/autoQueueTypes";
import { siteChannelIndex } from "../lib/site";
import { inspectChannelMedia } from "../lib/channelMedia";
-import { locationOfDataDir } from "../lib/storageLocations";
-import type { VolumeBins } from "../lib/storageVolumes";
+import {
+ locationOfDataDir,
+ type StorageLocation,
+} from "../lib/storageLocations";
+import {
+ detectLocationHealth,
+ type HealthVerdict,
+ type LocationHealthProbe,
+ type VolumeBins,
+} from "../lib/storageVolumes";
+import {
+ HEALTH_PROBE_INTERVAL_MS,
+ HEALTH_PROBE_TIMEOUT_MS,
+ NOT_ANSWERING,
+ noteLocationDetector,
+ pruneLocationHealth,
+ recordLocationHealth,
+ registerLocationHealth,
+ type HealthTransition,
+ type LocationHealthState,
+} from "../lib/storageHealth";
import { listChannelConfigs } from "./channels";
import { maybeAutoRepoint, probeAllLocations } from "./storageLocations";
@@ -185,6 +205,7 @@ export async function runStorageWatchPass(
const configs = await listChannelConfigs(paths);
let model: ChannelPriority = settings.channelPriority;
+ const pauseCauses = new Map<string, AutoPauseCause>();
for (const { slug, config } of configs) {
const wasAutoPaused = Boolean(model.channels[slug]?.autoPaused);
@@ -208,14 +229,28 @@ export async function runStorageWatchPass(
const loc = locationOfDataDir(dataDir, locations);
const probe = loc ? probes[loc.id] : undefined;
const locationDown = Boolean(loc) && probe?.status !== "available";
- const media = await inspectChannelMedia(paths, slug, config);
+ // FRESH: this pass is the detector, and the page memo is not what it asks.
+ // A location the health probe found not answering reads `stalled` here
+ // without a call, and a stall is `down` like any other.
+ const media = await inspectChannelMedia(paths, slug, config, {
+ fresh: true,
+ });
// `in-transition` is NEVER a reason to pause: a marker means a move is
// running or was interrupted, and the relocate job is precisely the thing
// that would then be refused by the state it created.
+ // A stall is down too, on a location or not (a root typed by hand gets its
+ // `stalled` from the watchdog alone).
const down =
media.status === "in-transition"
? false
- : locationDown || media.status === "unreachable";
+ : locationDown ||
+ media.status === "unreachable" ||
+ media.status === "stalled";
+ // WHICH down it is, for the pause record's words (autoPauseReasonOf).
+ const cause: AutoPauseCause =
+ media.status === "stalled" || probe?.status === "stalled"
+ ? "not-answering"
+ : "not-there";
if (down && !wasAutoPaused) {
// ONE BAD READ IS A SUSPICION, TWO IN A ROW IS A FACT. See rule 3.
@@ -230,7 +265,8 @@ export async function runStorageWatchPass(
continue;
}
const before = model;
- model = autoPauseForMedia(model, slug);
+ model = autoPauseForMedia(model, slug, new Date(), cause);
+ pauseCauses.set(slug, cause);
// autoPauseForMedia no-ops on a channel the OPERATOR already paused —
// which is right, and means "nothing changed" is a normal outcome here.
if (model !== before) {
@@ -286,7 +322,9 @@ export async function runStorageWatchPass(
)
: stored;
let merged: ChannelPriority = base;
- for (const slug of out.paused) merged = autoPauseForMedia(merged, slug);
+ for (const slug of out.paused) {
+ merged = autoPauseForMedia(merged, slug, new Date(), pauseCauses.get(slug));
+ }
for (const slug of out.restored) merged = restoreAfterMedia(merged, slug);
merged = sanitizeChannelPriority(merged);
// AND THE TREES, in the same write. Two writes to one settings file race each
@@ -308,23 +346,177 @@ export async function runStorageWatchPass(
return out;
}
-// ---------------------------------------------------------------------------
-// The cadence
-// ---------------------------------------------------------------------------
-
// Five minutes. A drive does not come and go on a timescale a person would
// notice faster than that, and every pass is one findmnt per location plus two
// stats per relocated channel — cheap, but not free, and this runs for the life
// of the process.
export const STORAGE_WATCH_INTERVAL_MS = 5 * 60_000;
+// ---------------------------------------------------------------------------
+// The health pass: is each location's drive ANSWERING
+// ---------------------------------------------------------------------------
+//
+// A SECOND CADENCE, AND A MUCH SHORTER ONE. The pass above asks "is the disk
+// here" every five minutes and pauses on two misses, which is right for a
+// cable pulled out. It is no help for a drive that is here and stalled — an
+// SMR disk in a USB enclosure resetting under a long write — because every
+// in-process call on that drive waits for it, and four waits stop the editor
+// answering at all. So every 15 s this reads each location's block device
+// counters in /sys (`detectLocationHealth`; a child `stat` of the root only
+// where no device can be named) and records the answer in `lib/storageHealth.ts`,
+// which every page and poll consults before it touches a drive. One `stalled`
+// answer marks a location at once; two clean answers in a row clear it (the
+// rules are that module's).
+//
+// READ-ONLY AND IN MEMORY. It writes no settings and pauses nothing: the pass
+// above sees a stalled location as down (its probe answers `stalled` without
+// asking) and pauses on its own cadence. So, like the boot probe, it is armed
+// ABOVE the idle gate (`startStorageHealthWatch`, from instrumentation): an idle
+// boot has no five-minute pass, but it has the health pass — without it nothing
+// registers the locations, and a stall the watchdog marks is never cleared.
+
+export type StorageHealthPassOpts = {
+ // Default: the configured locations, read from settings.
+ locations?: readonly StorageLocation[];
+ io?: { read: () => SiteSettings };
+ // findmnt, for naming each root's block device. Default: getPaths().
+ bins?: Pick<VolumeBins, "findmntBin">;
+ // Test seam. Default: `detectLocationHealth` — the block device's counters,
+ // or a child `stat` against a 3 s timer when no device can be named.
+ probe?: LocationHealthProbe;
+ now?: () => number;
+ log?: (line: string) => void;
+};
+
+export type StorageHealthPassResult = {
+ probed: number;
+ answers: Record<string, LocationHealthState>;
+ // Only the locations whose state changed.
+ transitions: HealthTransition[];
+};
+
+// A probe's answer as a verdict. A bare state names no detector; a probe that
+// threw is "could not ask": `ok`, never `stalled`.
+function asVerdict(answer: LocationHealthState | HealthVerdict): HealthVerdict {
+ return typeof answer === "string" ? { answer } : answer;
+}
+
+const STAT_CAUSE = `a stat of its root did not answer within ${HEALTH_PROBE_TIMEOUT_MS / 1000} s`;
+
+// Record one verdict. A verdict with no answer records nothing but the
+// detector that gave it.
+function recordVerdict(
+ loc: StorageLocation,
+ verdict: HealthVerdict,
+ now: number,
+): HealthTransition | null {
+ // The counters' device, for the watchdog's slow-or-stalled check; a stat
+ // verdict forgets it.
+ const device =
+ verdict.detector === "counters"
+ ? verdict.device
+ : verdict.detector === "stat"
+ ? null
+ : undefined;
+ if (verdict.answer === null) {
+ if (verdict.detector) noteLocationDetector(loc.id, verdict.detector, device);
+ return null;
+ }
+ return recordLocationHealth(loc, verdict.answer, {
+ now,
+ cause: verdict.cause ?? STAT_CAUSE,
+ ...(verdict.detector ? { detector: verdict.detector } : {}),
+ ...(device !== undefined ? { device } : {}),
+ });
+}
+
+export async function runStorageHealthPass(
+ opts: StorageHealthPassOpts = {},
+): Promise<StorageHealthPassResult> {
+ const log = opts.log ?? (() => {});
+ const locations =
+ opts.locations ?? (opts.io ?? DEFAULT_IO).read().storage.locations;
+ pruneLocationHealth(locations.map((l) => l.id));
+ // Every configured location has an entry before anything is asked, so the
+ // watchdog (lib/storageHealth.ts `onDrive`) can find a channel's location
+ // even before the counters have given a first verdict.
+ registerLocationHealth(locations);
+ const bins = opts.bins ?? getPaths();
+ const probe: LocationHealthProbe =
+ opts.probe ?? ((loc) => detectLocationHealth(loc, bins));
+ // Every location at once: each answer is bounded by the probe's own timer,
+ // so the pass is too, and one stalled drive does not delay the others.
+ const answers = await Promise.all(
+ locations.map(async (loc) => {
+ const verdict = await probe(loc).then(asVerdict, (): HealthVerdict => ({
+ answer: "ok",
+ }));
+ return [loc, verdict] as const;
+ }),
+ );
+ const out: StorageHealthPassResult = {
+ probed: answers.length,
+ answers: {},
+ transitions: [],
+ };
+ const now = opts.now?.() ?? Date.now();
+ for (const [loc, verdict] of answers) {
+ if (verdict.answer !== null) out.answers[loc.id] = verdict.answer;
+ const t = recordVerdict(loc, verdict, now);
+ if (!t) continue;
+ out.transitions.push(t);
+ if (t.to === "stalled") {
+ log(
+ `[storage] "${loc.id}": ${NOT_ANSWERING} — ${verdict.cause ?? STAT_CAUSE}; ` +
+ `pages and polls skip it until two passes in a row find it answering`,
+ );
+ } else if (t.from === "stalled") {
+ log(`[storage] "${loc.id}": answering again (${t.to})`);
+ }
+ }
+ return out;
+}
+
+// ONE LOCATION, NOW: what /storage's Refresh asks before its own probe, so the
+// operator pressing it after doing something about the drive gets an answer
+// taken afterwards. It counts as one answer like any other — a stalled location
+// still needs two clean ones in a row — and the counters give none when their
+// last sample is under MIN_COUNTER_INTERVAL_MS old. Nothing is pruned.
+export async function refreshLocationHealth(
+ loc: StorageLocation,
+ probe?: LocationHealthProbe,
+): Promise<LocationHealthState | null> {
+ registerLocationHealth([loc]);
+ const ask: LocationHealthProbe =
+ probe ?? ((l) => detectLocationHealth(l, getPaths()));
+ const verdict = await ask(loc).then(asVerdict, (): HealthVerdict => ({
+ answer: "ok",
+ }));
+ recordVerdict(loc, verdict, Date.now());
+ return verdict.answer;
+}
+
+// ---------------------------------------------------------------------------
+// The cadence
+// ---------------------------------------------------------------------------
+
+// Fifteen seconds (lib/storageHealth.ts says why). Each pass reads each
+// location's device counters in /sys (a findmnt only when the device is not
+// known yet or its /sys entry stopped reading; with no device, one short-lived
+// `stat`).
+export const STORAGE_HEALTH_INTERVAL_MS = HEALTH_PROBE_INTERVAL_MS;
+
// A per-module-copy singleton, deliberately left so: it is not a temp-file
// name (slice W folded every tmp + rename onto lib/jsonFile-server.ts, whose
// state is on globalThis), and its one caller is editor/instrumentation.ts,
-// so only one copy ever arms it.
+// so only one copy ever arms it. (The health STATE the second timer writes is
+// on globalThis — pages in another module copy read it.)
let timer: ReturnType<typeof setInterval> | null = null;
+let healthTimer: ReturnType<typeof setInterval> | null = null;
+let healthInFlight = false;
-// ARMED ONCE PER PROCESS. `unref()` so it never holds the event loop open — a
+// THE FIVE-MINUTE PASS, ARMED ONCE PER PROCESS, below the idle gate: it writes
+// settings (an auto-pause). `unref()` so it never holds the event loop open — a
// CLI that imports a controller must still exit.
export function startStorageWatch(
opts: StorageWatchOpts & { intervalMs?: number } = {},
@@ -343,7 +535,51 @@ export function startStorageWatch(
}
export function stopStorageWatch(): void {
- if (!timer) return;
- clearInterval(timer);
+ if (timer) clearInterval(timer);
timer = null;
}
+
+// THE FIFTEEN-SECOND HEALTH PASS, ARMED ONCE PER PROCESS, above the idle gate:
+// it is in memory and writes nothing (see the section header). It also runs
+// once at once, so the locations are registered and a drive that is already
+// stalled is on its way to being known before the first page.
+export function startStorageHealthWatch(
+ opts: {
+ io?: { read: () => SiteSettings };
+ bins?: Pick<VolumeBins, "findmntBin">;
+ probe?: LocationHealthProbe;
+ intervalMs?: number;
+ log?: (line: string) => void;
+ } = {},
+): boolean {
+ if (healthTimer) return false;
+ const health = () => {
+ // One at a time: a pass is bounded by its timers, but a pass that overran
+ // the interval must not stack a second one on top of it.
+ if (healthInFlight) return;
+ healthInFlight = true;
+ void runStorageHealthPass({
+ io: opts.io,
+ bins: opts.bins,
+ probe: opts.probe,
+ log: opts.log,
+ })
+ .catch((err) => {
+ (opts.log ?? console.warn)(
+ `[storage] health pass failed: ${(err as Error).message}`,
+ );
+ })
+ .finally(() => {
+ healthInFlight = false;
+ });
+ };
+ healthTimer = setInterval(health, opts.intervalMs ?? STORAGE_HEALTH_INTERVAL_MS);
+ healthTimer.unref?.();
+ health();
+ return true;
+}
+
+export function stopStorageHealthWatch(): void {
+ if (healthTimer) clearInterval(healthTimer);
+ healthTimer = null;
+}
diff --git a/common/lib/channelMedia.ts b/common/lib/channelMedia.ts
@@ -2,6 +2,15 @@ import path from "node:path";
import { lstat, readFile, readlink, rm, stat } from "node:fs/promises";
import type { Paths } from "./paths";
import type { ChannelConfig } from "./channelConfig";
+import {
+ DRIVE_CALL_BUDGET_MS,
+ NOT_ANSWERING,
+ isDriveNotAnswering,
+ onDrive,
+ sinceText,
+ stalledLocationForPath,
+ type LocationHealth,
+} from "./storageHealth";
// WHERE A CHANNEL'S MEDIA ACTUALLY IS, and whether it can be reached.
//
@@ -60,7 +69,12 @@ export type ChannelMediaStatus =
// A relocation is in flight (or was interrupted): the marker is present.
| "in-transition"
// Disk and config disagree, in either direction. Never guessed past.
- | "inconsistent";
+ | "inconsistent"
+ // Relocated onto a storage location whose drive is not answering
+ // (`lib/storageHealth.ts`). Answered from memory, WITHOUT a filesystem call:
+ // a call there would block one of the process's few I/O threads for as long
+ // as the drive takes to come back. Held and refused like `unreachable`.
+ | "stalled";
export type ChannelMediaLocation = {
// Always channelDir/data — the path every reader uses, relocated or not.
@@ -162,6 +176,7 @@ export async function clearRelocationMarker(
slug: string,
): Promise<void> {
await rm(relocationMarkerPath(paths, slug), { force: true });
+ forgetChannelMedia(slug);
}
// The `dataDir` field alone, read straight off config.json. Deliberately NOT
@@ -185,13 +200,124 @@ async function readConfiguredDataDir(
}
}
+// The configured target's drive is not answering: the answer, from memory. The
+// detail names the location and when it stopped, never a path — the badge's
+// title shows it, and /storage has the paths.
+export function stalledMediaLocation(
+ dataDir: string,
+ configured: string,
+ // Null when the drive is on no location the health state knows (a root typed
+ // by hand) and the watchdog found it not answering, or when the watchdog
+ // found it slow rather than stalled.
+ health: LocationHealth | null,
+ // The watchdog's own words, for a refusal that marked no location.
+ detail?: string,
+): ChannelMediaLocation {
+ return {
+ dataDir,
+ relocated: true,
+ target: configured,
+ status: "stalled",
+ detail: health
+ ? `${NOT_ANSWERING} (location "${health.label}", ${sinceText(health.since)})`
+ : (detail ??
+ `${NOT_ANSWERING} (a read did not answer within ${DRIVE_CALL_BUDGET_MS / 1000} s)`),
+ };
+}
+
+// THE STALL, for a caller holding a parsed config: the stalled location the
+// channel's media is on, or null. No I/O — the question every page and poll
+// that reads a channel's `data/` asks before it does.
+export function channelMediaStall(
+ config: Pick<ChannelConfig, "dataDir"> | null | undefined,
+): LocationHealth | null {
+ const dir = config?.dataDir?.trim();
+ return dir ? stalledLocationForPath(dir) : null;
+}
+
+// ---------------------------------------------------------------------------
+// The memo
+// ---------------------------------------------------------------------------
+//
+// FIVE SECONDS, PER CHANNEL, KEYED BY SLUG AND THE CONFIGURED TARGET. The home
+// page, /channels and the auto-queue status poll (every three seconds, four
+// lanes) each inspect every channel; without this each of them costs three
+// syscalls a relocated channel, every time, on a drive that may be the slow
+// one. The key carries the configured `dataDir`, so a move that rewrites it is
+// a new key at once, and the movers clear the memo outright
+// (`forgetChannelMedia`) whenever a marker is written or removed.
+//
+// THE STALL GATE IS ASKED BEFORE A REMEMBERED ANSWER IS GIVEN, so a drive that
+// stops answering is seen on the next call even when an `ok` from four seconds
+// ago is remembered. A remembered `in-transition` is given as it is: the
+// marker is asked before the gate (below), and the movers forget the channel
+// whenever they write or remove it.
+//
+// `fresh: true` BYPASSES IT, and every caller that decides something from the
+// answer passes it: the start-of-work guard (`assertChannelMediaReachable`),
+// the movers, the index and stats builds, the storage watch, eviction and the
+// re-point preflight. The memo is for pages and polls.
+//
+// ONE MAP PER PROCESS (on `globalThis`), for the reason `storageHealth.ts`
+// gives: the movers that clear it run in one bundle layer and the pages that
+// read it in another.
+
+export const CHANNEL_MEDIA_MEMO_MS = 5_000;
+
+export type InspectOptions = {
+ // Skip the memo: take a fresh answer, and do not remember it.
+ fresh?: boolean;
+ // Test seam for the clock.
+ now?: number;
+};
+
+type MediaMemo = Map<string, { at: number; location: ChannelMediaLocation }>;
+
+declare global {
+ // eslint-disable-next-line no-var
+ var __yttChannelMediaMemo__: MediaMemo | undefined;
+}
+
+function mediaMemo(): MediaMemo {
+ if (!globalThis.__yttChannelMediaMemo__) {
+ globalThis.__yttChannelMediaMemo__ = new Map();
+ }
+ return globalThis.__yttChannelMediaMemo__;
+}
+
+// Forget what the memo holds: for one channel, or for every channel. The
+// movers call it whenever the disk changes under a channel (a marker written or
+// removed, a link swapped, a location re-pointed), so the next page sees it.
+export function forgetChannelMedia(slug?: string): void {
+ const memo = mediaMemo();
+ if (slug === undefined) {
+ memo.clear();
+ return;
+ }
+ for (const key of [...memo.keys()]) {
+ if (key.split("\u0000")[1] === slug) memo.delete(key);
+ }
+}
+
+function memoKey(paths: ChannelMediaPaths, slug: string, configured?: string): string {
+ return `${paths.channelsDir}\u0000${slug}\u0000${configured ?? ""}`;
+}
+
+// Past this many entries, expired ones are swept on insert. A corpus has tens
+// of channels; this only matters to a process that inspects many corpora.
+const MEMO_SWEEP_AT = 512;
+
// Two stats and (at most) one small JSON read. Render-safe: nothing here walks a
// directory, so calling it per channel on a listing page costs three syscalls a
-// row.
+// row — or none, for five seconds after the last answer (see the memo above).
+// On a stalled location the one call that reaches the drive (the target's
+// `stat`) is not made; the marker, the link and config.json are on the corpus
+// disk and are read as usual.
export async function inspectChannelMedia(
paths: ChannelMediaPaths,
slug: string,
config?: Pick<ChannelConfig, "dataDir"> | null,
+ opts: InspectOptions = {},
): Promise<ChannelMediaLocation> {
const dataDir = channelMediaDir(paths, slug);
const configured =
@@ -201,6 +327,42 @@ export async function inspectChannelMedia(
? config.dataDir.trim()
: undefined;
+ const now = opts.now ?? Date.now();
+ const key = memoKey(paths, slug, configured);
+ const memo = mediaMemo();
+ if (!opts.fresh) {
+ const hit = memo.get(key);
+ if (hit && now - hit.at < CHANNEL_MEDIA_MEMO_MS) {
+ if (hit.location.status !== "in-transition" && configured) {
+ const stall = stalledLocationForPath(configured);
+ if (stall) return stalledMediaLocation(dataDir, configured, stall);
+ }
+ return { ...hit.location };
+ }
+ }
+ const location = await inspectOnDisk(paths, slug, dataDir, configured);
+ // A stall is not remembered: the health state is already its memory.
+ if (!opts.fresh && location.status !== "stalled") {
+ if (memo.size >= MEMO_SWEEP_AT) {
+ for (const [k, v] of memo) {
+ if (now - v.at >= CHANNEL_MEDIA_MEMO_MS) memo.delete(k);
+ }
+ }
+ memo.set(key, { at: now, location });
+ }
+ return { ...location };
+}
+
+async function inspectOnDisk(
+ paths: ChannelMediaPaths,
+ slug: string,
+ dataDir: string,
+ configured: string | undefined,
+): Promise<ChannelMediaLocation> {
+ // THE MARKER FIRST. It is in the channel dir, on the corpus disk, so reading
+ // it costs the drive nothing — and a channel mid-move reads `in-transition`
+ // whatever its drive is doing, which is what a resumed move and every guard
+ // key off.
const marker = await readRelocationMarker(paths, slug);
if (marker) {
return {
@@ -215,6 +377,15 @@ export async function inspectChannelMedia(
};
}
+ // THE GATE: a channel whose configured target is on a location whose drive
+ // is not answering is answered from memory, before the link is looked at and
+ // before the target's stat, which would hold an I/O thread for as long as the
+ // drive takes.
+ if (configured) {
+ const stall = stalledLocationForPath(configured);
+ if (stall) return stalledMediaLocation(dataDir, configured, stall);
+ }
+
let link: Awaited<ReturnType<typeof lstat>> | null = null;
try {
link = await lstat(dataDir);
@@ -266,8 +437,12 @@ export async function inspectChannelMedia(
// The link points at a DEEP path (<root>/<slug>/data), so an unmounted root
// gives ENOENT here. An empty mountpoint can never be mistaken for the
// media, which is the whole reason the suffix is fixed.
+ //
+ // THE ONE CALL HERE THAT REACHES THE DRIVE, so it goes through the
+ // watchdog: not made while the location is stalled, and a stat that has
+ // not answered in 3 s marks it stalled and answers `stalled` now.
try {
- const st = await stat(configured);
+ const st = await onDrive(configured, () => stat(configured));
if (!st.isDirectory()) {
return {
dataDir,
@@ -277,7 +452,10 @@ export async function inspectChannelMedia(
detail: `${configured} exists but is not a directory`,
};
}
- } catch {
+ } catch (err) {
+ if (isDriveNotAnswering(err)) {
+ return stalledMediaLocation(dataDir, configured, err.health, err.message);
+ }
return {
dataDir,
relocated: true,
@@ -316,6 +494,10 @@ export async function inspectChannelMedia(
// "ok" and "in-place" pass; everything else throws. An in-transition or
// inconsistent channel is refused for the same reason an unreachable one is:
// the caller would otherwise read a half-populated or empty dir as the truth.
+// A stalled one is refused because the work would block on the drive.
+//
+// ALWAYS FRESH: this is the start-of-work guard, and a remembered "ok" from a
+// few seconds ago is not what a job about to read `data/` should be told.
//
// THIS CHECK HAS A TWIN. `checkChannelReachable` in
// `umtool/report-to-video/cues.mjs` repeats the same statuses in plain `.mjs`,
@@ -329,7 +511,9 @@ export async function assertChannelMediaReachable(
slug: string,
config?: Pick<ChannelConfig, "dataDir"> | null,
): Promise<ChannelMediaLocation> {
- const location = await inspectChannelMedia(paths, slug, config);
+ const location = await inspectChannelMedia(paths, slug, config, {
+ fresh: true,
+ });
if (location.status === "ok" || location.status === "in-place") {
return location;
}
diff --git a/common/lib/channelMediaHold.ts b/common/lib/channelMediaHold.ts
@@ -33,6 +33,7 @@ export const HELD_REASON: Record<ChannelMediaStatus, string> = {
unreachable: "its media is not reachable (drive not mounted?)",
"in-transition": "a move of its media is in progress or was interrupted",
inconsistent: "its data link and its config disagree",
+ stalled: "its drive is not answering (a stalled disk)",
ok: "reachable",
"in-place": "reachable",
};
diff --git a/common/lib/channelPriority.ts b/common/lib/channelPriority.ts
@@ -183,8 +183,15 @@ export type ChannelAutoPause = {
reason: "storage";
since: string;
previousTier: StoredChannelTier;
+ cause?: AutoPauseCause;
};
+// Which storage trouble paused it: the drive is not there (unmounted,
+// unplugged), or it is there and not answering (lib/storageHealth.ts). A record
+// with none was written before the second existed, when the first was the only
+// one.
+export type AutoPauseCause = "not-there" | "not-answering";
+
export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = {
reason:
"One reason today. A union so a second one has somewhere to go, and so " +
@@ -195,6 +202,11 @@ export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = {
"The base tier the channel had before the machine paused it; what a " +
"restore puts back. Never `paused` (that would restore to paused — a " +
"no-op dressed as a restore).",
+ cause:
+ "`not-there` (the drive is unmounted or unplugged) or `not-answering` (it " +
+ "is there and does not answer: a stalled disk). Only what the words on " +
+ "/review, the rack and the channel page say. Optional: a record written " +
+ "before it existed reads as `not-there`.",
};
// Each field is documented in CHANNEL_PRIORITY_FIELD_DOCS below (rendered into SETTINGS.md).
@@ -387,12 +399,16 @@ function sanitizeAutoPause(
reason: "storage",
since: typeof r.since === "string" ? r.since : "",
previousTier: previousTier === "paused" ? DEFAULT_CHANNEL_TIER : previousTier,
+ ...(r.cause === "not-there" || r.cause === "not-answering"
+ ? { cause: r.cause }
+ : {}),
};
}
// --- Auto-pause: the machine's own pause, and its undo ----------------------
-// PAUSE A CHANNEL BECAUSE ITS DRIVE IS NOT THERE, recording what to put back.
+// PAUSE A CHANNEL BECAUSE ITS DRIVE IS NOT THERE (OR NOT ANSWERING), recording
+// what to put back, and which of the two it was.
//
// A NO-OP IN TWO CASES, and both matter. Already auto-paused: a flapping drive
// must not overwrite `previousTier` with the `paused` it wrote last time, which
@@ -403,6 +419,7 @@ export function autoPauseForMedia(
model: ChannelPriority,
slug: string,
now: Date = new Date(),
+ cause?: AutoPauseCause,
): ChannelPriority {
const key = slug.trim();
if (!key) return model;
@@ -421,6 +438,7 @@ export function autoPauseForMedia(
reason: "storage",
since: now.toISOString(),
previousTier,
+ ...(cause ? { cause } : {}),
},
},
},
@@ -466,6 +484,12 @@ export function autoPauseReasonOf(
const auto = model.channels[slug]?.autoPaused;
if (!auto) return null;
const since = auto.since ? ` since ${auto.since.slice(0, 10)}` : "";
+ if (auto.cause === "not-answering") {
+ return (
+ `Auto-paused — its media is on a drive that is not answering${since}. ` +
+ `It returns to ${auto.previousTier} on its own when the drive answers again.`
+ );
+ }
return (
`Auto-paused — its media is on a drive that is not there${since}. ` +
`It returns to ${auto.previousTier} on its own when the drive is back.`
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -113,6 +113,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "AUDIO_CHECK_BACKOFF_FACTOR", audience: "runtime", default: "the built-in factor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "The audio check's interval backoff factor, in (0, 1], for a one-off run." },
{ name: "ARCHILYZER_STATS_ALLOW_DOWNGRADE", audience: "runtime", default: "off", readBy: "common/controller/buildStats.ts", doc: "`1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions." },
{ name: "ARCHILYZER_INDEX_ALLOW_HELD", audience: "runtime", default: "off", readBy: "common/controller/buildIndex.ts", doc: "`1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel." },
+ { name: "UV_THREADPOOL_SIZE", audience: "runtime", default: "`16` for the editor (`4` is Node's own)", readBy: "Node's libuv (set by editor/package.json `start` and docker/entrypoint.sh)", doc: "Threads in Node's pool for filesystem calls. A call on a stalled drive holds one until the drive answers, so the editor starts with 16. It buys time for calls already in flight and isolates nothing: the storage health probe and its gate keep new calls off a stalled drive." },
{ name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." },
{ name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." },
{ name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },
diff --git a/common/lib/ports.test.ts b/common/lib/ports.test.ts
@@ -23,6 +23,11 @@ import {
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..");
const PACKAGES = ["", "common", "editor", "export", "homepage", "mcp", "umtool", "umtool/report-to-video"];
+// A script's `${NAME:-N}` that is a number and NOT a port, named one by one so
+// every other spelled default is still read as a port: the editor's `start`
+// gives Node's thread pool 16 threads (envVars.ts, UV_THREADPOOL_SIZE).
+const NUMERIC_NOT_PORTS = new Set(["UV_THREADPOOL_SIZE"]);
+
function scripts(dir: string): Record<string, string> {
const file = path.join(ROOT, dir, "package.json");
return (JSON.parse(readFileSync(file, "utf8")).scripts ?? {}) as Record<string, string>;
@@ -63,6 +68,7 @@ function found(): Found[] {
for (const [script, line] of Object.entries(scripts(pkg))) {
const where = `${pkg || "."}/package.json "${script}"`;
for (const m of line.matchAll(/\$\{([A-Z0-9_]+):-(\d+)\}/g)) {
+ if (NUMERIC_NOT_PORTS.has(m[1])) continue;
hits.push({ where, name: m[1], port: Number(m[2]) });
}
for (const m of line.matchAll(/--ports\s+(\S+)/g)) {
diff --git a/common/lib/storageHealth.test.ts b/common/lib/storageHealth.test.ts
@@ -0,0 +1,544 @@
+import { beforeEach, test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ DRIVE_CALLS_IN_FLIGHT,
+ DRIVE_CALL_BUDGET_MS,
+ DriveNotAnsweringError,
+ HEALTH_CLEAN_TO_CLEAR,
+ allLocationHealth,
+ driveCallsInFlight,
+ isDriveNotAnswering,
+ noteLocationDetector,
+ onDrive,
+ registerLocationHealth,
+ setCounterReader,
+ setDriveCallBudget,
+ locationHealth,
+ notAnsweringText,
+ pruneLocationHealth,
+ recordLocationHealth,
+ resetStorageHealth,
+ sinceText,
+ stalledLocation,
+ stalledLocationForPath,
+} from "./storageHealth";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/storageHealth.test.ts
+//
+// The rules the health state keeps (storageHealth.ts's header): one miss stalls
+// at once, two clean answers in a row clear it, a miss in between starts the
+// count again, and a new root starts the location over.
+
+beforeEach(() => {
+ resetStorageHealth();
+ setDriveCallBudget();
+ setCounterReader(undefined);
+});
+
+const USB = { id: "usb", label: "USB drive", root: "/mnt/usb/media" };
+
+test("one missed probe marks the location stalled at once", () => {
+ assert.equal(recordLocationHealth(USB, "ok", { now: 1_000 })?.to, "ok");
+ const t = recordLocationHealth(USB, "stalled", { now: 2_000, cause: "no answer" });
+ assert.deepEqual(t, { id: "usb", from: "ok", to: "stalled" });
+ const h = locationHealth("usb");
+ assert.equal(h?.state, "stalled");
+ assert.equal(h?.since, 2_000);
+ assert.equal(h?.cause, "no answer");
+});
+
+test("a location first seen stalled is stalled", () => {
+ const t = recordLocationHealth(USB, "stalled", { now: 5 });
+ assert.deepEqual(t, { id: "usb", from: null, to: "stalled" });
+ assert.equal(stalledLocation(USB)?.since, 5);
+});
+
+test("one clean probe after a stall does not clear it; two in a row do", () => {
+ assert.equal(HEALTH_CLEAN_TO_CLEAR, 2);
+ recordLocationHealth(USB, "ok", { now: 0 });
+ recordLocationHealth(USB, "stalled", { now: 10 });
+ assert.equal(recordLocationHealth(USB, "ok", { now: 20 }), null);
+ assert.equal(locationHealth("usb")?.state, "stalled");
+ // `since` stays the stall's start while it lasts.
+ assert.equal(locationHealth("usb")?.since, 10);
+ const t = recordLocationHealth(USB, "ok", { now: 30 });
+ assert.deepEqual(t, { id: "usb", from: "stalled", to: "ok" });
+ const h = locationHealth("usb");
+ assert.equal(h?.state, "ok");
+ assert.equal(h?.since, 30);
+ assert.equal(h?.cause, undefined);
+});
+
+test("a miss between two clean probes starts the count again", () => {
+ recordLocationHealth(USB, "stalled", { now: 0 });
+ recordLocationHealth(USB, "ok", { now: 1 });
+ assert.equal(recordLocationHealth(USB, "stalled", { now: 2 }), null);
+ // The stall's start is not moved by a second miss.
+ assert.equal(locationHealth("usb")?.since, 0);
+ recordLocationHealth(USB, "ok", { now: 3 });
+ assert.equal(locationHealth("usb")?.state, "stalled");
+ recordLocationHealth(USB, "ok", { now: 4 });
+ assert.equal(locationHealth("usb")?.state, "ok");
+});
+
+test("an absent answer is a clean one: an unplugged drive answers at once", () => {
+ recordLocationHealth(USB, "stalled", { now: 0 });
+ recordLocationHealth(USB, "absent", { now: 1 });
+ const t = recordLocationHealth(USB, "absent", { now: 2 });
+ assert.deepEqual(t, { id: "usb", from: "stalled", to: "absent" });
+ assert.equal(stalledLocation(USB), null);
+});
+
+test("a re-pointed root starts the location over", () => {
+ recordLocationHealth(USB, "stalled", { now: 0 });
+ const moved = { ...USB, root: "/mnt/elsewhere/media" };
+ // The stall belonged to the old root: the new root's lookup is not stalled,
+ // and neither is the old one's (the entry now describes the new root).
+ assert.equal(stalledLocation(moved), null);
+ const t = recordLocationHealth(moved, "ok", { now: 1 });
+ assert.deepEqual(t, { id: "usb", from: null, to: "ok" });
+ assert.equal(locationHealth("usb")?.root, "/mnt/elsewhere/media");
+ assert.equal(stalledLocation(USB), null);
+});
+
+test("the path gate matches like locationOfDataDir: longest root, strictly under", () => {
+ const parent = { id: "parent", label: "Parent", root: "/mnt/p" };
+ const child = { id: "child", label: "Child", root: "/mnt/p/archive" };
+ recordLocationHealth(parent, "stalled", { now: 0 });
+ recordLocationHealth(child, "ok", { now: 0 });
+ // A channel on the nested location answers for that location, not its parent.
+ assert.equal(stalledLocationForPath("/mnt/p/archive/chan/data"), null);
+ assert.equal(stalledLocationForPath("/mnt/p/chan/data")?.id, "parent");
+ // The root itself is not "under" it, and an unrelated path is on nothing.
+ assert.equal(stalledLocationForPath("/mnt/p"), null);
+ assert.equal(stalledLocationForPath("/elsewhere/chan/data"), null);
+ assert.equal(stalledLocationForPath(""), null);
+});
+
+test("pruning drops locations no longer configured", () => {
+ recordLocationHealth(USB, "stalled", { now: 0 });
+ recordLocationHealth({ id: "b", label: "B", root: "/b" }, "ok", { now: 0 });
+ pruneLocationHealth(["b"]);
+ assert.deepEqual(Object.keys(allLocationHealth()), ["b"]);
+ assert.equal(stalledLocationForPath("/mnt/usb/media/x/data"), null);
+});
+
+test("the state is one map per process, on globalThis", () => {
+ recordLocationHealth(USB, "stalled", { now: 0 });
+ // A second module copy reads the same object (the house pattern).
+ assert.equal(
+ globalThis.__yttStorageHealth__?.byId.get("usb")?.state,
+ "stalled",
+ );
+});
+
+test("the words: since a time today, or a date and time", () => {
+ const now = new Date(2026, 8, 29, 12, 0).getTime();
+ const today = new Date(2026, 8, 29, 11, 35).getTime();
+ const yesterday = new Date(2026, 8, 28, 23, 5).getTime();
+ assert.equal(sinceText(today, now), "since 11:35");
+ assert.equal(sinceText(yesterday, now), "since 2026-09-28 23:05");
+ assert.equal(notAnsweringText({ since: today }, now), "not answering since 11:35");
+});
+
+// ── the watchdog (`onDrive`) ────────────────────────────────────────────────
+// A call that never answers is a promise that never settles: exactly what a
+// read blocked on a stalled drive looks like from here. No drive is involved.
+
+const never = () => new Promise<never>(() => {});
+
+function deferred<T>() {
+ let resolve!: (v: T) => void;
+ let reject!: (e: Error) => void;
+ const promise = new Promise<T>((res, rej) => {
+ resolve = res;
+ reject = rej;
+ });
+ return { promise, resolve, reject };
+}
+
+test("a call that answers in time passes through, value or error, and frees its slot", async () => {
+ registerLocationHealth([USB]);
+ assert.equal(await onDrive("/mnt/usb/media/ch/data", async () => 42), 42);
+ await assert.rejects(
+ () => onDrive(USB, async () => {
+ throw new Error("ENOENT");
+ }),
+ /ENOENT/,
+ );
+ assert.equal(driveCallsInFlight("usb"), 0);
+ assert.equal(locationHealth("usb")?.state, "ok");
+});
+
+test("a call that never answers: stalled on the timer, the location marked at once, the call left to settle", async () => {
+ registerLocationHealth([USB], 0);
+ setDriveCallBudget(80);
+ const late = deferred<string>();
+ const started = Date.now();
+ await assert.rejects(
+ () => onDrive("/mnt/usb/media/ch/data", () => late.promise),
+ (err: unknown) =>
+ err instanceof DriveNotAnsweringError &&
+ isDriveNotAnswering(err) &&
+ err.health?.id === "usb" &&
+ /^drive not answering \(location "USB drive", since /.test(err.message),
+ );
+ assert.ok(Date.now() - started < 1_000);
+ const h = locationHealth("usb");
+ assert.equal(h?.state, "stalled");
+ assert.ok((h?.since ?? 0) >= started, "since is now, not the entry's first sighting");
+ assert.match(String(h?.cause), /a read in the editor did not answer within 0.08 s/);
+ // The slot is held until the call really returns.
+ assert.equal(driveCallsInFlight("usb"), 1);
+ late.resolve("finally");
+ await new Promise((r) => setImmediate(r));
+ assert.equal(driveCallsInFlight("usb"), 0);
+});
+
+test("a stalled location is refused without the call being made", async () => {
+ registerLocationHealth([USB]);
+ recordLocationHealth(USB, "stalled");
+ let made = 0;
+ await assert.rejects(
+ () => onDrive("/mnt/usb/media/ch/data", async () => ++made),
+ DriveNotAnsweringError,
+ );
+ await assert.rejects(() => onDrive(USB, async () => ++made), DriveNotAnsweringError);
+ assert.equal(made, 0);
+});
+
+test("at most four calls in flight on a location; the rest wait, and are refused without a call when it stalls", async () => {
+ assert.equal(DRIVE_CALLS_IN_FLIGHT, 4);
+ registerLocationHealth([USB]);
+ setDriveCallBudget(80);
+ let made = 0;
+ const calls = Array.from({ length: 7 }, () =>
+ onDrive(USB, () => {
+ made += 1;
+ return never();
+ }).then(
+ () => "answered",
+ (err: Error) => err.name,
+ ),
+ );
+ await new Promise((r) => setImmediate(r));
+ assert.equal(made, 4, "four in flight, three waiting");
+ const outcomes = await Promise.all(calls);
+ assert.deepEqual(outcomes, Array(7).fill("DriveNotAnsweringError"));
+ assert.equal(made, 4, "the three that waited were refused without a call");
+ assert.equal(locationHealth("usb")?.state, "stalled");
+});
+
+test("a waiting call runs when a slot frees, if the location is still answering", async () => {
+ registerLocationHealth([USB]);
+ const gates = Array.from({ length: 4 }, () => deferred<number>());
+ const first = gates.map((g) => onDrive(USB, () => g.promise));
+ let fifth = false;
+ const waiting = onDrive(USB, async () => {
+ fifth = true;
+ return 5;
+ });
+ await new Promise((r) => setImmediate(r));
+ assert.equal(fifth, false);
+ gates[0].resolve(1);
+ assert.equal(await waiting, 5);
+ for (const g of gates.slice(1)) g.resolve(0);
+ await Promise.all(first);
+ assert.equal(driveCallsInFlight("usb"), 0);
+});
+
+test("a path on no known location is raced and capped by its root, and names no location", async () => {
+ setDriveCallBudget(80);
+ let made = 0;
+ // Two channels under one hand-typed root share its four slots.
+ const calls = [
+ ...Array.from({ length: 3 }, () => "/hand/typed/chan-a/data"),
+ ...Array.from({ length: 3 }, () => "/hand/typed/chan-b/data"),
+ ].map((p) =>
+ onDrive(p, () => {
+ made += 1;
+ return never();
+ }).then(
+ () => "answered",
+ (err: DriveNotAnsweringError) => (err.health === null ? "refused" : "marked"),
+ ),
+ );
+ await new Promise((r) => setImmediate(r));
+ assert.equal(made, 4, "one root, four slots");
+ assert.deepEqual(await Promise.all(calls), Array(6).fill("refused"));
+ assert.equal(made, 4);
+ assert.deepEqual(allLocationHealth(), {});
+ // Every slot is held by a call given up on: the next is refused at once.
+ const started = Date.now();
+ await assert.rejects(() => onDrive("/hand/typed/chan-c/data", async () => ++made));
+ assert.ok(Date.now() - started < 50);
+ assert.equal(made, 4);
+});
+
+test("a probe of another root under a location's id has its own slots and does not rewrite that location", async () => {
+ registerLocationHealth([USB]);
+ setDriveCallBudget(50);
+ // Four candidate probes that never answer: they hold the CANDIDATE root's
+ // four slots, not the location's.
+ for (let i = 0; i < 4; i++) {
+ await assert.rejects(
+ () => onDrive({ ...USB, root: "/mnt/candidate" }, never),
+ DriveNotAnsweringError,
+ );
+ }
+ assert.equal(locationHealth("usb")?.root, "/mnt/usb/media");
+ assert.equal(locationHealth("usb")?.state, "ok");
+ assert.equal(driveCallsInFlight("usb"), 0);
+ // The real location's calls run as if nothing happened.
+ assert.deepEqual(
+ await Promise.all([1, 2, 3, 4, 5].map((n) => onDrive(USB, async () => n))),
+ [1, 2, 3, 4, 5],
+ );
+});
+
+test("M1: every slot held by a call given up on — a new call is refused within the budget, after the pass cleared the location", async () => {
+ registerLocationHealth([USB]);
+ setDriveCallBudget(80);
+ const hung = Array.from({ length: 4 }, () => onDrive(USB, never).catch(() => "gave up"));
+ assert.deepEqual(await Promise.all(hung), Array(4).fill("gave up"));
+ assert.equal(locationHealth("usb")?.state, "stalled");
+ // Two clean answers from the pass (in a reset loop's good moment).
+ recordLocationHealth(USB, "ok");
+ recordLocationHealth(USB, "ok");
+ assert.equal(locationHealth("usb")?.state, "ok");
+ let made = false;
+ const started = Date.now();
+ await assert.rejects(
+ () => onDrive(USB, async () => {
+ made = true;
+ }),
+ (err: unknown) => err instanceof DriveNotAnsweringError && err.health?.id === "usb",
+ );
+ assert.ok(Date.now() - started < 80, "refused at once, not after a wait");
+ assert.equal(made, false);
+ // And the location is marked again: none of the four has returned.
+ assert.equal(locationHealth("usb")?.state, "stalled");
+ assert.match(String(locationHealth("usb")?.cause), /4 reads on it have not answered/);
+});
+
+test("M1: every transition to stalled refuses the waiting calls at once, whoever decided it", async () => {
+ registerLocationHealth([USB]);
+ setDriveCallBudget(5_000);
+ const gates = Array.from({ length: 4 }, () => deferred<number>());
+ const inFlight = gates.map((g) => onDrive(USB, () => g.promise));
+ const waiting = onDrive(USB, async () => 5).then(
+ () => "ran",
+ (err: Error) => err.name,
+ );
+ await new Promise((r) => setImmediate(r));
+ const started = Date.now();
+ // The pass (not the watchdog) finds the drive stalled.
+ recordLocationHealth(USB, "stalled");
+ assert.equal(await waiting, "DriveNotAnsweringError");
+ assert.ok(Date.now() - started < 1_000);
+ for (const g of gates) g.resolve(0);
+ await Promise.all(inFlight);
+});
+
+test("L6: a call that times out while its disk is still completing requests is slow — refused, the location not marked", async () => {
+ registerLocationHealth([USB]);
+ noteLocationDetector("usb", "counters", "sdz1");
+ let completed = 100;
+ setCounterReader((device) => {
+ assert.equal(device, "sdz1");
+ completed += 7;
+ return { completed, inFlight: 2 };
+ });
+ setDriveCallBudget(60);
+ // Four slow calls and one waiting behind them.
+ const slow = Array.from({ length: 4 }, () =>
+ onDrive(USB, never).then(
+ () => "answered",
+ (err: Error) => err.message,
+ ),
+ );
+ const waiter = onDrive(USB, async () => 1).then(
+ () => "ran",
+ (err: Error) => err.message,
+ );
+ const outcomes = await Promise.all(slow);
+ for (const o of outcomes) assert.match(o, /^drive slow/);
+ assert.equal(locationHealth("usb")?.state, "ok", "slow is not stalled");
+ // Nothing on the drive returns: the waiter's wait runs out — refused, still
+ // without marking.
+ assert.match(await waiter, /nothing on it answered for 0.06 s while a read waited/);
+ assert.equal(locationHealth("usb")?.state, "ok");
+ // With the counters standing still, the same timeout marks it.
+ setCounterReader(() => ({ completed: 500, inFlight: 2 }));
+ resetStorageHealth();
+ registerLocationHealth([USB]);
+ noteLocationDetector("usb", "counters", "sdz1");
+ setCounterReader(() => ({ completed: 500, inFlight: 2 }));
+ await assert.rejects(() => onDrive(USB, never), DriveNotAnsweringError);
+ assert.equal(locationHealth("usb")?.state, "stalled");
+});
+
+test("the default budget is 3 s", async () => {
+ assert.equal(DRIVE_CALL_BUDGET_MS, 3_000);
+ registerLocationHealth([USB]);
+ const started = Date.now();
+ await assert.rejects(() => onDrive(USB, never), DriveNotAnsweringError);
+ const took = Date.now() - started;
+ assert.ok(took >= 3_000 && took < 4_500, `answered after ${took} ms`);
+});
+
+test("registering locations creates entries without an answer, and a moved root starts over", () => {
+ registerLocationHealth([USB], 5);
+ assert.equal(locationHealth("usb")?.state, "ok");
+ recordLocationHealth(USB, "stalled", { now: 6 });
+ registerLocationHealth([USB], 7);
+ assert.equal(locationHealth("usb")?.state, "stalled", "registering is not an answer");
+ registerLocationHealth([{ ...USB, root: "/mnt/new" }], 8);
+ assert.equal(locationHealth("usb")?.state, "ok");
+ assert.equal(locationHealth("usb")?.root, "/mnt/new");
+});
+
+// ── M4: the wait's deadline follows progress ───────────────────────────────
+
+const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms));
+
+test("M4: a healthy 64-wide walk of units at half the budget — no refusals, nothing marked", async () => {
+ registerLocationHealth([USB]);
+ setDriveCallBudget(100);
+ let answered = 0;
+ const started = Date.now();
+ const outcomes = await Promise.all(
+ Array.from({ length: 64 }, () =>
+ onDrive("/mnt/usb/media/ch/data", async () => {
+ await sleep(50);
+ answered += 1;
+ return "ok";
+ }).catch((err: Error) => err.message),
+ ),
+ );
+ // Sixteen rounds of 50 ms: the last call waited about 750 ms, far past the
+ // 100 ms budget, and was not refused — the drive kept answering.
+ assert.ok(Date.now() - started >= 700, `took ${Date.now() - started} ms`);
+ assert.deepEqual(outcomes, Array(64).fill("ok"));
+ assert.equal(answered, 64);
+ assert.equal(locationHealth("usb")?.state, "ok");
+ assert.equal(driveCallsInFlight("usb"), 0);
+});
+
+test("M4: the same deep queue on a hand-typed root — no refusals either", async () => {
+ setDriveCallBudget(100);
+ const outcomes = await Promise.all(
+ Array.from({ length: 32 }, () =>
+ onDrive("/hand/typed/ch/data", async () => {
+ await sleep(50);
+ return "ok";
+ }).catch((err: Error) => err.message),
+ ),
+ );
+ assert.deepEqual(outcomes, Array(32).fill("ok"));
+});
+
+test("M4: a queue behind four hung calls is still refused within the budget", async () => {
+ setDriveCallBudget(100);
+ // A hand-typed root: nothing marks it, so the refusal is the timeouts'.
+ const hung = Array.from({ length: 4 }, () =>
+ onDrive("/hand/typed/ch/data", never).catch(() => "gave up"),
+ );
+ const started = Date.now();
+ const waiters = Array.from({ length: 6 }, () =>
+ onDrive("/hand/typed/ch/data", async () => "ran").catch((err: Error) => err.name),
+ );
+ assert.deepEqual(await Promise.all(waiters), Array(6).fill("DriveNotAnsweringError"));
+ assert.ok(Date.now() - started < 400, `refused after ${Date.now() - started} ms`);
+ await Promise.all(hung);
+ // And on a configured location, the timeouts mark it and refuse the queue.
+ registerLocationHealth([USB]);
+ const hungHere = Array.from({ length: 4 }, () => onDrive(USB, never).catch(() => "gave up"));
+ const t0 = Date.now();
+ const waitersHere = Array.from({ length: 6 }, () =>
+ onDrive(USB, async () => "ran").catch((err: Error) => err.name),
+ );
+ assert.deepEqual(await Promise.all(waitersHere), Array(6).fill("DriveNotAnsweringError"));
+ assert.ok(Date.now() - t0 < 400);
+ assert.equal(locationHealth("usb")?.state, "stalled");
+ await Promise.all(hungHere);
+});
+
+test("M4: a call that returns late frees its slot for the calls waiting behind it", async () => {
+ registerLocationHealth([USB]);
+ noteLocationDetector("usb", "counters", "sdz1");
+ // The counters keep completing, so the late calls are slow, not stalled.
+ let completed = 0;
+ setCounterReader(() => ({ completed: (completed += 5), inFlight: 3 }));
+ setDriveCallBudget(80);
+ // Four calls that take 150 ms: refused as slow at 80 ms, returning at 150.
+ const slow = Array.from({ length: 4 }, () =>
+ onDrive(USB, () => sleep(150)).then(
+ () => "answered",
+ (err: Error) => err.message,
+ ),
+ );
+ // Four more queue at 70 ms, deadline 70 + 80 + 20: the returns at 150 ms
+ // come first and hand them the slots.
+ await sleep(70);
+ const behind = Array.from({ length: 4 }, () =>
+ onDrive(USB, async () => "ran").catch((err: Error) => err.message),
+ );
+ for (const o of await Promise.all(slow)) assert.match(o, /^drive slow/);
+ assert.deepEqual(await Promise.all(behind), Array(4).fill("ran"));
+ assert.equal(locationHealth("usb")?.state, "ok");
+ // The first return can feed all four waiters; let the other three slow
+ // calls return too, so their releases do not land in the next test's state.
+ await sleep(120);
+ assert.equal(driveCallsInFlight("usb"), 0);
+});
+
+// ── the overdue refusal: slow or stalled, by the counters ──────────────────
+
+test("every slot held by a slow unit on a disk still completing: the next call is refused, nothing marked", async () => {
+ registerLocationHealth([USB]);
+ noteLocationDetector("usb", "counters", "sdz1");
+ let completed = 0;
+ setCounterReader(() => ({ completed: (completed += 3), inFlight: 4 }));
+ setDriveCallBudget(60);
+ // Four units slower than the budget (they return long after): each is
+ // refused as slow at its timeout, and holds its slot, overdue.
+ const slow = Array.from({ length: 4 }, () =>
+ onDrive(USB, never).catch((err: Error) => err.message),
+ );
+ for (const o of await Promise.all(slow)) assert.match(o, /^drive slow/);
+ let made = false;
+ const started = Date.now();
+ await assert.rejects(
+ () => onDrive(USB, async () => {
+ made = true;
+ }),
+ (err: unknown) =>
+ err instanceof DriveNotAnsweringError &&
+ err.health === null &&
+ /^drive slow \(4 reads on it are past 0.06 s/.test(err.message),
+ );
+ assert.ok(Date.now() - started < 60, "refused at once");
+ assert.equal(made, false);
+ assert.equal(locationHealth("usb")?.state, "ok", "slow, not stalled");
+});
+
+test("every slot held by an overdue call on a disk completing nothing: the next call is refused and marks it", async () => {
+ registerLocationHealth([USB]);
+ noteLocationDetector("usb", "counters", "sdz1");
+ setCounterReader(() => ({ completed: 500, inFlight: 4 }));
+ setDriveCallBudget(60);
+ const hung = Array.from({ length: 4 }, () => onDrive(USB, never).catch(() => "gave up"));
+ await Promise.all(hung);
+ assert.equal(locationHealth("usb")?.state, "stalled");
+ // The pass clears it; the four have still not returned.
+ recordLocationHealth(USB, "ok");
+ recordLocationHealth(USB, "ok");
+ await assert.rejects(
+ () => onDrive(USB, async () => 1),
+ (err: unknown) => err instanceof DriveNotAnsweringError && err.health?.id === "usb",
+ );
+ assert.equal(locationHealth("usb")?.state, "stalled");
+ assert.match(String(locationHealth("usb")?.cause), /4 reads on it have not answered/);
+});
diff --git a/common/lib/storageHealth.ts b/common/lib/storageHealth.ts
@@ -0,0 +1,749 @@
+import { locationOfDataDir, type StorageLocation } from "./storageLocations";
+
+// IS A STORAGE LOCATION'S DRIVE ANSWERING RIGHT NOW — the in-memory answer every
+// page and poll asks before it touches the drive.
+//
+// A drive can be mounted and still not answer. An SMR disk in a USB enclosure
+// under a long write stalls, the enclosure resets, and every filesystem call
+// that has to reach the disk blocks for about 30 seconds. Node runs those calls
+// on libuv's thread pool (four threads by default), so four of them block the
+// whole editor: no page, no poll and no job log answers until the disk does.
+// "Not mounted" does not describe that (a `stat` there does not fail, it hangs),
+// and no in-process call can find it out without paying the hang itself.
+//
+// SO SOMETHING THAT CANNOT HANG ASKS, AND THIS MODULE REMEMBERS WHAT IT SAID.
+// Two detectors write here. The health pass (`controller/storageWatch.ts`,
+// every 15 s) reads each location's block device counters in /sys, which never
+// touch the drive (`detectLocationHealth` in `storageVolumes.ts`; a child `stat`
+// of the root raced against 3 s only where no device can be named). And
+// `onDrive` below races every in-process call the gate covers against 3 s and
+// marks the location the moment one does not answer. Everything that would
+// touch the drive in-process asks this state first and, on a stalled location,
+// answers without the call: `inspectChannelMedia` reports `stalled`, the
+// free-space column reads "—", the probe reads "Not answering", the recency
+// layer skips the tail read.
+//
+// THE RULES, which are what keep a flaky drive from flapping the UI:
+// - ONE `stalled` answer marks the location `stalled` at once. A drive that
+// did not answer will not answer the next page either, and every page that
+// asks costs a thread.
+// - TWO consecutive clean answers clear it. A clean answer is anything else:
+// the counters moving or idle, or a child `stat` answering in time whether
+// the root was there (`ok`) or not (`absent`: an unmounted drive answers
+// ENOENT at once, and that is a different problem, which
+// `inspectChannelMedia` already reports). One clean answer in the middle of
+// a reset loop is not recovery.
+// - A ROOT CHANGE (a re-point) starts the location over: the old root's stall
+// says nothing about the new one.
+//
+// IN MEMORY ONLY. A stall is a fact about this minute, not about the corpus;
+// persisting it would outlive the reset loop that caused it. A restarted
+// process starts with no stall, and the first pass (at arm time, on an idle
+// boot too) registers the locations; the watchdog re-learns a stall the moment
+// a page reaches the drive.
+//
+// ONE MAP PER PROCESS, NOT PER MODULE COPY. The watch that probes is armed from
+// `editor/instrumentation.ts`, and the pages that read are another bundle
+// layer; Next can load this module once for each. The map lives on
+// `globalThis`, the house pattern (`lib/jsonFile-server.ts`, `jobs/registry.ts`),
+// so the writer and every reader see one map.
+//
+// PURE of I/O, and deliberately without execa, so `lib/channelMedia.ts` can ask
+// it without pulling a subprocess module into everything that imports that.
+
+// One probe's answer, and a location's state.
+export type LocationHealthState =
+ // The root answered in time and is a directory.
+ | "ok"
+ // The root did not answer within the probe's budget.
+ | "stalled"
+ // The root answered in time, and is not a directory (not mounted, or gone).
+ | "absent";
+
+// What decided a location's state. `counters`: the block device's own request
+// counters (`/sys/class/block/<dev>/stat`), which never touch the drive;
+// `stat`: a child `stat` of the root, when no device could be named (a
+// container, no findmnt, no /sys entry). A stall marked by `onDrive`'s watchdog
+// keeps the detector the last pass used; its `cause` says what did not answer.
+export type HealthDetector = "counters" | "stat";
+
+export type LocationHealth = {
+ id: string;
+ label: string;
+ root: string;
+ state: LocationHealthState;
+ detector?: HealthDetector;
+ // When the current state began, ms since epoch.
+ since: number;
+ // When the last probe (or observation) was recorded.
+ checkedAt: number;
+ // Consecutive clean answers since the location was marked stalled. It clears
+ // at HEALTH_CLEAN_TO_CLEAR.
+ cleanStreak: number;
+ // What did not answer, for a stalled location: the probe's own words.
+ cause?: string;
+ // The block device the counters detector reads for this location, when it
+ // could name one. The watchdog reads its counters too (see `onDrive`).
+ device?: string;
+};
+
+// The probe's budget. A `stat` of a directory on a healthy disk answers in
+// microseconds; three seconds is a thousand times that, and short enough that
+// a page asking during a stall has not waited long.
+export const HEALTH_PROBE_TIMEOUT_MS = 3_000;
+// How often the watch asks. Short, because the gate is only as current as the
+// last answer: a stall that began just after a probe costs every page that
+// touches the drive until the next one.
+export const HEALTH_PROBE_INTERVAL_MS = 15_000;
+// Clean answers in a row that clear a stall.
+export const HEALTH_CLEAN_TO_CLEAR = 2;
+
+// The one wording of the state, for every surface that shows it.
+export const NOT_ANSWERING = "drive not answering";
+
+// A call waiting for a slot. `resolve` answers whether it took the slot (a
+// waiter whose own wait already timed out does not); `rearm` restarts its
+// deadline, which a call returning on its key does (see `acquireSlot`).
+type Waiter = {
+ resolve: () => boolean;
+ reject: (err: Error) => void;
+ rearm: () => void;
+};
+
+// A call the watchdog gave up on that has not returned yet: when it began,
+// and the device's counters then (null when the counters detector has named
+// no device for its location).
+type OverdueCall = { startedAt: number; before: BlockStatSample | null };
+
+type HealthState = {
+ byId: Map<string, LocationHealth>;
+ // `onDrive`'s bookkeeping, by slot key (see `resolveWhere`): calls in
+ // flight, the ones among them the watchdog has already given up on, and
+ // the calls waiting for a slot.
+ inFlight?: Map<string, number>;
+ overdue?: Map<string, OverdueCall[]>;
+ waiters?: Map<string, Waiter[]>;
+ // Test seam: the watchdog's budget.
+ budgetMs?: number;
+ // Reads a block device's counters, synchronously and without touching the
+ // drive (set by `storageVolumes.ts`, which owns /sys). Absent: no check.
+ readCounters?: (device: string) => BlockStatSample | null;
+};
+
+declare global {
+ // eslint-disable-next-line no-var
+ var __yttStorageHealth__: HealthState | undefined;
+}
+
+type FilledState = HealthState &
+ Required<Pick<HealthState, "inFlight" | "overdue" | "waiters">>;
+
+function healthState(): FilledState {
+ if (!globalThis.__yttStorageHealth__) {
+ globalThis.__yttStorageHealth__ = { byId: new Map() };
+ }
+ const s = globalThis.__yttStorageHealth__;
+ // Filled lazily: a dev server's hot reload keeps an object made by an older
+ // copy of this module.
+ s.inFlight ??= new Map();
+ s.overdue ??= new Map();
+ s.waiters ??= new Map();
+ return s as FilledState;
+}
+
+// Test seam, and the escape hatch for a process that wants to forget. Calls
+// still waiting for a slot are released to run.
+export function resetStorageHealth(): void {
+ const s = healthState();
+ s.byId.clear();
+ s.inFlight.clear();
+ s.overdue.clear();
+ for (const q of s.waiters.values()) for (const w of q) w.resolve();
+ s.waiters.clear();
+}
+
+// `storageVolumes.ts` registers its /sys reader here (see `onDrive`, L6 in
+// the record: a call that times out while its disk is still completing
+// requests is slow, not stalled). A test passes its own, or undefined.
+export function setCounterReader(
+ read: ((device: string) => BlockStatSample | null) | undefined,
+): void {
+ healthState().readCounters = read;
+}
+
+export type HealthTransition = {
+ id: string;
+ from: LocationHealthState | null;
+ to: LocationHealthState;
+};
+
+// Record one answer about one location. Returns the transition when the state
+// changed, null when it did not.
+export function recordLocationHealth(
+ loc: Pick<StorageLocation, "id" | "label" | "root">,
+ answer: LocationHealthState,
+ opts: {
+ now?: number;
+ cause?: string;
+ detector?: HealthDetector;
+ // The counters' device; `null` forgets it (the stat detector answered).
+ device?: string | null;
+ } = {},
+): HealthTransition | null {
+ const t = recordAnswer(loc, answer, opts);
+ // EVERY TRANSITION TO STALLED refuses the calls waiting for a slot on the
+ // location, whoever decided it (the pass, the watchdog, a Refresh): they
+ // would otherwise wait for a slot held by a call the drive is not answering.
+ if (t?.to === "stalled") {
+ refuseWaiters(slotKeyOfLocation(loc.id), stalledLocation(loc));
+ }
+ return t;
+}
+
+function recordAnswer(
+ loc: Pick<StorageLocation, "id" | "label" | "root">,
+ answer: LocationHealthState,
+ opts: {
+ now?: number;
+ cause?: string;
+ detector?: HealthDetector;
+ device?: string | null;
+ },
+): HealthTransition | null {
+ const now = opts.now ?? Date.now();
+ const map = healthState().byId;
+ const prev = map.get(loc.id);
+ const label = loc.label || loc.id;
+ // First sighting, or the root moved under the same id: start over.
+ if (!prev || prev.root !== loc.root) {
+ map.set(loc.id, {
+ id: loc.id,
+ label,
+ root: loc.root,
+ state: answer,
+ ...(opts.detector ? { detector: opts.detector } : {}),
+ ...(opts.device ? { device: opts.device } : {}),
+ since: now,
+ checkedAt: now,
+ cleanStreak: 0,
+ ...(answer === "stalled" && opts.cause ? { cause: opts.cause } : {}),
+ });
+ return { id: loc.id, from: null, to: answer };
+ }
+ prev.label = label;
+ prev.checkedAt = now;
+ if (opts.detector) prev.detector = opts.detector;
+ if (opts.device === null) delete prev.device;
+ else if (opts.device) prev.device = opts.device;
+ if (answer === "stalled") {
+ prev.cleanStreak = 0;
+ if (prev.state === "stalled") return null;
+ const from = prev.state;
+ prev.state = "stalled";
+ prev.since = now;
+ if (opts.cause) prev.cause = opts.cause;
+ return { id: loc.id, from, to: "stalled" };
+ }
+ if (prev.state === "stalled") {
+ prev.cleanStreak += 1;
+ if (prev.cleanStreak < HEALTH_CLEAN_TO_CLEAR) return null;
+ prev.state = answer;
+ prev.since = now;
+ prev.cleanStreak = 0;
+ delete prev.cause;
+ return { id: loc.id, from: "stalled", to: answer };
+ }
+ if (prev.state === answer) return null;
+ const from = prev.state;
+ prev.state = answer;
+ prev.since = now;
+ return { id: loc.id, from, to: answer };
+}
+
+// Make sure every configured location has an entry, without recording an
+// answer: a new location starts `ok`, and a known one whose root moved starts
+// over. The watchdog below can only mark a location it can find, and it finds
+// a channel's location among these entries.
+export function registerLocationHealth(
+ locations: ReadonlyArray<Pick<StorageLocation, "id" | "label" | "root">>,
+ now: number = Date.now(),
+): void {
+ const map = healthState().byId;
+ for (const loc of locations) {
+ const prev = map.get(loc.id);
+ if (prev && prev.root === loc.root) {
+ prev.label = loc.label || loc.id;
+ continue;
+ }
+ recordLocationHealth(loc, "ok", { now });
+ }
+}
+
+// Which detector decided a location's last answer, when it gave none (the
+// counters' first sample has nothing to compare with).
+export function noteLocationDetector(
+ id: string,
+ detector: HealthDetector,
+ device?: string | null,
+): void {
+ const h = healthState().byId.get(id);
+ if (!h) return;
+ h.detector = detector;
+ if (device === null) delete h.device;
+ else if (device) h.device = device;
+}
+
+// Drop every location that is no longer configured.
+export function pruneLocationHealth(liveIds: Iterable<string>): void {
+ const keep = new Set(liveIds);
+ const map = healthState().byId;
+ for (const id of [...map.keys()]) if (!keep.has(id)) map.delete(id);
+}
+
+export function locationHealth(id: string): LocationHealth | undefined {
+ const h = healthState().byId.get(id);
+ return h ? { ...h } : undefined;
+}
+
+// Every location's last answer, by id. A copy: the caller may keep it.
+export function allLocationHealth(): Record<string, LocationHealth> {
+ const out: Record<string, LocationHealth> = {};
+ for (const [id, h] of healthState().byId) out[id] = { ...h };
+ return out;
+}
+
+// THE GATE. The stalled location `p` is on, or null. `p` is a channel's
+// `dataDir` (or anything under a location's root); the match is the one
+// `locationOfDataDir` makes, longest root first, so a channel on a nested
+// location answers for that location and not its parent.
+export function stalledLocationForPath(p: string): LocationHealth | null {
+ const map = healthState().byId;
+ if (map.size === 0 || !p) return null;
+ const entries = [...map.values()];
+ const loc = locationOfDataDir(
+ p,
+ entries.map((h) => ({ id: h.id, label: h.label, root: h.root, autoRepoint: false })),
+ );
+ if (!loc) return null;
+ const h = map.get(loc.id);
+ return h && h.state === "stalled" ? { ...h } : null;
+}
+
+// The location with this id, when it is stalled AND still at this root. The
+// gate for a caller that holds a location rather than a channel path
+// (`probeLocation`, `volumeFreeBytes`).
+export function stalledLocation(
+ loc: Pick<StorageLocation, "id" | "root">,
+): LocationHealth | null {
+ const h = healthState().byId.get(loc.id);
+ return h && h.state === "stalled" && h.root === loc.root ? { ...h } : null;
+}
+
+// "since 11:35" in the server's clock, or "since 2026-09-28 11:35" when the
+// stall began on another day than `now`.
+export function sinceText(since: number, now: number = Date.now()): string {
+ const d = new Date(since);
+ const n = new Date(now);
+ const pad = (x: number) => String(x).padStart(2, "0");
+ const hm = `${pad(d.getHours())}:${pad(d.getMinutes())}`;
+ const sameDay =
+ d.getFullYear() === n.getFullYear() &&
+ d.getMonth() === n.getMonth() &&
+ d.getDate() === n.getDate();
+ return sameDay
+ ? `since ${hm}`
+ : `since ${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())} ${hm}`;
+}
+
+// "not answering since 11:35" — the one line /storage, the /channels volume bar
+// and the channel's Storage panel show for a stalled location.
+export function notAnsweringText(h: Pick<LocationHealth, "since">, now?: number): string {
+ return `not answering ${sinceText(h.since, now)}`;
+}
+
+// ---------------------------------------------------------------------------
+// The watchdog: every call the gate covers, raced against 3 s
+// ---------------------------------------------------------------------------
+//
+// THE DETECTOR THAT CANNOT BE FOOLED BY A CACHE. The 15 s pass reads the
+// block device's counters (or, with no device, a child `stat`), and a stall
+// that starts between two passes is not seen by it. A page or a poll that
+// actually reaches the drive is: `onDrive` runs the call against a 3 s timer,
+// and a call that has not answered by then marks its location `stalled` at
+// once (since now) and throws `DriveNotAnsweringError`. The call itself is left
+// to settle on its own: its thread is held until the drive answers, which is
+// the stated limit.
+//
+// THE BUDGET COVERS A WHOLE UNIT OF WORK. A caller sends a unit through as one
+// call (a video directory's few reads, a page's reads of one video), so a slow
+// drive that is still answering can be marked by one long unit. Unless its
+// disk is visibly still completing requests: on a timeout, when the counters
+// detector has named the location's device, its counters are read (from /sys,
+// never the drive) and compared with a reading taken when the call began; if
+// requests completed meanwhile the drive is slow, not stalled, and only this
+// call is refused.
+//
+// AT MOST DRIVE_CALLS_IN_FLIGHT CALLS PER SLOT KEY ARE IN FLIGHT. A walk of
+// `data/` fans out 64 wide, and a stall mid-walk would otherwise put all 64 in
+// libuv's queue before the watchdog fired. The rest wait in a queue of our own:
+// - a transition to `stalled` (from any detector) refuses them at once;
+// - a wait is refused (without marking anything) only when no call on its key
+// has returned for the budget plus a small grace: every call that returns
+// restarts every waiter's deadline, so a deep queue on a drive that is busy
+// but answering waits as long as it takes;
+// - when every slot is held by a call the watchdog already gave up on, a new
+// call is refused at once, and the location marked stalled again unless
+// the disk has been completing requests since the oldest of those calls
+// began (slow, not stalled — the same test as a timeout's): none of those
+// calls has returned, whatever the last pass said.
+// A slot is released when its call really returns, not when the watchdog gave
+// up on it. So on one location at most four threads wait on its drive for the
+// calls that come through here — every page and poll path, and the snapshot
+// walk. A job's own reads that do not come through here are not capped.
+//
+// THE SLOT KEY: a configured location's id; for a probe of another root under a
+// location's id, that root; for a path on no configured location (a root typed
+// by hand), the root the path is under (`<root>/<slug>/data` → `<root>`). Only a
+// configured location can be marked.
+//
+// Do not nest `onDrive` for one key: the inner call would wait for a slot the
+// outer one holds.
+
+export const DRIVE_CALL_BUDGET_MS = 3_000;
+export const DRIVE_CALLS_IN_FLIGHT = 4;
+
+// Test seam: shorten (or restore, with no argument) the watchdog's budget.
+export function setDriveCallBudget(ms?: number): void {
+ healthState().budgetMs = ms;
+}
+
+function driveCallBudget(): number {
+ return healthState().budgetMs ?? DRIVE_CALL_BUDGET_MS;
+}
+
+export class DriveNotAnsweringError extends Error {
+ // The stalled location, when a known one was marked.
+ readonly health: LocationHealth | null;
+ constructor(health: LocationHealth | null, detail?: string) {
+ super(
+ detail ??
+ (health
+ ? `${NOT_ANSWERING} (location "${health.label}", ${sinceText(health.since)})`
+ : `${NOT_ANSWERING} (a read did not answer within ${driveCallBudget() / 1000} s)`),
+ );
+ this.name = "DriveNotAnsweringError";
+ this.health = health;
+ }
+}
+
+export function isDriveNotAnswering(err: unknown): err is DriveNotAnsweringError {
+ return err instanceof Error && err.name === "DriveNotAnsweringError";
+}
+
+type Where = string | Pick<StorageLocation, "id" | "label" | "root">;
+
+type Resolved = {
+ key: string;
+ // The location to mark, when marking is right.
+ loc: Pick<StorageLocation, "id" | "label" | "root"> | null;
+};
+
+function slotKeyOfLocation(id: string): string {
+ return `loc:${id}`;
+}
+
+// The root a path on no configured location is under: `<root>/<slug>/data`
+// (relocatedDataDir's shape) gives `<root>`; anything else is its own key.
+function rootOfUnknownPath(p: string): string {
+ const clean = p.replace(/\/+$/, "");
+ const parts = clean.split("/");
+ return parts.length > 2 && parts[parts.length - 1] === "data"
+ ? parts.slice(0, -2).join("/") || "/"
+ : clean;
+}
+
+function resolveWhere(where: Where): Resolved {
+ const map = healthState().byId;
+ if (typeof where !== "string") {
+ const entry = map.get(where.id);
+ // A probe of a candidate root under a location's id is its own key, and
+ // marks nothing: racing it is right, rewriting that location is not.
+ if (entry && entry.root !== where.root) {
+ return { key: `root:${where.root}`, loc: null };
+ }
+ return { key: slotKeyOfLocation(where.id), loc: where };
+ }
+ const found =
+ map.size > 0 && where
+ ? locationOfDataDir(
+ where,
+ [...map.values()].map((h) => ({
+ id: h.id,
+ label: h.label,
+ root: h.root,
+ autoRepoint: false,
+ })),
+ )
+ : null;
+ if (found) return { key: slotKeyOfLocation(found.id), loc: found };
+ return { key: `root:${rootOfUnknownPath(where)}`, loc: null };
+}
+
+function refuseWaiters(key: string, health: LocationHealth | null): void {
+ const s = healthState();
+ const q = s.waiters.get(key) ?? [];
+ s.waiters.delete(key);
+ for (const w of q) w.reject(new DriveNotAnsweringError(health));
+}
+
+// Take a slot on `key`, or wait for one (raced against the budget), or be
+// refused at once when every slot is held by an overdue call.
+async function acquireSlot(r: Resolved): Promise<void> {
+ const s = healthState();
+ const n = s.inFlight.get(r.key) ?? 0;
+ if (n < DRIVE_CALLS_IN_FLIGHT) {
+ s.inFlight.set(r.key, n + 1);
+ return;
+ }
+ const overdue = s.overdue.get(r.key) ?? [];
+ if (overdue.length >= DRIVE_CALLS_IN_FLIGHT) {
+ // EVERY SLOT IS HELD BY A CALL THE WATCHDOG GAVE UP ON: refused at once.
+ // Marked stalled only when the disk has not been completing requests
+ // since the oldest of them began — the watchdog's own slow-or-stalled test
+ // (see `onDrive`): four slow reads on a busy disk are not a stall.
+ const oldest = overdue.reduce((a, b) => (b.startedAt < a.startedAt ? b : a));
+ const now = readDeviceCounters(r.loc);
+ if (oldest.before && now && now.completed > oldest.before.completed) {
+ throw new DriveNotAnsweringError(
+ null,
+ `drive slow (${DRIVE_CALLS_IN_FLIGHT} reads on it are past ` +
+ `${driveCallBudget() / 1000} s, while its disk is still completing others)`,
+ );
+ }
+ let health: LocationHealth | null = null;
+ if (r.loc) {
+ recordLocationHealth(r.loc, "stalled", {
+ cause: `${DRIVE_CALLS_IN_FLIGHT} reads on it have not answered`,
+ });
+ health = stalledLocation(r.loc);
+ }
+ throw new DriveNotAnsweringError(health);
+ }
+ const budget = driveCallBudget();
+ // THE WAIT'S DEADLINE FOLLOWS PROGRESS, NOT THE QUEUE. A waiter is refused
+ // only when NO call on its key has returned for a full budget — "nothing on
+ // this drive answered for 3 s", the condition the watchdog exists for.
+ // Every call that returns on the key (in time or late) re-arms every waiter
+ // behind it, so a deep queue on a drive that is busy but answering (a
+ // 64-wide walk of units that each take a second) is never refused for its
+ // depth alone. Timed from when the call queued, it was: a waiter at depth d
+ // waits about d/4 units, whatever the drive is doing.
+ //
+ // PLUS A SMALL GRACE. A waiter's timer is created when it queues — before
+ // the race timers of calls that took their slots in the same tick — so with
+ // equal deadlines the waiters would give up a moment before the calls they
+ // wait behind, and be refused without the mark those calls are about to
+ // make. The grace lets them time out first.
+ const grace = Math.min(250, Math.round(budget / 4));
+ await new Promise<void>((resolve, reject) => {
+ let settled = false;
+ let timer: ReturnType<typeof setTimeout> | undefined;
+ const giveUp = () => {
+ if (settled) return;
+ settled = true;
+ const q = s.waiters.get(r.key);
+ if (q) {
+ const i = q.indexOf(waiter);
+ if (i >= 0) q.splice(i, 1);
+ }
+ reject(
+ new DriveNotAnsweringError(
+ r.loc ? stalledLocation(r.loc) : null,
+ `${NOT_ANSWERING} (nothing on it answered for ${budget / 1000} s while a read waited)`,
+ ),
+ );
+ };
+ const arm = () => {
+ if (timer) clearTimeout(timer);
+ timer = setTimeout(giveUp, budget + grace);
+ };
+ const waiter: Waiter = {
+ resolve: () => {
+ if (settled) return false;
+ settled = true;
+ if (timer) clearTimeout(timer);
+ resolve();
+ return true;
+ },
+ reject: (err) => {
+ if (settled) return;
+ settled = true;
+ if (timer) clearTimeout(timer);
+ reject(err);
+ },
+ rearm: () => {
+ if (!settled) arm();
+ },
+ };
+ arm();
+ const q = s.waiters.get(r.key) ?? [];
+ q.push(waiter);
+ s.waiters.set(r.key, q);
+ });
+ // Resolved by a release that handed its slot over: the count is unchanged.
+}
+
+// A slot is released when its call returns (or, on a refusal after a wait,
+// without a call). A return is progress: every waiter on the key has its
+// deadline restarted, then the slot goes to the first waiter still waiting.
+function releaseSlot(key: string): void {
+ const s = healthState();
+ const q = s.waiters.get(key);
+ if (q) for (const w of q) w.rearm();
+ while (q && q.length > 0) {
+ const next = q.shift() as Waiter;
+ if (next.resolve()) return;
+ }
+ s.inFlight.set(key, Math.max(0, (s.inFlight.get(key) ?? 1) - 1));
+}
+
+// How many calls are in flight on a location through `onDrive` (for tests and
+// for a reader that wants to say so).
+export function driveCallsInFlight(id: string): number {
+ return healthState().inFlight.get(slotKeyOfLocation(id)) ?? 0;
+}
+
+function readDeviceCounters(loc: Resolved["loc"]): BlockStatSample | null {
+ if (!loc) return null;
+ const s = healthState();
+ const device = s.byId.get(loc.id)?.device;
+ if (!device || !s.readCounters) return null;
+ try {
+ return s.readCounters(device);
+ } catch {
+ return null;
+ }
+}
+
+// Run `call` against the drive `where` is on: refused at once when that
+// location is stalled, queued behind DRIVE_CALLS_IN_FLIGHT calls already in
+// flight on its key, and raced against the budget. Throws
+// DriveNotAnsweringError for a refusal or a timeout; any other error is the
+// call's own.
+export async function onDrive<T>(where: Where, call: () => Promise<T>): Promise<T> {
+ const r = resolveWhere(where);
+ const refused = r.loc ? stalledLocation(r.loc) : null;
+ if (refused) throw new DriveNotAnsweringError(refused);
+ await acquireSlot(r);
+ // Stalled while this call waited: refused, and the slot passed on.
+ const late = r.loc ? stalledLocation(r.loc) : null;
+ if (late) {
+ releaseSlot(r.key);
+ throw new DriveNotAnsweringError(late);
+ }
+ const s = healthState();
+ let released = false;
+ const release = () => {
+ if (released) return;
+ released = true;
+ releaseSlot(r.key);
+ };
+ const startedAt = Date.now();
+ const before = readDeviceCounters(r.loc);
+ let pending: Promise<T>;
+ try {
+ pending = call();
+ } catch (err) {
+ release();
+ throw err;
+ }
+ const TIMED_OUT = Symbol("timed out");
+ let timer: ReturnType<typeof setTimeout> | undefined;
+ // Not unref'd: it is cleared the moment the call answers, and while the call
+ // is outstanding the timer is what must fire.
+ const timeout = new Promise<typeof TIMED_OUT>((resolve) => {
+ timer = setTimeout(() => resolve(TIMED_OUT), driveCallBudget());
+ });
+ let answer: T | typeof TIMED_OUT;
+ try {
+ answer = await Promise.race([pending, timeout]);
+ } catch (err) {
+ release();
+ throw err;
+ } finally {
+ if (timer) clearTimeout(timer);
+ }
+ if (answer !== TIMED_OUT) {
+ release();
+ return answer;
+ }
+ // The call is still waiting on the drive. Its slot stays held, and counted
+ // overdue, until it returns; nothing awaits it.
+ const entry: OverdueCall = { startedAt, before };
+ s.overdue.set(r.key, [...(s.overdue.get(r.key) ?? []), entry]);
+ const done = () => {
+ const left = (s.overdue.get(r.key) ?? []).filter((e) => e !== entry);
+ if (left.length > 0) s.overdue.set(r.key, left);
+ else s.overdue.delete(r.key);
+ release();
+ };
+ pending.then(done, done);
+ const after = readDeviceCounters(r.loc);
+ if (before && after && after.completed > before.completed) {
+ // SLOW, NOT STALLED: the disk completed requests while this call waited.
+ throw new DriveNotAnsweringError(
+ null,
+ `drive slow (a read did not answer within ${driveCallBudget() / 1000} s, ` +
+ `while its disk was still completing others)`,
+ );
+ }
+ if (r.loc) {
+ recordLocationHealth(r.loc, "stalled", {
+ cause: `a read in the editor did not answer within ${driveCallBudget() / 1000} s`,
+ });
+ throw new DriveNotAnsweringError(stalledLocation(r.loc));
+ }
+ refuseWaiters(r.key, null);
+ throw new DriveNotAnsweringError(null);
+}
+
+// ---------------------------------------------------------------------------
+// The block device's counters
+// ---------------------------------------------------------------------------
+//
+// `/sys/class/block/<dev>/stat` is the kernel's own count of a device's
+// requests (Documentation/block/stat.rst): field 1 reads completed, 5 writes
+// completed, 9 requests in flight now. Reading it never touches the drive. A
+// drive that is merely slow, even one grinding through a long write, keeps
+// completing requests; one in a reset loop has requests in flight and
+// completes none. So, between two samples a pass apart:
+//
+// stalled ⇔ in flight at both samples AND nothing completed between (reads,
+// writes, discards, flushes)
+// ok ⇔ anything else (nothing in flight at one of them, or completions
+// moved)
+//
+// and the health rules above turn one `stalled` into a stall and two `ok`s in
+// a row into its end.
+
+export type BlockStatSample = { completed: number; inFlight: number };
+
+// Completed: reads (field 1) + writes (5), and, on kernels that count them,
+// discards (12) and flushes (16) — in_flight counts those too, so a long flush
+// alone (an SMR drive emptying its media cache) must not read as nothing
+// completing.
+export function parseBlockStat(line: string): BlockStatSample | null {
+ const f = line.trim().split(/\s+/).map(Number);
+ if (f.length < 9 || f.slice(0, 9).some((n) => !Number.isFinite(n))) return null;
+ const opt = (i: number) => (Number.isFinite(f[i]) ? f[i] : 0);
+ return { completed: f[0] + f[4] + opt(11) + opt(15), inFlight: f[8] };
+}
+
+export function countersVerdict(
+ prev: BlockStatSample,
+ cur: BlockStatSample,
+): "ok" | "stalled" {
+ return prev.inFlight > 0 && cur.inFlight > 0 && cur.completed === prev.completed
+ ? "stalled"
+ : "ok";
+}
diff --git a/common/lib/storageHealthCounters.test.ts b/common/lib/storageHealthCounters.test.ts
@@ -0,0 +1,231 @@
+import { beforeEach, test } from "node:test";
+import assert from "node:assert/strict";
+import path from "node:path";
+import { tmpdir } from "node:os";
+import { chmod, mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
+import {
+ blockDeviceName,
+ detectLocationHealth,
+ MIN_COUNTER_INTERVAL_MS,
+ resetHealthDetector,
+} from "./storageVolumes";
+import { countersVerdict, parseBlockStat } from "./storageHealth";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/storageHealthCounters.test.ts
+//
+// THE COUNTERS DETECTOR. No test stalls a real drive: findmnt is a fake that
+// names a device, and /sys/class/block is a temp directory whose `stat` files
+// the test writes — a stalled device is one whose in-flight count stays up
+// while its completions stand still.
+
+beforeEach(() => resetHealthDetector());
+
+// A real line from this machine's /sys/class/block/<dev>/stat (17 fields).
+const LINE =
+ "368126412 135952867 36463191157 80435430 29670281 46781260 4001255365 167469727 0 44151394 253218297 877589 0 861936568 4100450 578669 1212688";
+
+test("the stat line: reads + writes (+ discards + flushes) completed, and requests in flight", () => {
+ // Fields 1, 5, 12 and 16 completed; field 9 in flight.
+ assert.deepEqual(parseBlockStat(LINE), {
+ completed: 368126412 + 29670281 + 877589 + 578669,
+ inFlight: 0,
+ });
+ // A long flush alone moves completions: not a stall.
+ const before = parseBlockStat("10 0 0 0 5 0 0 0 1 0 0 0 0 0 0 7 0");
+ const after = parseBlockStat("10 0 0 0 5 0 0 0 1 0 0 0 0 0 0 8 0");
+ assert.equal(countersVerdict(before!, after!), "ok");
+ // An 11-field line from an older kernel parses the same way.
+ assert.deepEqual(parseBlockStat("10 0 0 0 5 0 0 0 3 0 0\n"), { completed: 15, inFlight: 3 });
+ assert.equal(parseBlockStat(""), null);
+ assert.equal(parseBlockStat("1 2 3"), null);
+ assert.equal(parseBlockStat("a b c d e f g h i"), null);
+});
+
+test("the verdict over a pair of samples", () => {
+ const s = (completed: number, inFlight: number) => ({ completed, inFlight });
+ // In flight at both, nothing completed between: stalled.
+ assert.equal(countersVerdict(s(100, 2), s(100, 5)), "stalled");
+ // Completions moved: slow, perhaps, but answering.
+ assert.equal(countersVerdict(s(100, 2), s(101, 2)), "ok");
+ // Nothing in flight at either end: idle.
+ assert.equal(countersVerdict(s(100, 0), s(100, 0)), "ok");
+ assert.equal(countersVerdict(s(100, 0), s(100, 4)), "ok");
+ assert.equal(countersVerdict(s(100, 4), s(100, 0)), "ok");
+});
+
+test("a mount source's device name: a partition, a bind or subvolume suffix, nothing that is not /dev", async () => {
+ assert.equal(await blockDeviceName("/dev/no-such-disk1"), "no-such-disk1");
+ assert.equal(await blockDeviceName("/dev/no-such-disk2[/@home]"), "no-such-disk2");
+ assert.equal(await blockDeviceName("tmpfs"), null);
+ assert.equal(await blockDeviceName("server:/export"), null);
+ assert.equal(await blockDeviceName(""), null);
+});
+
+// ── the detector end to end, with a fake findmnt and a fake /sys ────────────
+
+const FAKE_FINDMNT = `#!/usr/bin/env node
+import { readFileSync } from "node:fs";
+import path from "node:path";
+const c = JSON.parse(readFileSync(path.join(import.meta.dirname, "control.json"), "utf8"));
+if (c.sleepMs) Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, c.sleepMs);
+if (c.exit) process.exit(c.exit);
+process.stdout.write(JSON.stringify({ filesystems: [{ source: c.source, uuid: c.uuid ?? null }] }) + "\\n");
+`;
+
+type H = {
+ root: string;
+ bins: { findmntBin: string };
+ sys: string;
+ control: (c: Record<string, unknown>) => Promise<void>;
+ counters: (device: string, completed: number, inFlight: number) => Promise<void>;
+};
+
+async function withHarness(fn: (h: H) => Promise<void>): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-counters-"));
+ try {
+ const bin = path.join(dir, "fake-findmnt.mjs");
+ await writeFile(bin, FAKE_FINDMNT);
+ await chmod(bin, 0o755);
+ const root = path.join(dir, "media");
+ await mkdir(root);
+ const sys = path.join(dir, "sys-block");
+ await fn({
+ root,
+ bins: { findmntBin: bin },
+ sys,
+ control: (c) => writeFile(path.join(dir, "control.json"), JSON.stringify(c)),
+ counters: async (device, completed, inFlight) => {
+ await mkdir(path.join(sys, device), { recursive: true });
+ // Reads `completed`, writes 0, in flight `inFlight`, the rest zero.
+ await writeFile(
+ path.join(sys, device, "stat"),
+ `${completed} 0 0 0 0 0 0 0 ${inFlight} 0 0 0 0 0 0 0 0\n`,
+ );
+ },
+ });
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+const loc = (root: string, uuid?: string) => ({
+ id: "usb",
+ root,
+ ...(uuid ? { volume: { uuid, mountpoint: root, relPath: "" } } : {}),
+});
+
+test("counters: first sample no verdict; stuck → stalled; two clean samples; each pass apart", async () => {
+ await withHarness(async (h) => {
+ await h.control({ source: "/dev/fakedisk1", uuid: "u-1" });
+ const at = (i: number) => ({ sysBlockDir: h.sys, now: i * 15_000 });
+ await h.counters("fakedisk1", 100, 2);
+ assert.deepEqual(await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(1)), {
+ answer: null,
+ detector: "counters",
+ device: "fakedisk1",
+ });
+ // Still two in flight, nothing completed in 15 s.
+ await h.counters("fakedisk1", 100, 2);
+ const stuck = await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(2));
+ assert.equal(stuck.answer, "stalled");
+ assert.equal(stuck.detector, "counters");
+ assert.match(String(stuck.cause), /^its disk \(fakedisk1\) had 2 request\(s\) in flight and completed none in 15 s$/);
+ // It drains: nothing in flight.
+ await h.counters("fakedisk1", 140, 0);
+ assert.equal((await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(3))).answer, "ok");
+ // Busy and moving: still ok.
+ await h.counters("fakedisk1", 190, 3);
+ assert.equal((await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(4))).answer, "ok");
+ });
+});
+
+test("counters: a second sample sooner than the minimum interval gives no verdict and keeps the first", async () => {
+ await withHarness(async (h) => {
+ await h.control({ source: "/dev/fakedisk1" });
+ await h.counters("fakedisk1", 100, 2);
+ await detectLocationHealth(loc(h.root), h.bins, { sysBlockDir: h.sys, now: 0 });
+ const soon = await detectLocationHealth(loc(h.root), h.bins, {
+ sysBlockDir: h.sys,
+ now: MIN_COUNTER_INTERVAL_MS - 1,
+ });
+ assert.equal(soon.answer, null);
+ // Compared with the FIRST sample, not the refused one.
+ const later = await detectLocationHealth(loc(h.root), h.bins, {
+ sysBlockDir: h.sys,
+ now: 15_000,
+ });
+ assert.equal(later.answer, "stalled");
+ });
+});
+
+test("no device → the child stat, and the verdict says so", async () => {
+ await withHarness(async (h) => {
+ const opts = { sysBlockDir: h.sys, now: 0 };
+ // A tmpfs or network source names no block device.
+ await h.control({ source: "tmpfs" });
+ assert.deepEqual(await detectLocationHealth(loc(h.root), h.bins, opts), {
+ answer: "ok",
+ detector: "stat",
+ });
+ // findmnt cannot say (no such path, a container without it).
+ await h.control({ exit: 1 });
+ assert.equal((await detectLocationHealth(loc(h.root), h.bins, opts)).detector, "stat");
+ assert.equal(
+ (await detectLocationHealth(loc(path.join(h.root, "gone")), h.bins, opts)).answer,
+ "absent",
+ );
+ // A device with no /sys entry.
+ await h.control({ source: "/dev/nodisk9" });
+ assert.equal((await detectLocationHealth(loc(h.root), h.bins, opts)).detector, "stat");
+ // A root whose filesystem is not the location's recorded volume.
+ await h.counters("fakedisk1", 1, 0);
+ await h.control({ source: "/dev/fakedisk1", uuid: "someone-else" });
+ assert.equal(
+ (await detectLocationHealth(loc(h.root, "u-1"), h.bins, opts)).detector,
+ "stat",
+ );
+ // No findmnt binary at all.
+ assert.equal(
+ (
+ await detectLocationHealth(
+ loc(h.root),
+ { findmntBin: path.join(h.root, "no-findmnt") },
+ opts,
+ )
+ ).detector,
+ "stat",
+ );
+ });
+});
+
+test("a device already named is read without findmnt; findmnt again only when its /sys entry stops reading", async () => {
+ await withHarness(async (h) => {
+ await h.control({ source: "/dev/fakedisk1" });
+ await h.counters("fakedisk1", 100, 2);
+ await detectLocationHealth(loc(h.root), h.bins, { sysBlockDir: h.sys, now: 0 });
+ // findmnt now hangs: it is not asked, the known device is read.
+ await h.control({ sleepMs: 5_000, source: "/dev/fakedisk1" });
+ const started = Date.now();
+ const v = await detectLocationHealth(loc(h.root), h.bins, {
+ sysBlockDir: h.sys,
+ now: 15_000,
+ timeoutMs: 300,
+ });
+ assert.ok(Date.now() - started < 250, "no findmnt was run");
+ assert.equal(v.detector, "counters");
+ assert.equal(v.answer, "stalled");
+ // The device is gone from /sys (replugged under another name): findmnt is
+ // asked again — here it does not answer, so no device, and the child stat.
+ await rm(path.join(h.sys, "fakedisk1"), { recursive: true });
+ const again = await detectLocationHealth(loc(h.root), h.bins, {
+ sysBlockDir: h.sys,
+ now: 30_000,
+ timeoutMs: 300,
+ });
+ assert.equal(again.detector, "stat");
+ // The detector's samples are on globalThis (the pass and a Refresh run in
+ // different module copies and compare against one previous sample).
+ assert.ok(globalThis.__yttHealthDetector__);
+ });
+});
diff --git a/common/lib/storageHealthProbe.test.ts b/common/lib/storageHealthProbe.test.ts
@@ -0,0 +1,112 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import path from "node:path";
+import { tmpdir } from "node:os";
+import { chmod, mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
+import { probeLocationHealth } from "./storageVolumes";
+import { HEALTH_PROBE_TIMEOUT_MS } from "./storageHealth";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/storageHealthProbe.test.ts
+//
+// THE HEALTH PROBE: `stat` on a location's root as a CHILD PROCESS, raced
+// against a timer. No test stalls a real drive: a stalled `stat` is a fake
+// binary that never answers (a child blocked in the kernel looks the same from
+// here — it does not exit), and the case that matters is that the answer
+// arrives on the timer while the child is still running.
+
+async function withDir(fn: (dir: string) => Promise<void>): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-health-"));
+ try {
+ await fn(dir);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+// A fake `stat`: sleeps `sleepMs` (a blocking sleep, so it is simply not
+// answering), then prints `out` and exits `code`.
+async function fakeStat(
+ dir: string,
+ opts: { sleepMs?: number; out?: string; code?: number },
+): Promise<string> {
+ const bin = path.join(dir, "fake-stat.mjs");
+ await writeFile(
+ bin,
+ `#!/usr/bin/env node
+if (${opts.sleepMs ?? 0} > 0) {
+ Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ${opts.sleepMs ?? 0});
+}
+process.stdout.write(${JSON.stringify(opts.out ?? "")} + "\\n");
+process.exit(${opts.code ?? 0});
+`,
+ );
+ await chmod(bin, 0o755);
+ return bin;
+}
+
+test("the real stat: a directory is ok, a missing path and a file are absent", async () => {
+ await withDir(async (dir) => {
+ const root = path.join(dir, "media");
+ await mkdir(root);
+ await writeFile(path.join(dir, "a-file"), "x");
+ assert.equal(await probeLocationHealth({ root }), "ok");
+ assert.equal(await probeLocationHealth({ root: path.join(dir, "gone") }), "absent");
+ assert.equal(await probeLocationHealth({ root: path.join(dir, "a-file") }), "absent");
+ assert.equal(await probeLocationHealth({ root: " " }), "absent");
+ });
+});
+
+test("a child that never answers is 'stalled' on the timer, without waiting for it", async () => {
+ await withDir(async (dir) => {
+ // Twenty seconds asleep: if the probe waited for the child, this test would
+ // take that long.
+ const statBin = await fakeStat(dir, { sleepMs: 20_000, out: "directory" });
+ const started = Date.now();
+ const answer = await probeLocationHealth(
+ { root: dir },
+ { statBin, timeoutMs: 400 },
+ );
+ const took = Date.now() - started;
+ assert.equal(answer, "stalled");
+ assert.ok(took >= 400, `answered after ${took} ms, before the timer`);
+ assert.ok(took < 3_000, `answered after ${took} ms`);
+ });
+});
+
+test("the default budget is 3 s", async () => {
+ assert.equal(HEALTH_PROBE_TIMEOUT_MS, 3_000);
+ await withDir(async (dir) => {
+ const statBin = await fakeStat(dir, { sleepMs: 20_000, out: "directory" });
+ const started = Date.now();
+ assert.equal(await probeLocationHealth({ root: dir }, { statBin }), "stalled");
+ const took = Date.now() - started;
+ assert.ok(took >= 3_000 && took < 6_000, `answered after ${took} ms`);
+ });
+});
+
+test("an answer inside the budget is taken as given", async () => {
+ await withDir(async (dir) => {
+ const slowDir = await fakeStat(dir, { sleepMs: 100, out: "directory" });
+ assert.equal(
+ await probeLocationHealth({ root: dir }, { statBin: slowDir, timeoutMs: 2_500 }),
+ "ok",
+ );
+ const aFile = await fakeStat(dir, { out: "regular file" });
+ assert.equal(await probeLocationHealth({ root: dir }, { statBin: aFile }), "absent");
+ const missing = await fakeStat(dir, { code: 1 });
+ assert.equal(await probeLocationHealth({ root: dir }, { statBin: missing }), "absent");
+ });
+});
+
+test("a stat that cannot be started fails open: ok, never stalled", async () => {
+ await withDir(async (dir) => {
+ assert.equal(
+ await probeLocationHealth(
+ { root: dir },
+ { statBin: path.join(dir, "no-such-stat") },
+ ),
+ "ok",
+ );
+ });
+});
diff --git a/common/lib/storageVolumes.ts b/common/lib/storageVolumes.ts
@@ -1,9 +1,22 @@
import path from "node:path";
-import { lstat, stat } from "node:fs/promises";
+import { readFileSync } from "node:fs";
+import { lstat, readFile, realpath, stat } from "node:fs/promises";
import { execa } from "execa";
import type { Paths } from "./paths";
import { getFreeBytes } from "./diskSpace";
import type { StorageLocation, StorageVolume } from "./storageLocations";
+import {
+ HEALTH_PROBE_TIMEOUT_MS,
+ countersVerdict,
+ isDriveNotAnswering,
+ onDrive,
+ parseBlockStat,
+ setCounterReader,
+ stalledLocation,
+ type BlockStatSample,
+ type HealthDetector,
+ type LocationHealthState,
+} from "./storageHealth";
// STORAGE VOLUME PROBES — is this location's disk here, and if not, where?
//
@@ -33,7 +46,11 @@ export type StorageLocationStatus =
| "absent"
// The root is not there and we have no identity to look for — nothing to say
// beyond "that path does not exist".
- | "missing";
+ | "missing"
+ // The drive did not answer the last health probe (`lib/storageHealth.ts`), so
+ // this probe did not ask: an in-process `stat` there would hold one of the
+ // process's I/O threads until the drive answered. No identity, no free space.
+ | "stalled";
// `known: false` is the fail-open answer and is NOT a problem report: it means
// the probe could not ask (no findmnt, a container, a timeout), not that the
@@ -226,13 +243,25 @@ export async function probeLocation(
const timeoutMs = opts.findmntTimeoutMs ?? FINDMNT_TIMEOUT_MS;
const root = loc.root.trim();
+ // A DRIVE THAT IS NOT ANSWERING IS NOT ASKED. The `stat` and `statfs` below
+ // run in-process, and on a stalled disk each holds an I/O thread until the
+ // drive comes back; the health pass (below) already asked without touching
+ // it. Both go through `onDrive`'s watchdog, too: one that has not answered
+ // in 3 s marks the location stalled and this probe answers so.
+ const stalledProbe: StorageLocationProbe = {
+ status: "stalled",
+ identity: { known: false },
+ };
+ if (stalledLocation(loc)) return stalledProbe;
+
// AVAILABILITY IS `stat`, AND ONLY `stat`. A root that is a directory is
// available even when every identity probe below fails — see the header.
let isDir = false;
if (root !== "") {
try {
- isDir = (await stat(root)).isDirectory();
- } catch {
+ isDir = (await onDrive(loc, () => stat(root))).isDirectory();
+ } catch (err) {
+ if (isDriveNotAnswering(err)) return stalledProbe;
isDir = false;
}
}
@@ -245,7 +274,13 @@ export async function probeLocation(
// `null` and every byte formatter downstream gets a surprise. An
// unmeasurable root reports no free space at all, which is the honest
// answer and the one the field is already optional for.
- const free = await getFreeBytes(root);
+ let free: number;
+ try {
+ free = await onDrive(loc, () => getFreeBytes(root));
+ } catch (err) {
+ if (isDriveNotAnswering(err)) return stalledProbe;
+ throw err;
+ }
// Two ways a mount will not be there after a reboot: udisks put it under
// /run/media (or /media) because a human plugged it in, or there is no
// fstab entry naming its UUID. The fstab call is skipped when the
@@ -304,6 +339,323 @@ export async function probeLocation(
}
}
+// ---------------------------------------------------------------------------
+// The health probe: is the drive ANSWERING, asked from a child process
+// ---------------------------------------------------------------------------
+//
+// `probeLocation` answers "is the disk here" with an in-process `stat`, which is
+// right until the disk is here and not answering: then that `stat` does not
+// fail, it waits — for as long as the drive takes, on one of the four threads
+// libuv runs every filesystem call on. A child process waiting in the kernel
+// holds none of them. So this runs `stat` on the root as a SUBPROCESS and races
+// it against a timer, and the timer's answer is `stalled`.
+//
+// THE TIMER WINS, AND NOTHING WAITS FOR THE CHILD. A process in uninterruptible
+// I/O cannot be killed until the I/O returns, so awaiting its exit (which is
+// what execa's own `timeout` does) would hand the wait straight back to us.
+// The child is sent SIGKILL, which it takes when the drive lets it, and its
+// promise settles into a handler nobody awaits.
+//
+// FAILS OPEN, like everything else here: a `stat` that could not be started
+// (no binary) is "could not ask", which is `ok`, never `stalled`. Only a child
+// that started and did not answer in time is a stall.
+
+export type HealthProbeOptions = {
+ timeoutMs?: number;
+ // Test seam: the binary to run (it is called as `<bin> -L -c %F -- <root>`).
+ statBin?: string;
+};
+
+// One pass's answer about one location. `answer: null` is no verdict (the
+// counters' first sample, or a second one taken too soon after the last).
+export type HealthVerdict = {
+ answer: LocationHealthState | null;
+ detector?: HealthDetector;
+ // For a `stalled` answer: what did not answer, in words with no path.
+ cause?: string;
+ // The block device the counters were read from.
+ device?: string;
+};
+
+// What the health pass asks about a location. A bare state is a verdict with
+// no detector named (the tests' scripted probes).
+export type LocationHealthProbe = (
+ loc: Pick<StorageLocation, "id" | "label" | "root" | "volume">,
+) => Promise<LocationHealthState | HealthVerdict>;
+
+export async function probeLocationHealth(
+ loc: Pick<StorageLocation, "root">,
+ opts: HealthProbeOptions = {},
+): Promise<LocationHealthState> {
+ const root = loc.root.trim();
+ if (root === "") return "absent";
+ const timeoutMs = opts.timeoutMs ?? HEALTH_PROBE_TIMEOUT_MS;
+ let child: ReturnType<typeof execa>;
+ try {
+ child = execa(opts.statBin ?? "stat", ["-L", "-c", "%F", "--", root], {
+ buffer: true,
+ reject: false,
+ stdin: "ignore",
+ });
+ } catch {
+ return "ok";
+ }
+ const answered: Promise<LocationHealthState> = child.then(
+ (res) => {
+ // No exit code: the binary never ran (or was killed after the race was
+ // already decided). Could not ask.
+ if (typeof res.exitCode !== "number") return "ok";
+ if (res.exitCode !== 0) return "absent";
+ const out = typeof res.stdout === "string" ? res.stdout.trim() : "";
+ return out === "directory" ? "ok" : "absent";
+ },
+ () => "ok",
+ );
+ let timer: ReturnType<typeof setTimeout> | undefined;
+ const timedOut = new Promise<LocationHealthState>((resolve) => {
+ timer = setTimeout(() => resolve("stalled"), timeoutMs);
+ timer.unref?.();
+ });
+ const answer = await Promise.race([answered, timedOut]);
+ if (timer) clearTimeout(timer);
+ if (answer === "stalled") {
+ try {
+ child.kill("SIGKILL");
+ } catch {
+ /* already gone */
+ }
+ }
+ return answer;
+}
+
+// ---------------------------------------------------------------------------
+// The counters detector: the block device's own request counters
+// ---------------------------------------------------------------------------
+//
+// THE STAT PROBE ABOVE CAN BE ANSWERED FROM THE KERNEL'S CACHE. A root's inode
+// is cached whenever anything has used the drive lately, so a child `stat` of
+// it answers in microseconds while the reads that actually reach the device
+// wait out a reset loop. The block device's counters (lib/storageHealth.ts,
+// `countersVerdict`) are the device's own account of what it has done, and
+// reading them touches only /sys. So the health pass asks this first:
+//
+// 1. the root's device: `findmnt -J -T <root> -o SOURCE,UUID` as a child
+// raced against 3 s (a findmnt stuck resolving the root holds nothing of
+// ours), the `[subvolume]` suffix a bind or btrfs mount adds taken off,
+// `/dev/mapper/<x>` resolved to its `dm-N`, then the basename. A partition
+// and a mapper device both have `/sys/class/block/<name>/stat`. Asked only
+// when the root has no device yet or its device's /sys entry cannot be read
+// (a drive replugged under another name): a findmnt per pass would leave
+// one child stuck per pass during a long stall. One that times out names
+// none this pass; one whose UUID is not the location's recorded one names
+// none (the root is then a directory on some other filesystem).
+// 2. that device's `stat` line, compared with the previous pass's sample for
+// the location (same device, at least MIN_COUNTER_INTERVAL_MS earlier).
+//
+// NO DEVICE (a container, no findmnt, a network or tmpfs mount, no /sys entry)
+// FALLS BACK TO THE CHILD `stat`, and the verdict says which detector answered.
+
+export type DetectorOptions = HealthProbeOptions & {
+ // Test seams: where /sys/class/block is, and the clock.
+ sysBlockDir?: string;
+ now?: number;
+};
+
+export const SYS_BLOCK_DIR = "/sys/class/block";
+// Two samples closer than this are not compared: a healthy drive can have a
+// request in flight at two instants a moment apart without completing one.
+// The pass is 15 s apart; a /storage Refresh just after a pass gives no verdict.
+export const MIN_COUNTER_INTERVAL_MS = 10_000;
+
+type CounterSample = BlockStatSample & { device: string; at: number };
+
+type DetectorState = {
+ samples: Map<string, CounterSample>;
+ deviceByRoot: Map<string, string>;
+};
+
+declare global {
+ // eslint-disable-next-line no-var
+ var __yttHealthDetector__: DetectorState | undefined;
+}
+
+// ON globalThis, the house pattern: the health pass runs in instrumentation's
+// module copy and /storage's Refresh in a page's, and they must compare
+// against the same previous sample.
+function detectorState(): DetectorState {
+ globalThis.__yttHealthDetector__ ??= {
+ samples: new Map(),
+ deviceByRoot: new Map(),
+ };
+ return globalThis.__yttHealthDetector__;
+}
+
+export function resetHealthDetector(): void {
+ detectorState().samples.clear();
+ detectorState().deviceByRoot.clear();
+}
+
+// THE WATCHDOG'S READING of a device's counters (lib/storageHealth.ts,
+// `onDrive`): synchronous, so it does not wait behind the thread pool it is
+// judging, and cheap — a /sys read is answered by the kernel from memory and
+// never reaches the drive.
+function readCountersNow(device: string): BlockStatSample | null {
+ try {
+ return parseBlockStat(readFileSync(path.join(SYS_BLOCK_DIR, device, "stat"), "utf8"));
+ } catch {
+ return null;
+ }
+}
+setCounterReader(readCountersNow);
+
+type Raced = { answered: false } | ({ answered: true } & Run);
+
+// `run`, but raced against a timer that nobody waits past: a child stuck in
+// the kernel is sent SIGKILL and left to exit when it can.
+async function runRaced(bin: string, args: string[], timeoutMs: number): Promise<Raced> {
+ let child: ReturnType<typeof execa>;
+ try {
+ child = execa(bin, args, { buffer: true, reject: false, stdin: "ignore" });
+ } catch {
+ return { answered: true, ok: false, exitCode: undefined, stdout: "", stderr: "" };
+ }
+ const answered: Promise<Raced> = child.then(
+ (res) => {
+ const exitCode = typeof res.exitCode === "number" ? res.exitCode : undefined;
+ return {
+ answered: true as const,
+ ok: exitCode === 0,
+ exitCode,
+ stdout: typeof res.stdout === "string" ? res.stdout : "",
+ stderr: typeof res.stderr === "string" ? res.stderr : "",
+ };
+ },
+ () => ({ answered: true as const, ok: false, exitCode: undefined, stdout: "", stderr: "" }),
+ );
+ let timer: ReturnType<typeof setTimeout> | undefined;
+ const timedOut = new Promise<Raced>((resolve) => {
+ timer = setTimeout(() => resolve({ answered: false }), timeoutMs);
+ });
+ const out = await Promise.race([answered, timedOut]);
+ if (timer) clearTimeout(timer);
+ if (!out.answered) {
+ try {
+ child.kill("SIGKILL");
+ } catch {
+ /* already gone */
+ }
+ }
+ return out;
+}
+
+// The block device name under /sys/class/block for a mount SOURCE, or null.
+export async function blockDeviceName(source: string): Promise<string | null> {
+ const bare = source.replace(/\[.*\]$/, "").trim();
+ if (!bare.startsWith("/dev/")) return null;
+ // /dev/mapper/<x> and /dev/disk/by-*/<x> are links to the kernel's name.
+ // Resolving them reads /dev, never the drive.
+ const resolved = await realpath(bare).catch(() => bare);
+ const name = path.basename(resolved);
+ return name && name !== "dev" ? name : null;
+}
+
+async function blockDeviceOfRoot(
+ loc: Pick<StorageLocation, "root" | "volume">,
+ bins: Pick<VolumeBins, "findmntBin">,
+ timeoutMs: number,
+): Promise<{ device: string | null; timedOut: boolean }> {
+ const res = await runRaced(
+ bins.findmntBin,
+ ["-J", "-T", loc.root, "-o", "SOURCE,UUID"],
+ timeoutMs,
+ );
+ if (!res.answered) return { device: null, timedOut: true };
+ if (!res.ok) return { device: null, timedOut: false };
+ let fs0: Record<string, unknown> | undefined;
+ try {
+ fs0 = (JSON.parse(res.stdout) as { filesystems?: Record<string, unknown>[] })
+ ?.filesystems?.[0];
+ } catch {
+ return { device: null, timedOut: false };
+ }
+ const source = typeof fs0?.source === "string" ? fs0.source : "";
+ const uuid = typeof fs0?.uuid === "string" ? fs0.uuid : "";
+ const recorded = loc.volume?.uuid?.trim() ?? "";
+ if (recorded && uuid && uuid !== recorded) return { device: null, timedOut: false };
+ return { device: await blockDeviceName(source), timedOut: false };
+}
+
+async function readBlockStat(
+ device: string,
+ sysBlockDir: string,
+): Promise<BlockStatSample | null> {
+ try {
+ return parseBlockStat(await readFile(path.join(sysBlockDir, device, "stat"), "utf8"));
+ } catch {
+ return null;
+ }
+}
+
+// One location's verdict for the health pass: the counters when its device can
+// be named and read, the child `stat` otherwise.
+export async function detectLocationHealth(
+ loc: Pick<StorageLocation, "id" | "root" | "volume">,
+ bins: Pick<VolumeBins, "findmntBin">,
+ opts: DetectorOptions = {},
+): Promise<HealthVerdict> {
+ const now = opts.now ?? Date.now();
+ const timeoutMs = opts.timeoutMs ?? HEALTH_PROBE_TIMEOUT_MS;
+ const sysBlockDir = opts.sysBlockDir ?? SYS_BLOCK_DIR;
+ const detector = detectorState();
+ // The device this root was last known on, if its /sys entry still reads;
+ // findmnt only when there is none (see step 1 above).
+ let device: string | null = detector.deviceByRoot.get(loc.root) ?? null;
+ let sample = device ? await readBlockStat(device, sysBlockDir) : null;
+ if (!sample) {
+ detector.deviceByRoot.delete(loc.root);
+ const mapped = loc.root.trim()
+ ? await blockDeviceOfRoot(loc, bins, timeoutMs)
+ : { device: null, timedOut: false };
+ device = mapped.device;
+ sample = device ? await readBlockStat(device, sysBlockDir) : null;
+ }
+ if (device) {
+ if (sample) {
+ detector.deviceByRoot.set(loc.root, device);
+ const prev = detector.samples.get(loc.id);
+ if (prev && prev.device === device && now - prev.at < MIN_COUNTER_INTERVAL_MS) {
+ return { answer: null, detector: "counters", device };
+ }
+ detector.samples.set(loc.id, { ...sample, device, at: now });
+ if (!prev || prev.device !== device) {
+ return { answer: null, detector: "counters", device };
+ }
+ const answer = countersVerdict(prev, sample);
+ return {
+ answer,
+ detector: "counters",
+ device,
+ ...(answer === "stalled"
+ ? {
+ cause:
+ `its disk (${device}) had ${sample.inFlight} request(s) in flight and ` +
+ `completed none in ${Math.round((now - prev.at) / 1000)} s`,
+ }
+ : {}),
+ };
+ }
+ }
+ detector.samples.delete(loc.id);
+ const answer = await probeLocationHealth(loc, opts);
+ return {
+ answer,
+ detector: "stat",
+ ...(answer === "stalled"
+ ? { cause: `a stat of its root did not answer within ${timeoutMs / 1000} s` }
+ : {}),
+ };
+}
+
// udisksctl availability, memoised per binary path. Same shape as a digest
// app's probe (`digestApps.ts` claudeCode.probe): `--version`, reject:false,
// short timeout. Memoised because /storage asks once per render and the answer
@@ -401,6 +753,12 @@ export async function probeLocationMemo(
opts: ProbeOptions & { refresh?: boolean; now?: number } = {},
): Promise<MemoizedProbe> {
const now = opts.now ?? Date.now();
+ // The stall is asked before the memo, so a remembered "available" from a few
+ // seconds ago does not outlive the drive's answer. Not remembered either:
+ // the health probe's state is already the memory.
+ if (stalledLocation(loc)) {
+ return { status: "stalled", identity: { known: false }, probedAt: now };
+ }
const hit = probeMemo.get(loc.id);
if (
!opts.refresh &&
diff --git a/common/views/pipeline/stageStatus.ts b/common/views/pipeline/stageStatus.ts
@@ -601,7 +601,8 @@ export function computeStageStatuses(
mediaStatus === "in-transition"
? "running"
: mediaStatus === "unreachable" ||
- mediaStatus === "inconsistent"
+ mediaStatus === "inconsistent" ||
+ mediaStatus === "stalled"
? "danger"
: "neutral",
};
diff --git a/common/views/storage.test.ts b/common/views/storage.test.ts
@@ -399,3 +399,36 @@ test("a location row carries the clips share of its bytes", () => {
// A SUBSET, not a sibling: the clips are already inside `bytes`.
assert.equal(row?.bytes, 10_000_000_000);
});
+
+test("a location whose drive is not answering reads so, and says since when", () => {
+ const { rows } = buildStorageRows({
+ locations: [loc("cold", "/mnt/cold")],
+ defaultLocationId: "",
+ probes: {
+ cold: { status: "stalled", identity: { known: false }, probedAt: NOW },
+ },
+ notAnswering: { cold: "not answering since 11:35 — a stat of its root did not answer within 3 s." },
+ rollups: { cold: rollup({ locationId: "cold", total: 2, unreachable: 2 }) },
+ registry: NO_JOBS,
+ now: NOW,
+ });
+ const row = rows[0];
+ assert.equal(row.status, "stalled");
+ assert.equal(row.statusLabel, "Not answering");
+ assert.match(String(row.notAnswering), /^not answering since 11:35/);
+ assert.equal(row.freeBytes, undefined);
+ // Neither re-point nor mount is offered for a drive that is here and stalled.
+ assert.equal(action(row, "repoint").offered, false);
+ assert.match(String(action(row, "repoint").withheld), /not answering/);
+ assert.equal(action(row, "mount").offered, false);
+ // A row with no entry says nothing.
+ const quiet = buildStorageRows({
+ locations: [loc("cold", "/mnt/cold")],
+ defaultLocationId: "",
+ probes: {},
+ rollups: {},
+ registry: NO_JOBS,
+ now: NOW,
+ }).rows[0];
+ assert.equal(quiet.notAnswering, undefined);
+});
diff --git a/common/views/storage.ts b/common/views/storage.ts
@@ -103,6 +103,9 @@ export type StorageRow = {
freeBytes?: number;
// How long ago the probe behind this row was taken, in ms.
lastProbeAgeMs: number;
+ // "not answering since 11:35 — …", while the health probe finds this
+ // location's drive not answering (lib/storageHealth.ts). Absent otherwise.
+ notAnswering?: string;
warning?: string;
candidateRoot?: string;
// Why every action on this row is withheld, or null. One re-point runs at a
@@ -201,6 +204,10 @@ export type StorageRowsInputs = {
// omitted, because a row that vanishes when a probe fails is worse than one
// that says it does not know.
probes: Record<string, MemoizedProbe | undefined>;
+ // By location id: the sentence for a location whose drive is not answering.
+ // Worded by the shell (lib/storageHealth.ts's `notAnsweringText`), because
+ // this module takes types only from lib/.
+ notAnswering?: Record<string, string | undefined>;
rollups: Record<string, LocationRollup | undefined>;
registry: RegistryReader;
udisksctlAvailable?: boolean;
@@ -217,6 +224,7 @@ export const STORAGE_STATUS_LABEL: Record<StorageLocationStatus, string> = {
unmounted: "Not mounted",
absent: "Not attached",
missing: "Missing",
+ stalled: "Not answering",
};
export const REPOINT_JOB_KIND = "repoint-storage-location";
@@ -386,6 +394,9 @@ export function buildStorageRows(i: StorageRowsInputs): StorageRowsPayload {
clipsText: storageClipsText(clipsBytes),
...(probe?.freeBytes !== undefined ? { freeBytes: probe.freeBytes } : {}),
lastProbeAgeMs: probe ? Math.max(0, i.now - probe.probedAt) : 0,
+ ...(i.notAnswering?.[loc.id]
+ ? { notAnswering: i.notAnswering[loc.id] }
+ : {}),
...(probe?.warning ? { warning: probe.warning } : {}),
...(candidateRoot ? { candidateRoot } : {}),
busy,
diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh
@@ -222,6 +222,12 @@ editor)
log "idle boot: schedulers, auto-queue runners and sweeps stay STOPPED"
fi
cd /repo/editor
+ # Sixteen threads for Node's filesystem pool instead of four. A call on a
+ # drive that has stalled holds its thread until the drive answers; with four,
+ # four such calls stop the editor answering at all. More threads buy time for
+ # calls already in flight — they isolate nothing (the storage health probe
+ # and its gate are what keep new calls off a stalled drive).
+ export UV_THREADPOOL_SIZE="${UV_THREADPOOL_SIZE:-16}"
exec "$(next_bin /repo/editor)" start --port "${EDITOR_PORT:-3001}"
;;
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -5,6 +5,7 @@
- **A stats build keeps the stats of a channel whose drive is not mounted, and will not undo a newer version's stats.** A channel whose media is on a drive that is not mounted (or is being moved) is left as it was instead of being read as a channel with no videos; a stats rebuild that has to start over refuses until the drive is back. A stats build refuses to clear stats written by a newer version of the editor; set `ARCHILYZER_STATS_ALLOW_DOWNGRADE=1` to roll back on purpose. Its log also says apart how many videos were downloaded since the last index build (they catch up after the next one) and how many the index skipped (no upload date, or it failed on them).
- **An index build keeps a channel whose drive is not mounted, instead of dropping it from the sites.** **Build index**, a site build's data phase and `archilyzer index` read a channel whose media is on a drive that is not mounted (or is being moved, or whose link and config disagree) as a channel with no videos: they removed its videos from the index, and the next site build published the channel as gone. Such a channel is now left as the last build had it — its videos stay in the index, its pages stay as they were, and the sites built next still list it — and the log names it, with its storage location: one line per channel, ` Held: N channel(s), K video(s) kept.` at the end of the `Diff:` line, and the channels again on the last line. A data folder that fails to read is held the same way, and a channel with no data folder at all is said in the log instead of passed over. An index rebuild that has to start over (after an update that changes the index's format, or with no index yet) refuses while any channel is held and says which; mount the drive first, or set `ARCHILYZER_INDEX_ALLOW_HELD=1` to rebuild without that channel until its drive is back and the index is built again — on the command for a command-line build (`ARCHILYZER_INDEX_ALLOW_HELD=1 pnpm archilyzer index`), or in the editor's own environment, with a restart, for **Build index** and the site builds started from the editor.
- **umtool's build no longer lists its e2e test data, the e2e server's build folder or `.env.local` among a route's files.** The clip-audio route named its cache files in a way the bundler read as a pattern reaching into umtool's hidden folders, so its list of files took in the e2e fixture (where the tests link the song data), the e2e dev server's build folder and the env file: 1,704 of its 2,167 entries. It now lists what the other routes list (463). Those folders and env files are also excluded from every route's list, and `pnpm test:scripts` reads the last umtool build's lists back and fails on any such entry. A checkout whose umtool build predates its code (this change included) skips that check, saying so, until umtool is rebuilt (`pnpm --filter umtool exec next build`). Nothing changes when umtool runs.
+- **A drive that stops answering no longer stops the editor answering.** When a storage location's drive is mounted but not answering (an SMR disk in a USB enclosure resetting under a long write), every page and poll that touched it waited on it, and a few such waits froze the whole editor until the drive came back. Every 15 seconds the editor now reads each location's disk activity counters from the kernel, which never waits on the drive: a disk with requests waiting and none finished since the last look is marked **Not answering**, and the mark comes off after two looks in a row find it working. Where no disk can be named (in a container, say) it asks the drive from a separate process with a 3-second limit instead. Any page or poll that reads the drive also gives up after 3 seconds and marks it the same way, and no more than four such reads wait on one drive at a time. While it is marked, the editor's pages and polls do not read that drive: `/storage` shows the location as **Not answering** with the time it stopped and how it is watched (**Refresh** asks the drive again), the `/channels` volume chip reads "not answering since HH:MM" and its channels' badges "not answering", their videos list, video pages and Cleanup stage say so instead of reading the drive, `/saved-videos` names the channels it did not read, and the index and stats builds keep those channels as they do for an unmounted drive, and a channel that is in the middle of a move still shows as moving. Jobs for those channels are refused until the drive answers, and a channel paused automatically for it says the drive is not answering rather than not there. The drive check keeps running on an editor started with `ARCHILYZER_IDLE_BOOT`. Pages and polls also reuse each channel's media check for 5 seconds. The editor's `start` script and the container now give Node 16 threads for file access instead of 4 (`UV_THREADPOOL_SIZE`); that buys time for reads already waiting on a drive, and a read that was already waiting when the drive stalled still waits until the drive answers.
- **Building the homepage now publishes the source: a read-only git mirror, its raw tree and a fresh tarball, behind a gate.** `archilyzer build homepage`, the `/sites` Homepage jobs and `pnpm ops build-homepage` run `archilyzer source publish` between compose and `next build`. It makes a fresh clone of the private `main` (the repository itself is never rewritten), rewrites that copy with git-filter-repo using your scrub rules (file contents and commit messages; your home directory becomes `/home/user` without a rule), and publishes it under `homepage/public` for `git clone https://archilyzer.pages.dev/source/archilyzer.git`, beside `/source/tree/` and the Downloads tarball. Before anything is written, every object of the rewritten history and every file about to be published is searched for every string you have denied; **one hit refuses the build**, and its log names the string only by where you wrote it (`denylist line 3 (len 5)`) and each hit by its object, field and byte offset — never a byte of the object. **A refusal withdraws the source**: the last publish is removed from `homepage/public` and the last build's copy from `homepage/out`, and **Deploy homepage refuses** a build whose source was not audited under today's rules and today's `main` ("run `archilyzer build homepage`, then deploy"). The rules live outside the repo, in `~/.config/archilyzer/source-scrub.txt` and `source-denylist.txt` (`ARCHILYZER_CONFIG_DIR`, `SOURCE_SCRUB_FILE`, `SOURCE_DENYLIST_FILE`); **without them the build refuses**, naming the missing file. **Put everything private in the denylist before any deploy, a preview included**: previews are public, and every deployment stays reachable at its own address until you delete it. Install git-filter-repo once (`pipx install git-filter-repo`; the editor's process needs `~/.local/bin` on its `PATH` to find it) — without it the build fetches it through `pipx run`, which needs the network — and gitleaks if you want its secret scan too. An unchanged `main` with unchanged rules is skipped, so a rebuild costs about 20 seconds only when something moved. A checkout with no git repository (the docker image, a tarball install) builds with the /source page's empty state. `archilyzer source publish --check` audits without writing, `archilyzer source audit <clone>/.git` checks any clone, `archilyzer build homepage --no-source` removes the published source instead, and `archilyzer doctor` reports the tools, the two files (rule counts and permissions, never their contents) and the last publish. `create-archives.sh` is gone. See PUBLISH.md, "The source mirror (homepage)".
- **umtool reads the corpus from its checkout (or `TRANSCRIPTS_DIR`), and the song project's data defaults to `~/.local/share/archilyzer/song`.** If yours is elsewhere, link it there before restarting umtool: `mkdir -p ~/.local/share/archilyzer && ln -s <where the data is> ~/.local/share/archilyzer/song` (the data stays where it is). With no `CHANNELS_DIR`, umtool reads the corpus at `$TRANSCRIPTS_DIR/channels`, else the checkout's own `transcripts/channels`; it used to fall back to an absolute path that existed on one machine only. The song project's videos default to `~/reports/quartering-uh-song/videos`; `SONG_DIR` and `VIDEO_ROOT` still win. The song project's tracked manifests record their paths relative to the song folders, and the twenty one-off `umtool/song/*.sh` run logs, which only ever ran on the machine that wrote them, are gone.
- **umtool's production build no longer reads the corpus folder.** Since umtool began finding the corpus from its checkout (the bullet above), `next build` treated the checkout's whole `transcripts/channels` as files to bundle. On a real archive it ran out of memory and was killed, so umtool could not be rebuilt. The build now ignores that folder and finishes in about 25 s at under 1 GB, the same as a checkout with no corpus. Nothing changes when umtool runs.
diff --git a/editor/app/api/channels/[slug]/videos/[id]/files/[name]/route.ts b/editor/app/api/channels/[slug]/videos/[id]/files/[name]/route.ts
@@ -4,6 +4,13 @@ import { stat } from "node:fs/promises";
import { NextResponse } from "next/server";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { makeSafeController } from "yt-dlp-transcript-common/lib/safeStreamController";
+import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
+import { channelMediaStall } from "yt-dlp-transcript-common/lib/channelMedia";
+import {
+ NOT_ANSWERING,
+ isDriveNotAnswering,
+ onDrive,
+} from "yt-dlp-transcript-common/lib/storageHealth";
export const dynamic = "force-dynamic";
@@ -123,10 +130,24 @@ export async function GET(
return NextResponse.json({ error: "Forbidden" }, { status: 403 });
}
+ // A drive that is not answering is not asked: the stat and the stream would
+ // each wait on it. 503, because it is a state that passes. The stat goes
+ // through the watchdog, so a drive that stops answering now is a 503 too;
+ // the stream that follows a stat that answered is not raced.
+ const notAnswering = () =>
+ NextResponse.json(
+ { error: `Media not read: ${NOT_ANSWERING}.` },
+ { status: 503, headers: { "retry-after": "15" } },
+ );
+ const channelConfig = await readChannelConfig(paths, slug);
+ if (channelMediaStall(channelConfig)) return notAnswering();
+ const drive = channelConfig?.dataDir?.trim();
+
let stats;
try {
- stats = await stat(fullPath);
- } catch {
+ stats = await (drive ? onDrive(drive, () => stat(fullPath)) : stat(fullPath));
+ } catch (err) {
+ if (isDriveNotAnswering(err)) return notAnswering();
return NextResponse.json({ error: "Not found" }, { status: 404 });
}
if (!stats.isFile()) {
diff --git a/editor/app/api/test/invalidate-cache/route.ts b/editor/app/api/test/invalidate-cache/route.ts
@@ -4,6 +4,8 @@ import { resetSnapshotScheduler } from "yt-dlp-transcript-common/jobs/snapshotSc
import { resetChannelSnapshotMemo } from "yt-dlp-transcript-common/controller/channels";
import { resetStorageProbeMemo } from "yt-dlp-transcript-common/controller/storageLocations";
import { resetVideoTitleMemo } from "yt-dlp-transcript-common/controller/videoTitles";
+import { forgetChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia";
+import { resetStorageHealth } from "yt-dlp-transcript-common/lib/storageHealth";
import { testRouteDenied } from "../_guard";
export const dynamic = "force-dynamic";
@@ -112,6 +114,12 @@ function invalidate() {
// testInfo.outputPath varies by accident rather than on purpose; cleared here
// so it is on purpose.
resetStorageProbeMemo();
+ // And the two memories a stalled drive lives in: inspectChannelMedia's
+ // five-second answers (keyed by channels dir, slug and the configured target,
+ // which a reset fixture reproduces exactly) and each location's health, which
+ // a previous spec's location id would otherwise carry into this one.
+ forgetChannelMedia();
+ resetStorageHealth();
// And the video list's metadata.info.json title memo. It is keyed by the
// channel's data/ mtime, which resetData() changes by recreating the dir — but
// a spec that rewrites a title in place inside one mtime tick would otherwise
diff --git a/editor/app/channels/[slug]/components/MediaNotAnswering.tsx b/editor/app/channels/[slug]/components/MediaNotAnswering.tsx
@@ -0,0 +1,67 @@
+import Link from "next/link";
+import {
+ notAnsweringText,
+ type LocationHealth,
+} from "yt-dlp-transcript-common/lib/storageHealth";
+
+// WHAT A PAGE THAT READS A CHANNEL'S `data/` SHOWS WHILE ITS DRIVE IS NOT
+// ANSWERING, instead of reading it.
+//
+// The videos list and the video page read the channel's media directory on
+// every render — a readdir, a stat per file, a metadata head read per title. On
+// a drive that is mounted and not answering (lib/storageHealth.ts) each of those
+// calls waits for the drive, on one of the few threads every page and poll in
+// this process shares. So the page asks the health state first, and on a
+// stalled drive it says so and reads nothing. The rest of the channel (its
+// overview, its report, its settings) comes off the corpus disk and still
+// renders.
+//
+// Server-only: it takes the health entry from the page, which read it from
+// memory — or from the DriveNotAnsweringError a page's read got from `onDrive`'s
+// watchdog.
+export function MediaNotAnswering({
+ slug,
+ stall,
+ what,
+}: {
+ slug: string;
+ // The stalled location, or null when the drive is on no location the health
+ // state knows and a read of it did not answer within 3 s.
+ stall: LocationHealth | null;
+ // What would have been shown: "The video list", "This video".
+ what: string;
+}) {
+ return (
+ <div className="flex flex-col gap-3">
+ <p
+ role="status"
+ aria-label="media not answering"
+ className="rounded border border-destructive/50 bg-destructive/5 px-3 py-2 text-sm text-destructive"
+ >
+ {stall ? (
+ <>
+ {what} reads this channel's media, which is on “
+ {stall.label}” — a drive that is {notAnsweringText(stall)}.
+ Nothing is read from it until it answers again (it is checked every
+ 15 s).
+ </>
+ ) : (
+ <>
+ {what} reads this channel's media, and a read of its drive did
+ not answer within 3 s. Nothing more is read from it on this page.
+ </>
+ )}
+ </p>
+ <p className="text-sm text-muted-foreground">
+ <Link href={`/channels/${slug}`} className="underline hover:text-foreground">
+ The channel
+ </Link>{" "}
+ still shows its report, and{" "}
+ <Link href="/storage" className="underline hover:text-foreground">
+ Storage
+ </Link>{" "}
+ shows the drive.
+ </p>
+ </div>
+ );
+}
diff --git a/editor/app/channels/[slug]/components/stages/StorageStage.tsx b/editor/app/channels/[slug]/components/stages/StorageStage.tsx
@@ -93,7 +93,8 @@ type Props = {
clipsBytes: number | null;
// Free space on the volume the media is on RIGHT NOW — the platter for a
// relocated channel, the corpus disk otherwise.
- freeBytes: number;
+ // Null when the drive did not answer the statfs ("—").
+ freeBytes: number | null;
volumeDir: string;
// Why both buttons are off, or null when they are live. Running/queued jobs
// for this channel, or a relocation marker left by an interrupted move.
@@ -151,7 +152,7 @@ export function StorageStage({
</dd>
<dt className="text-muted-foreground">Free on that volume</dt>
<dd aria-label="free on media volume">
- {formatBytes(freeBytes)}{" "}
+ {freeBytes === null ? "—" : formatBytes(freeBytes)}{" "}
<span className="text-xs text-muted-foreground font-mono">
({volumeDir})
</span>
@@ -214,7 +215,8 @@ export function StorageStage({
// a reason, one click earlier.
unvouched={
location.status === "inconsistent" ||
- location.status === "unreachable"
+ location.status === "unreachable" ||
+ location.status === "stalled"
? `This channel's media location is ${location.status}: ${
location.detail ?? "disk and config do not agree"
} Moving back would delete the relocated copy, so it is refused until the location reads "relocated · reachable".`
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -40,7 +40,15 @@ import {
type ShardOp,
} from "yt-dlp-transcript-common/controller/shard";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { inspectChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia";
+import {
+ channelMediaStall,
+ inspectChannelMedia,
+} from "yt-dlp-transcript-common/lib/channelMedia";
+import { MediaNotAnswering } from "./components/MediaNotAnswering";
+import {
+ isDriveNotAnswering,
+ onDrive,
+} from "yt-dlp-transcript-common/lib/storageHealth";
import { getFreeBytes } from "yt-dlp-transcript-common/lib/diskSpace";
import {
platformQueueKey,
@@ -515,12 +523,34 @@ export default async function ChannelDetailPage({
/>
);
case "cleanup": {
+ // The saved-video summary below reads a pointer in every video dir, and
+ // this stage's actions all act on the media: on a drive that is not
+ // answering the stage says so instead (see MediaNotAnswering).
+ const stall = channelMediaStall(config);
+ if (stall) {
+ return (
+ <MediaNotAnswering slug={slug} stall={stall} what="The Cleanup stage" />
+ );
+ }
// Saved-video store summary for this channel + whether backups are
- // configured, for the Retention & persistence section.
+ // configured, for the Retention & persistence section. Its reads go
+ // through the watchdog (`notAnswering`): a drive that stops answering
+ // on the way is named there, and the stage says so.
+ const notAnswering: string[] = [];
const savedTotals = await savedVideoTotals({
paths,
channelSlug: slug,
+ notAnswering,
});
+ if (notAnswering.length > 0) {
+ return (
+ <MediaNotAnswering
+ slug={slug}
+ stall={channelMediaStall(config)}
+ what="The Cleanup stage"
+ />
+ );
+ }
return (
<CleanupStage
slug={slug}
@@ -586,7 +616,19 @@ export default async function ChannelDetailPage({
// marker — reading the file again here was a second read of the same
// bytes that could disagree with the status rendered beside it.
const marker = media.marker ?? null;
- const freeBytes = await getFreeBytes(volumeDir);
+ // A statfs of the channel's drive goes through the watchdog
+ // (lib/storageHealth.ts): refused on a stalled location, given up on
+ // after 3 s, and read "—" either way.
+ let freeBytes: number | null;
+ try {
+ freeBytes =
+ volumeDir === media.target && media.target
+ ? await onDrive(media.target, () => getFreeBytes(volumeDir))
+ : await getFreeBytes(volumeDir);
+ } catch (err) {
+ if (!isDriveNotAnswering(err)) throw err;
+ freeBytes = null;
+ }
// The SAME two conditions storageActions.ts refuses on, stated here as
// prose so the button is off with a reason rather than off and silent —
// and stated in the action too, because a disabled button is a courtesy
diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx
@@ -42,6 +42,12 @@ import { TagsPanel } from "./components/TagsPanel";
import { MetadataHistoryDetails } from "./components/MetadataHistoryDetails";
import { loadVideoTags } from "./lib/videoTags";
import { loadVideoOperationPanels } from "./lib/videoOperationPanels";
+import { channelMediaStall } from "yt-dlp-transcript-common/lib/channelMedia";
+import {
+ isDriveNotAnswering,
+ onDrive,
+} from "yt-dlp-transcript-common/lib/storageHealth";
+import { MediaNotAnswering } from "../../components/MediaNotAnswering";
export const dynamic = "force-dynamic";
@@ -80,8 +86,18 @@ export async function generateMetadata({
params: Promise<{ slug: string; id: string }>;
}): Promise<Metadata> {
const { slug, id } = await params;
- const meta = await loadMeta(slug, id);
- const subject = meta.title ?? id;
+ // The title is read off the drive; a drive that is not answering is not
+ // asked, and one that does not answer in 3 s is given up on.
+ const drive = (await readChannelConfig(getPaths(), slug))?.dataDir?.trim();
+ let subject = id;
+ try {
+ const meta = await (drive
+ ? onDrive(drive, () => loadMeta(slug, id))
+ : loadMeta(slug, id));
+ subject = meta.title ?? id;
+ } catch (err) {
+ if (!isDriveNotAnswering(err)) throw err;
+ }
return { title: `${subject} — Video — ${slug}` };
}
@@ -93,63 +109,116 @@ export default async function VideoDetailPage({
const { slug, id } = await params;
const config = await readChannelConfig(getPaths(), slug);
if (!config) notFound();
- const dirData = await loadVideoDir(slug, id);
- const meta = await loadMeta(slug, id);
- const videoDir = path.join(getPaths().channelsDir, slug, "data", id);
- const downloadOutcome = await loadDownloadOutcome(videoDir);
- const availabilityRecord = await loadAvailability(videoDir);
- const availabilityHistory = availabilityRecord?.history ?? [];
- // Every rewrite of metadata.info.json that changed its bytes (release 10
- // slice N). One small bounded file, read once; null → nothing drawn.
- const metadataHistory = metadataHistoryView(
- await loadMetadataHistory(videoDir),
- Date.now(),
- );
- const doNotClean = await isDoNotClean(videoDir);
- const excludedFromTruncatedCheck =
- await isExcludedFromTruncatedCheck(videoDir);
- const savedVideo = await loadSavedVideo(videoDir);
- // The windows another tool asked this editor to fetch. One readdir of
- // data/<id>/clips/ plus a stat per file — and no per-CHANNEL count anywhere,
- // because that would be a walk of every video dir to draw one number.
- const clipWindows = await listClipWindows(videoDir);
- // Where each subtitle track came from, so the panel can say "YouTube
- // auto-captions" vs "manual captions" — and offer to replace the former with a
- // transcript of our own. One 4 KB head read per VTT (see subtitleProvenance).
- const vttProvenance: Record<string, SubtitleProvenance> = {};
- for (const f of dirData.files) {
- if (!isTranscriptVtt(f.name)) continue;
- vttProvenance[f.name] = await resolveVttProvenance(videoDir, f.name);
+ // Everything below reads the video's directory on the channel's drive; a
+ // drive that is not answering is not read. See MediaNotAnswering.
+ const stall = channelMediaStall(config);
+ if (stall) {
+ return <MediaNotAnswering slug={slug} stall={stall} what="This video's page" />;
}
- const cov = await readTranscriptCoverage(videoDir);
- const coverage = cov
- ? {
- lastCueEnd: cov.cov.lastCueEnd,
- duration: cov.cov.duration,
- coverage: cov.cov.coverage,
- incomplete:
- !excludedFromTruncatedCheck &&
- isIncompleteTranscript(cov.cov, {
- isLivestream: cov.isLivestream,
- }),
- }
- : null;
+ // EVERY READ BELOW IS OF THIS VIDEO'S DIRECTORY, on the channel's drive when
+ // it is relocated, so they go through the watchdog as one unit: a drive that
+ // has not answered them in 3 s is marked stalled and the page says so.
+ const loadAll = async () => {
+ const dirData = await loadVideoDir(slug, id);
+ const meta = await loadMeta(slug, id);
+ const videoDir = path.join(getPaths().channelsDir, slug, "data", id);
+ const downloadOutcome = await loadDownloadOutcome(videoDir);
+ const availabilityRecord = await loadAvailability(videoDir);
+ const availabilityHistory = availabilityRecord?.history ?? [];
+ // Every rewrite of metadata.info.json that changed its bytes (release 10
+ // slice N). One small bounded file, read once; null → nothing drawn.
+ const metadataHistory = metadataHistoryView(
+ await loadMetadataHistory(videoDir),
+ Date.now(),
+ );
+ const doNotClean = await isDoNotClean(videoDir);
+ const excludedFromTruncatedCheck =
+ await isExcludedFromTruncatedCheck(videoDir);
+ const savedVideo = await loadSavedVideo(videoDir);
+ // The windows another tool asked this editor to fetch. One readdir of
+ // data/<id>/clips/ plus a stat per file — and no per-CHANNEL count anywhere,
+ // because that would be a walk of every video dir to draw one number.
+ const clipWindows = await listClipWindows(videoDir);
+ // Where each subtitle track came from, so the panel can say "YouTube
+ // auto-captions" vs "manual captions" — and offer to replace the former with a
+ // transcript of our own. One 4 KB head read per VTT (see subtitleProvenance).
+ const vttProvenance: Record<string, SubtitleProvenance> = {};
+ for (const f of dirData.files) {
+ if (!isTranscriptVtt(f.name)) continue;
+ vttProvenance[f.name] = await resolveVttProvenance(videoDir, f.name);
+ }
+ const cov = await readTranscriptCoverage(videoDir);
+ const coverage = cov
+ ? {
+ lastCueEnd: cov.cov.lastCueEnd,
+ duration: cov.cov.duration,
+ coverage: cov.cov.coverage,
+ incomplete:
+ !excludedFromTruncatedCheck &&
+ isIncompleteTranscript(cov.cov, {
+ isLivestream: cov.isLivestream,
+ }),
+ }
+ : null;
- // ONE PANEL PER REGISTRY OPERATION, each carrying the state that entry's own
- // state() reports. The page used to re-derive the digest's freshness here —
- // its own target resolution and its own per-section fold beside the
- // registry's — which is how a page and a work list end up describing the same
- // disk differently.
- const panels = await loadVideoOperationPanels({
- paths: getPaths(),
- channelSlug: slug,
- videoId: id,
- settings: getSettings(),
- });
+ // ONE PANEL PER REGISTRY OPERATION, each carrying the state that entry's own
+ // state() reports. The page used to re-derive the digest's freshness here —
+ // its own target resolution and its own per-section fold beside the
+ // registry's — which is how a page and a work list end up describing the same
+ // disk differently.
+ const panels = await loadVideoOperationPanels({
+ paths: getPaths(),
+ channelSlug: slug,
+ videoId: id,
+ settings: getSettings(),
+ });
- // The curated tags on this video, with each one's provenance. Reads tags.json
- // plus (for rule hits) ONE key out of the transcript index — not a scan.
- const tagsView = await loadVideoTags(slug, id);
+ // The curated tags on this video, with each one's provenance. Reads tags.json
+ // plus (for rule hits) ONE key out of the transcript index — not a scan.
+ const tagsView = await loadVideoTags(slug, id);
+ return {
+ dirData,
+ meta,
+ videoDir,
+ downloadOutcome,
+ availabilityHistory,
+ metadataHistory,
+ doNotClean,
+ excludedFromTruncatedCheck,
+ savedVideo,
+ clipWindows,
+ vttProvenance,
+ coverage,
+ panels,
+ tagsView,
+ };
+ };
+ const drive = config.dataDir?.trim();
+ let loaded: Awaited<ReturnType<typeof loadAll>>;
+ try {
+ loaded = await (drive ? onDrive(drive, loadAll) : loadAll());
+ } catch (err) {
+ if (!isDriveNotAnswering(err)) throw err;
+ return (
+ <MediaNotAnswering slug={slug} stall={err.health} what="This video's page" />
+ );
+ }
+ const {
+ dirData,
+ meta,
+ videoDir,
+ downloadOutcome,
+ availabilityHistory,
+ metadataHistory,
+ doNotClean,
+ excludedFromTruncatedCheck,
+ savedVideo,
+ clipWindows,
+ vttProvenance,
+ coverage,
+ panels,
+ tagsView,
+ } = loaded;
const registry = getRegistry();
const existingQueues = registry.activeQueueNames();
diff --git a/editor/app/channels/[slug]/videos/page.tsx b/editor/app/channels/[slug]/videos/page.tsx
@@ -36,6 +36,12 @@ import { computeVideoRows, readDataDirVideoIds } from "../lib/videoRowsServer";
import { normalizeBuckets } from "yt-dlp-transcript-common/views/pipeline/stageStatus";
import { attachCuratedTags } from "../lib/videoTagRows";
import { readChannelVideoTitles } from "yt-dlp-transcript-common/controller/videoTitles";
+import { channelMediaStall } from "yt-dlp-transcript-common/lib/channelMedia";
+import {
+ isDriveNotAnswering,
+ onDrive,
+} from "yt-dlp-transcript-common/lib/storageHealth";
+import { MediaNotAnswering } from "../components/MediaNotAnswering";
export const dynamic = "force-dynamic";
@@ -100,6 +106,13 @@ export default async function ChannelVideosPage({
// A social channel has posts, not videos — there is no data directory to list
// and nothing here would render. 404 rather than an empty workspace.
if (isSocialChannel(config)) notFound();
+ // THE LIST IS READ OFF THE DRIVE (a readdir of data/, a head read per title,
+ // the selected video's files), so a drive that is not answering is not read:
+ // the page says so instead. See MediaNotAnswering.
+ const stall = channelMediaStall(config);
+ if (stall) {
+ return <MediaNotAnswering slug={slug} stall={stall} what="The video list" />;
+ }
const registry = getRegistry();
const existingQueues = registry.activeQueueNames();
@@ -131,14 +144,30 @@ export default async function ChannelVideosPage({
);
const channelDataDir = path.join(paths.channelsDir, slug, "data");
- const channelDataDirIds = await readDataDirVideoIds(channelDataDir);
- // What each video is called: one key range over the transcript index, one
- // read of the channel's metadata-scan.json, and a head read of
- // metadata.info.json only for what those two did not name. Measured at
- // ~80 ms for a synthetic 5,000-id channel (plans/release-8.md, slice V).
- const titles = await readChannelVideoTitles(paths, slug, [
- ...new Set([...channelDataDirIds, ...(snapshot.undownloadedIds ?? [])]),
- ]);
+ // THE READS OF THE DRIVE go through the watchdog when the channel is
+ // relocated: a drive that has not answered them in 3 s is marked stalled and
+ // the page says so instead (see MediaNotAnswering).
+ const drive = config.dataDir?.trim();
+ const onMedia = <T,>(call: () => Promise<T>): Promise<T> =>
+ drive ? onDrive(drive, call) : call();
+ let channelDataDirIds: string[];
+ let titles: Awaited<ReturnType<typeof readChannelVideoTitles>>;
+ try {
+ [channelDataDirIds, titles] = await onMedia(async () => {
+ const ids = await readDataDirVideoIds(channelDataDir);
+ // What each video is called: one key range over the transcript index,
+ // one read of the channel's metadata-scan.json, and a head read of
+ // metadata.info.json only for what those two did not name. Measured at
+ // ~80 ms for a synthetic 5,000-id channel (plans/release-8.md, slice V).
+ const named = await readChannelVideoTitles(paths, slug, [
+ ...new Set([...ids, ...(snapshot.undownloadedIds ?? [])]),
+ ]);
+ return [ids, named] as const;
+ });
+ } catch (err) {
+ if (!isDriveNotAnswering(err)) throw err;
+ return <MediaNotAnswering slug={slug} stall={err.health} what="The video list" />;
+ }
const rows = computeVideoRows({
channelDataDirIds,
snapshot,
@@ -188,8 +217,9 @@ export default async function ChannelVideosPage({
);
if (selectedVideoId) {
const videoDir = path.join(channelDataDir, selectedVideoId);
- const [dirData, title, outcome, availabilityRecord, cov, excludedTrunc] =
- await Promise.all([
+ let loadedVideo;
+ try {
+ loadedVideo = await onMedia(() => Promise.all([
loadVideoDir(channelDataDir, selectedVideoId),
// Already read for the list — the title map covers every row.
Promise.resolve(titles.get(selectedVideoId)?.title ?? null),
@@ -197,7 +227,13 @@ export default async function ChannelVideosPage({
loadAvailability(videoDir),
readTranscriptCoverage(videoDir),
isExcludedFromTruncatedCheck(videoDir),
- ]);
+ ]));
+ } catch (err) {
+ if (!isDriveNotAnswering(err)) throw err;
+ return <MediaNotAnswering slug={slug} stall={err.health} what="The video list" />;
+ }
+ const [dirData, title, outcome, availabilityRecord, cov, excludedTrunc] =
+ loadedVideo;
const coverage = cov
? {
lastCueEnd: cov.cov.lastCueEnd,
diff --git a/editor/app/channels/components/ChannelVolumeBar.tsx b/editor/app/channels/components/ChannelVolumeBar.tsx
@@ -42,6 +42,9 @@ export type ChannelVolume = {
// statfs of the root, or undefined when the root is not there (an unmounted
// drive). Never inferred from a parent — see volumeFreeBytes.
freeBytes?: number;
+ // "not answering since 11:35", while the health probe finds the drive not
+ // answering (its free space is then not asked either).
+ notAnswering?: string;
};
export function ChannelVolumeBar({
@@ -96,10 +99,14 @@ export function ChannelVolumeBar({
detail={
`${v.channels} ch · ${formatBytes(v.bytes)}` +
(v.unmeasured > 0 ? ` +${v.unmeasured}?` : "") +
- ` · ${v.freeBytes === undefined ? "free —" : `${formatBytes(v.freeBytes)} free`}`
+ (v.notAnswering
+ ? ` · ${v.notAnswering}`
+ : ` · ${v.freeBytes === undefined ? "free —" : `${formatBytes(v.freeBytes)} free`}`)
}
title={
- v.freeBytes === undefined
+ v.notAnswering
+ ? `${v.label}: the drive is ${v.notAnswering}. Pages and polls do not touch it until it answers twice in a row; its channels read "not answering".`
+ : v.freeBytes === undefined
? `${v.label}: the root is not there — an unmounted drive reports no free space rather than its parent's.`
: `${v.label}: ${v.channels} channel(s) hold ${formatBytes(v.bytes)}${
v.unmeasured > 0
diff --git a/editor/app/channels/page.tsx b/editor/app/channels/page.tsx
@@ -8,6 +8,10 @@ import {
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { inspectChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia";
import {
+ notAnsweringText,
+ stalledLocation,
+} from "yt-dlp-transcript-common/lib/storageHealth";
+import {
getSite,
listSiteIds,
listSites,
@@ -220,6 +224,14 @@ export default async function ChannelsPage({
// TWO SYSCALLS PER VOLUME, NOT A PROBE. See volumeFreeBytes: this table draws
// 71 rows on every auto-refresh and the rule is that tables never shell out.
const freeByVolume = await volumeFreeBytes({ paths, locations });
+ // A DRIVE THAT IS NOT ANSWERING, said on its chip. From memory — the health
+ // pass (the block device's counters every 15 s) or the watchdog on a read is
+ // what found it (lib/storageHealth.ts); the table asks nothing.
+ const notAnsweringByVolume: Record<string, string | undefined> = {};
+ for (const loc of locations) {
+ const stall = stalledLocation(loc);
+ if (stall) notAnsweringByVolume[loc.id] = notAnsweringText(stall);
+ }
// ONE ROW PER CHANNEL, off the shared builder (common/views/channelRow.ts):
// the dashboard and the operation pages build theirs the same way. The view
// carries no `config` — this table is a client component, and a channel
@@ -292,6 +304,9 @@ export default async function ChannelsPage({
bytes: measured.reduce((sum, c) => sum + (c.mediaBytes ?? 0), 0),
unmeasured: rows.length - measured.length,
freeBytes: freeByVolume[id],
+ ...(notAnsweringByVolume[id]
+ ? { notAnswering: notAnsweringByVolume[id] }
+ : {}),
};
})
// The unnamed-root chip only exists when something is actually on one.
diff --git a/editor/app/components/MediaLocationBadge.tsx b/editor/app/components/MediaLocationBadge.tsx
@@ -10,18 +10,19 @@ import type { ChannelRowMedia } from "yt-dlp-transcript-common/views/channelRow"
// the filesystem; the inspect() call that produces the location happens on the
// server, once per row, and only its result travels.
//
-// WHAT THE FIVE STATUSES LOOK LIKE, and why there are only three appearances:
+// WHAT THE SIX STATUSES LOOK LIKE, and why there are only three appearances:
//
// in-place → NOTHING. The overwhelming majority of channels are in place,
// and a badge on every row saying "normal" is noise that makes
// the two that matter harder to see, not easier.
// ok → neutral. Relocated and reachable is a fact worth stating (the
// bytes are not on the corpus disk) but it is not a problem.
-// everything → red. unreachable, in-transition and inconsistent are all
-// else "do not trust what this channel's dirs say right now": the
+// everything → red. unreachable, in-transition, inconsistent and stalled are
+// else all "do not trust what this channel's dirs say right now": the
// first because the drive is not mounted, the second because a
// move is half-done, the third because disk and config disagree
-// and nothing here is willing to guess which one is right.
+// and nothing here is willing to guess which one is right, the
+// fourth because the drive is mounted and not answering.
//
// The `detail` string is the operator's prose from inspect() — the drive path,
// the phase, the disagreement — and it goes on `title` so a row badge carries
@@ -52,6 +53,7 @@ const LABELS: Record<ChannelMediaStatus, string | null> = {
unreachable: "Media unreachable",
"in-transition": "Media moving",
inconsistent: "Media inconsistent",
+ stalled: "Media not answering",
};
// The one-word state, for the compact rendering of a NAMED location: "on
@@ -63,6 +65,7 @@ const SHORT_STATUS: Record<ChannelMediaStatus, string | null> = {
unreachable: "unreachable",
"in-transition": "moving",
inconsistent: "inconsistent",
+ stalled: "not answering",
};
// Null means "draw nothing" — an in-place channel, or no location at all (a
diff --git a/editor/app/saved-videos/page.tsx b/editor/app/saved-videos/page.tsx
@@ -22,7 +22,11 @@ export default async function SavedVideosPage() {
const paths = getPaths();
const settings = getSettings();
const backup = settings.savedVideoBackup;
- const entries = await listSavedVideos({ paths });
+ // Channels whose drive is not answering are not read (one pointer read per
+ // video dir, each of which would wait on it); the page names them.
+ const notAnswering: string[] = [];
+ const entries = await listSavedVideos({ paths, notAnswering });
+ notAnswering.sort();
const scheduler = await readSchedulerState(paths);
const byChannel = new Map<string, ChannelSummary>();
@@ -96,6 +100,21 @@ export default async function SavedVideosPage() {
<section aria-label="per-channel saved videos" className="flex flex-col gap-2">
<h2 className="text-base font-semibold">By channel</h2>
+ {notAnswering.length > 0 && (
+ <p
+ role="status"
+ aria-label="saved videos not read"
+ className="rounded border border-destructive/50 bg-destructive/5 px-3 py-2 text-sm text-destructive"
+ >
+ Not read, because the drive their media is on is not answering:{" "}
+ {notAnswering.join(", ")}. Their saved videos are not in the counts
+ above until it answers again (see{" "}
+ <Link href="/storage" className="underline">
+ Storage
+ </Link>
+ ).
+ </p>
+ )}
{channels.length === 0 ? (
<p className="text-sm text-muted-foreground">
No saved videos yet. Set a channel's keep-latest window, then run
diff --git a/editor/app/storage/actions.ts b/editor/app/storage/actions.ts
@@ -33,6 +33,12 @@ import { enqueueRepointJob } from "./lib/repointJob";
import { enqueueSavedVideosRelocation } from "./lib/savedVideosJob";
import { enqueueEvictClipWindows } from "./lib/evictClipsJob";
import { savedVideosStoreBusyReason } from "./lib/storeBusy";
+import { refreshLocationHealth } from "yt-dlp-transcript-common/controller/storageWatch";
+import { forgetChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia";
+import {
+ locationHealth,
+ notAnsweringText,
+} from "yt-dlp-transcript-common/lib/storageHealth";
// THE SIX THINGS AN OPERATOR MAY DO TO A STORAGE LOCATION.
//
@@ -244,6 +250,24 @@ export async function refreshStorageLocationAction(
const settings = getSettings();
const location = settings.storage.locations.find((l) => l.id === id);
if (!location) return { ok: false, error: `There is no storage location "${id}".` };
+ // IS IT ANSWERING, asked first and without touching the drive (the block
+ // device's counters in /sys, or a child `stat` raced against 3 s where no
+ // device can be named): the probe below runs in-process, and on a stalled
+ // drive it is not asked at all. The counters give no answer within 10 s of
+ // the pass's last sample. The channels' remembered answers go too — the
+ // operator has just done something about the drive.
+ await refreshLocationHealth(location);
+ forgetChannelMedia();
+ const health = locationHealth(location.id);
+ if (health?.state === "stalled") {
+ revalidateStorage();
+ return {
+ ok: true,
+ note:
+ `${location.label}: ${notAnsweringText(health)} — ${health.cause ?? "its root did not answer"}. ` +
+ `Pages skip this drive until it answers twice in a row (checked every 15 s).`,
+ };
+ }
const probe = await probeLocationMemo(location, paths, { refresh: true });
const wrote = await recordProbedIdentity({ locationId: id, probe });
diff --git a/editor/app/storage/buildStorage.ts b/editor/app/storage/buildStorage.ts
@@ -3,6 +3,12 @@ import { getSettings } from "yt-dlp-transcript-common/lib/settings";
import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
import { getFreeBytes } from "yt-dlp-transcript-common/lib/diskSpace";
import { udisksctlAvailable } from "yt-dlp-transcript-common/lib/storageVolumes";
+import {
+ isDriveNotAnswering,
+ notAnsweringText,
+ onDrive,
+ stalledLocation,
+} from "yt-dlp-transcript-common/lib/storageHealth";
import { listChannelBriefs } from "yt-dlp-transcript-common/controller/channels";
import {
channelsOnLocation,
@@ -81,11 +87,40 @@ export async function buildStorage(): Promise<StorageRowsPayload> {
// situation the operator opened the page to understand. The store's SIZE is
// the only thing cached; its location, status and marker are read fresh every
// render, because those are the safety facts.
+ // THE DRIVES THAT ARE NOT ANSWERING, in words. From memory: the health pass
+ // (the block device's counters every 15 s) or the watchdog on a page's read
+ // is what found it (lib/storageHealth.ts).
+ const now = Date.now();
+ const notAnswering: Record<string, string> = {};
+ for (const loc of locations) {
+ const stall = stalledLocation(loc);
+ if (stall) {
+ const watched =
+ stall.detector === "counters"
+ ? " Watched through its disk's request counters."
+ : stall.detector === "stat"
+ ? " Watched with a stat of its root (no disk could be named here)."
+ : "";
+ notAnswering[loc.id] =
+ `${notAnsweringText(stall, now)} — ${stall.cause ?? "its root did not answer"}. ` +
+ `Pages and polls skip this drive until it answers twice in a row.${watched}`;
+ }
+ }
const store = await inspectSavedVideosStore(paths, settings);
- const storeMeasured =
- store.status === "unreachable" || store.status === "in-transition"
- ? { bytes: 0, files: 0 }
- : await measureTreeCached(paths.savedVideosDir);
+ // A store on a location walks that location's drive: through the watchdog,
+ // so a drive that stops answering mid-walk leaves the size at 0 (the store's
+ // status line says why on the next render) instead of holding the page.
+ const storeLocation = locations.find((l) => l.id === store.locationId);
+ let storeMeasured = { bytes: 0, files: 0 };
+ if (store.status !== "unreachable" && store.status !== "in-transition") {
+ try {
+ storeMeasured = await (storeLocation
+ ? onDrive(storeLocation, () => measureTreeCached(paths.savedVideosDir))
+ : measureTreeCached(paths.savedVideosDir));
+ } catch (err) {
+ if (!isDriveNotAnswering(err)) throw err;
+ }
+ }
return buildStorageRows({
locations,
savedVideos: {
@@ -106,9 +141,10 @@ export async function buildStorage(): Promise<StorageRowsPayload> {
},
defaultLocationId: settings.storage.defaultLocationId,
probes,
+ notAnswering,
rollups,
registry: getRegistry(),
udisksctlAvailable: udisksctl,
- now: Date.now(),
+ now,
});
}
diff --git a/editor/app/storage/components/StorageLocationsTable.tsx b/editor/app/storage/components/StorageLocationsTable.tsx
@@ -254,6 +254,15 @@ function LocationCard({
<dd aria-label="location last probe">{formatAge(row.lastProbeAgeMs)}</dd>
</dl>
+ {row.notAnswering && (
+ <p
+ role="status"
+ aria-label="location not answering"
+ className="text-xs rounded border border-destructive/50 bg-destructive/5 px-3 py-2 text-destructive"
+ >
+ {row.notAnswering}
+ </p>
+ )}
{row.warning && (
<p
role="status"
diff --git a/editor/instrumentation.ts b/editor/instrumentation.ts
@@ -4,8 +4,10 @@
//
// See editor/app/scheduler/heartbeat.ts and SCHEDULED_SYNC.md.
//
-// Everything armed here except the shutdown reaper is skipped when the process
-// boots idle (ARCHILYZER_IDLE_BOOT) — see isIdleBoot below.
+// Everything armed here that starts or writes work is skipped when the process
+// boots idle (ARCHILYZER_IDLE_BOOT) — see isIdleBoot below. What stays armed
+// only stops work or only reads: the shutdown reaper, the persisted-pause
+// restore, the storage boot probe and the drive health pass.
//
// The one STATIC import in this file, and safe as one because idleBoot.ts
// imports nothing and touches no Node API: the Edge bundle's static Node-API
@@ -83,8 +85,25 @@ export async function register() {
/* a failed re-pause must not block server readiness */
}
- // ONE PASS OVER THE STORAGE LOCATIONS, and it runs on an IDLE BOOT TOO —
- // the only thing below the reaper that does.
+ // THE DRIVE HEALTH PASS, every 15 s — and ON AN IDLE BOOT TOO, like the
+ // probe below. It reads each location's block device counters (never the
+ // drive) and keeps, in memory only, which drives are not answering; every
+ // page and poll asks it before touching a drive, and its watchdog marks a
+ // drive a page reached and got no answer from. It writes nothing and starts
+ // no work, and without it nothing would ever clear such a mark. The
+ // five-minute pass that may auto-pause channels is armed below the idle gate.
+ // See common/controller/storageWatch.ts and common/lib/storageHealth.ts.
+ try {
+ const { startStorageHealthWatch } = await import(
+ "yt-dlp-transcript-common/controller/storageWatch"
+ );
+ startStorageHealthWatch({ log: (line) => console.log(line) });
+ } catch {
+ /* a health pass that fails to arm must not block server readiness */
+ }
+
+ // ONE PASS OVER THE STORAGE LOCATIONS, and it runs on an IDLE BOOT TOO: it
+ // only reads, like the health pass above.
//
// The pass probes each location (is the disk here, and if not, where?) and,
// for a location the operator armed with `autoRepoint`, re-points it to
@@ -165,10 +184,11 @@ export async function register() {
// is the thing that looks, on a five-minute cadence, and auto-pauses (and
// later restores) the channels on a location that is not there.
//
- // BELOW THE IDLE GATE, deliberately, and unlike the boot probe above: this
- // one WRITES settings.channelPriority, and a container pointed at somebody
- // else's corpus for the first time has no business rewriting that corpus's
- // priority document. See common/controller/storageWatch.ts.
+ // BELOW THE IDLE GATE, deliberately, and unlike the boot probe and the
+ // health pass above: this one WRITES settings.channelPriority, and a
+ // container pointed at somebody else's corpus for the first time has no
+ // business rewriting that corpus's priority document. See
+ // common/controller/storageWatch.ts.
try {
const { startStorageWatch } = await import(
"yt-dlp-transcript-common/controller/storageWatch"
diff --git a/editor/package.json b/editor/package.json
@@ -8,7 +8,7 @@
"dev:test": "WORKER_TOKEN=test-worker-token TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json EDITOR_CHANGELOG_FILE=$(pwd)/test-changelog.md EXPORT_CHANGELOG_FILE=$(pwd)/test-export-changelog.md YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs GALLERY_DL_BIN=$(pwd)/e2e/fixtures/bin/fake-gallery-dl.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null DIARIZE_BIN=$(pwd)/e2e/fixtures/bin/fake-diarize.mjs FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs FFPROBE_BIN=$(pwd)/e2e/fixtures/bin/fake-ffprobe.mjs OLLAMA_URL=http://127.0.0.1:${OLLAMA_STUB_PORT:-11435} CLAUDE_BIN=$(pwd)/e2e/fixtures/bin/fake-claude.mjs FINDMNT_BIN=$(pwd)/e2e/fixtures/bin/fake-findmnt.mjs UDISKSCTL_BIN=$(pwd)/e2e/fixtures/bin/fake-udisksctl.mjs next dev --port ${PORT:-3011}",
"start:test": "WORKER_TOKEN=test-worker-token TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json EDITOR_CHANGELOG_FILE=$(pwd)/test-changelog.md EXPORT_CHANGELOG_FILE=$(pwd)/test-export-changelog.md YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs GALLERY_DL_BIN=$(pwd)/e2e/fixtures/bin/fake-gallery-dl.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null DIARIZE_BIN=$(pwd)/e2e/fixtures/bin/fake-diarize.mjs FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs FFPROBE_BIN=$(pwd)/e2e/fixtures/bin/fake-ffprobe.mjs OLLAMA_URL=http://127.0.0.1:${OLLAMA_STUB_PORT:-11435} CLAUDE_BIN=$(pwd)/e2e/fixtures/bin/fake-claude.mjs FINDMNT_BIN=$(pwd)/e2e/fixtures/bin/fake-findmnt.mjs UDISKSCTL_BIN=$(pwd)/e2e/fixtures/bin/fake-udisksctl.mjs next start --port ${PORT:-3011}",
"build": "next build",
- "start": "next start --port ${EDITOR_PORT:-3001}",
+ "start": "UV_THREADPOOL_SIZE=${UV_THREADPOOL_SIZE:-16} next start --port ${EDITOR_PORT:-3001}",
"lint": "eslint",
"e2e": "node ../scripts/queue-lock.mjs --ports PORT:3011,EXPORT_PORT:3010,OLLAMA_STUB_PORT:11435 -- playwright test",
"e2e:ui": "playwright test --ui"
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -7622,3 +7622,88 @@ this section is stale, by +16 near the top and +203 at the end; they are not rew
no digest, which is removed (`digests.remove`, `:962`/`:965`). `mtimes` is then written with the
video's current mtimes, so the loss lasts until any of its tracked mtimes (metadata, transcript,
subs, availability, digest) moves.
+
+## The storage health gate (verified 2026-09-29, branch `r15/drive-stall`)
+
+The record is [`release-15.md`](release-15.md), "Slice DS, as shipped", with the parent's rulings
+(Q1–Q5) and the review's fixes (M1–M3, L1–L10). Anchors are at the branch tip after the merge of
+`main` `bab894db`.
+
+- **A drive can be mounted and not answering.** Every in-process fs call on it waits on one of
+ libuv's threads (4 by default, 16 in the editor's `start`) until it answers (~30 s for the observed
+ USB reset loop); only a child process isolates a call. A child `stat` of a location's ROOT does not
+ detect it reliably: the root's inode is in the kernel's cache whenever the drive was used lately.
+- **The state is `common/lib/storageHealth.ts`**, one map on `globalThis.__yttStorageHealth__` (the
+ pass writes it from instrumentation's module copy; pages read it from theirs), with each entry's
+ `detector` and the counters' `device`. `recordLocationHealth` (`:182`): one `stalled` answer stalls
+ at once, and every transition to stalled refuses `onDrive`'s waiting calls; `HEALTH_CLEAN_TO_CLEAR`
+ (2) clean answers in a row clear it; `absent` is clean; a new root starts over.
+ `registerLocationHealth` (`:267`) creates entries with no answer. `stalledLocationForPath`
+ (`:319`) matches like `locationOfDataDir`; `stalledLocation` (`:335`) is by id AND root.
+- **Detector 1, every 15 s: the block device's counters** (`detectLocationHealth`,
+ `lib/storageVolumes.ts:601`). The root's device from the last pass (findmnt `-J -T <root> -o
+ SOURCE,UUID`, raced against 3 s, only when there is none or its `/sys` entry stops reading;
+ another volume's UUID names none; `[…]` stripped, `/dev/mapper` resolved, basename); then
+ `/sys/class/block/<dev>/stat` (`parseBlockStat`, `storageHealth.ts:735`): completed = fields 1 + 5
+ + 12 + 16 (reads, writes, discards, flushes), in flight = field 9. Stalled ⇔ in flight at both
+ samples AND nothing completed between; samples at least `MIN_COUNTER_INTERVAL_MS` (10 s) apart; the
+ first gives no verdict. in_flight counts only requests dispatched to the driver: one requeued
+ during a host reset is not counted, so a sample in that window can read clean (the watchdog covers
+ it). No device → the child `stat -L -c %F` probe (`probeLocationHealth`, `:386`). The samples are
+ on `globalThis.__yttHealthDetector__` (the pass and /storage's Refresh share them).
+- **Detector 2, on every gated call: `onDrive(where, call)`** (`storageHealth.ts:633`). Refused with
+ no call on a stalled location; otherwise raced against `DRIVE_CALL_BUDGET_MS` (3 s; test seam
+ `setDriveCallBudget`); the budget covers the whole unit passed in. A timeout marks the location
+ stalled (since now) and throws `DriveNotAnsweringError`, leaving the call to settle — unless the
+ location's device counters (read synchronously from `/sys` through the reader `storageVolumes.ts`
+ registers with `setCounterReader`, `:509`) moved since the call began: then the call is refused
+ as slow and nothing is marked. At most `DRIVE_CALLS_IN_FLIGHT` (4) calls per slot key in flight
+ (`acquireSlot`, `storageHealth.ts:501`): the rest queue in JS. A waiting call's deadline follows progress: every call that
+ returns on the key (in time or late) restarts it (`releaseSlot`, `:599`, re-arms every waiter), and a
+ waiting call is refused unmarked only when nothing on the key has returned for the budget plus a
+ quarter of it (at most 250 ms) — never for the queue's depth alone. Waiting calls are refused at
+ once by any transition to stalled; and when every slot is held by a call already past its budget
+ (`overdue`, each kept with the counters reading from when it began), a new call is refused at once,
+ and the location is marked stalled again only if the disk has completed nothing since the oldest of
+ them began (otherwise "drive slow", unmarked). A slot is freed when its call really returns. The
+ slot key: a configured location's id; a probe of another root under its id, that root (marks
+ nothing); a path on no configured location, the root it is under (`rootOfUnknownPath`, `:457`;
+ marks nothing). Do not nest it for one key. The timer is not unref'd.
+- **The cadence** is `runStorageHealthPass` (`controller/storageWatch.ts:433`): prune, register, then
+ every location concurrently; every 15 s from `startStorageHealthWatch` (`:546`), plus one at arm
+ time, armed by `editor/instrumentation.ts` ABOVE the idle gate (it writes nothing). The five-minute
+ pass (`startStorageWatch`, `:521`) stays below it. A CLI process has no pass (its inspects are still
+ raced). `refreshLocationHealth` (`:485`) is /storage's Refresh.
+- **The gate order in `inspectChannelMedia`** (`lib/channelMedia.ts:316`): config → memo (a
+ remembered `in-transition` is returned as is, anything else is gated first) → the relocation
+ marker (`:362`, corpus disk) → the gate (`:380`) → the link (corpus disk) → the target's `stat`
+ through `onDrive` (`:445`). A stall is never memoised. Other gated calls: `probeLocation`
+ (`storageVolumes.ts:255`, its stat and statfs through `onDrive`) and its memo; `volumeFreeBytes`
+ (`controller/storageLocations.ts:288`, `:325`, `:334`); `readChannelStat` (`controller/channels.ts:215`,
+ the walk through `onDrive`, `null` on a stall); the snapshot walk (`controller/channelSnapshot.ts:751`,
+ its listing, keep-latest keys and per-video unit); the recency tail reads; the move-root check; the
+ saved-video store; `listSavedVideos` with `notAnswering`; the videos list, the video page, the
+ Cleanup stage, the Storage stage's statfs and the media file route. `channelMediaStall(config)`
+ (`channelMedia.ts:231`) is the no-I/O question for a holder of a config.
+- **`stalled` is a sixth `ChannelMediaStatus` and a sixth `StorageLocationStatus`** ("Not
+ answering"). `HELD_REASON`, `MediaLocationBadge`'s two tables and `STORAGE_STATUS_LABEL` are the
+ `Record`s that make tsc name every table a seventh would need. `isMediaHeld` holds it, so both
+ pool-wide builds hold a stalled channel; the storage watch counts it as down (two passes pause),
+ on a location or not, and the pause record carries `cause: "not-answering"` (`ChannelAutoPause`,
+ `lib/channelPriority.ts`; absent = not there).
+- **`inspectChannelMedia` is memoised for 5 s** (`CHANNEL_MEDIA_MEMO_MS`, `:265`), keyed by channels
+ dir, slug and configured `dataDir`, on `globalThis.__yttChannelMediaMemo__`. `{ fresh: true }`
+ skips it and does not store; the deciders that pass it are listed in the record (the guard and its
+ six callers, both movers, both builds, the watch, eviction, the re-point preflight, doctor). The
+ runners' tick shares the status poll's `buildChannelWork` and so reads the memo.
+ `forgetChannelMedia` (`:291`) is called by the channel mover's marker writes and clear,
+ `clearRelocationMarker`, a re-point, /storage's Refresh and the e2e `invalidate-cache` route.
+- **`UV_THREADPOOL_SIZE`** defaults to 16 in `editor/package.json`'s `start` and in
+ `docker/entrypoint.sh`. `ports.test.ts` reads every `${NAME:-N}` in a script as a port and names it
+ as the one exception (`NUMERIC_NOT_PORTS`, `common/lib/ports.test.ts:29`).
+- **Not covered:** a call already in flight when the drive stalls (at most four per drive for the
+ calls through `onDrive` — every page and poll path and the snapshot walk; a job's own reads that do
+ not go through it, `measureTree`, the index build's processing phase, the snapshot's sequential
+ reconcile pass, are not capped); a hand-typed root is capped but never marked; a drive already
+ stalled at boot before the second counter sample, unless a page reaches it; per-click server
+ actions; the file route's stream; the corpus disk itself.
diff --git a/plans/release-15.md b/plans/release-15.md
@@ -17,7 +17,7 @@ prompt carries its ruling, and this record carries what was built. Rules:
| Slice | Branch | What | Owns |
|---|---|---|---|
| IG | `r15/index-hold` | The index build holds an unreachable channel instead of emptying it | `common/controller/buildIndex.ts` + new `buildIndex.test.ts`, `common/controller/buildStats.ts` (the hold's words move to a shared module), new `common/lib/channelMediaHold.ts`, `common/lib/envVars.ts`, `ENVIRONMENT.md`; records: `plans/{STATE,FACTS,stats-cache-key}.md` |
-| DS | `r15/drive-stall` | A stalled drive does not stop the editor answering | per its prompt |
+| DS | `r15/drive-stall` | A stalled drive does not stop the editor answering | new `common/lib/storageHealth.ts`; `lib/{storageVolumes,channelMedia,channelMediaHold}.ts`, `controller/storageWatch.ts` and the gated callers; `/storage`, `/channels`, the videos pages; `UV_THREADPOOL_SIZE` (`editor/package.json`, `docker/entrypoint.sh`, `envVars.ts`) |
| UT | `r15/umtool-trace` | umtool's build stops tracing the whole `umtool/` folder | per its prompt |
**Order:** IG → DS. DS adds a health gate inside `inspectChannelMedia`, which IG's hold calls
@@ -463,4 +463,355 @@ on the real fixture, which carries `.env.local`, `.next-shots` and `test-results
two of them names the excludes do not cover. Until that rebuild the post-build check skips in the
primary, saying why. umtool's code changes nothing at run time.
+### Slice DS, as shipped — a stalled drive does not stop the editor answering (2026-09-29)
+
+Branch `r15/drive-stall` off `main` `ccf90892` (slice IG merged), worktree `~/Projects/r12-paths-fix`
+(block #12: editor 4201, test 4211, export 4210), one Opus implementer. Scratch files `ds-*` in the
+job's `tmp`. The ruling: a drive that is mounted and not answering must not stop the editor
+answering. Two detectors find the stall without the editor waiting on the drive, and pages and polls
+do not touch it in-process while it is not answering. The parent's five rulings on the first pass
+(Q1–Q5) and the review's fixes (M1–M3, L1–L10) are applied; the tables at the end map each to its
+commit.
+
+**What was wrong.** Node runs every filesystem call on libuv's thread pool, four threads by default.
+On a drive that has stalled (an SMR disk in a USB enclosure resetting under a long write) each call
+blocks its thread for about 30 s. The home page, `/channels` and the three-second auto-queue status
+poll each `stat`ted every relocated channel's target in-process and uncached; the one-second job-list
+poll walked the whole `data/` of every channel with a job listed, 64 calls wide; and the recency layer
+read metadata tails 32 at a time. Four blocked calls were enough for no page, poll or job log to
+answer. The storage watch saw nothing of it: its five-minute pass asks "is the disk here", and a
+stalled disk is here.
+
+- **The health state** (new `common/lib/storageHealth.ts`): one map per process on `globalThis`
+ (`__yttStorageHealth__`, the house pattern: the watch writes it from instrumentation's module copy
+ and pages read it from theirs), by location id: state `ok | stalled | absent`, `since`, the last
+ check, a clean streak, the cause, and the `detector` that gave the last verdict. In memory only.
+ - One `stalled` answer marks the location stalled at once. Two clean answers in a row clear it (an
+ `absent` answer is clean). A miss in between starts the count again. A re-pointed root starts the
+ location over. Locations no longer configured are pruned; every configured one is registered
+ (`registerLocationHealth`) before a pass asks anything, so the watchdog can find a channel's
+ location before the first verdict.
+ - Pure of I/O and without execa, so `lib/channelMedia.ts` can ask it.
+- **Detector 1, every 15 s: the block device's own counters** (`detectLocationHealth`,
+ `common/lib/storageVolumes.ts`; ruling Q1(a)). A child `stat` of the root is answered from the
+ kernel's inode cache whenever the drive was used lately, so it can say "ok" while the reads that
+ reach the device wait out a reset loop. Instead, per location per pass:
+ - the root's device: `findmnt -J -T <root> -o SOURCE,UUID` as a child raced against 3 s, a
+ `[subvolume]` suffix taken off, `/dev/mapper/*` resolved to its `dm-N` (a read of `/dev`), the
+ basename. Asked only when the root has no device yet or its device's `/sys` entry stops reading
+ (a replug under another name): a findmnt per pass would leave one child stuck per pass during a
+ long stall (L10). A UUID other than the location's recorded one names no device (the root is
+ then a directory on another filesystem, not the drive).
+ - `/sys/class/block/<dev>/stat`, which never touches the drive: completed = reads (field 1) +
+ writes (5) + discards (12) + flushes (16) where the kernel counts them (in_flight counts those
+ too, so a long SMR media-cache flush alone moves completions; L2), and requests in flight (9).
+ Against the previous sample for the location (same device, at least `MIN_COUNTER_INTERVAL_MS` =
+ 10 s earlier): **stalled ⇔ in flight at both AND nothing completed between**; anything else is
+ clean. The first sample gives no verdict. The samples are on `globalThis`
+ (`__yttHealthDetector__`), so the pass and `/storage`'s Refresh, in different module copies,
+ compare against one previous sample (L5).
+ - **No device** (a container, no findmnt, a tmpfs or network source, no `/sys` entry) falls back to
+ the child `stat` probe (`probeLocationHealth`): `stat -L -c %F -- <root>` raced against 3 s, the
+ child SIGKILLed and not waited for; `directory` is `ok`, anything else `absent`, no binary `ok`.
+ - The verdict names its detector (`"counters" | "stat"`), the health state records it (and the
+ counters' device, which the watchdog reads), and `/storage`'s line says which watched the drive.
+- **Detector 2, on every gated call: a 3 s watchdog** (`onDrive(where, call)`,
+ `lib/storageHealth.ts`; ruling Q1(b)). The detector that cannot be fooled by a cache: a page or poll
+ that actually reaches the drive finds out.
+ - Refused at once, with no call, when the location is stalled.
+ - Otherwise raced against `DRIVE_CALL_BUDGET_MS` (3 s). A call that has not answered marks its
+ location stalled (since now, cause "a read in the editor did not answer within 3 s") and throws
+ `DriveNotAnsweringError`; the caller answers `stalled`. The call is left to settle on its own:
+ its thread is the stated limit.
+ - **Slow is not stalled** (L6): on a timeout, when the counters detector has named the location's
+ device, its counters are read (synchronously, from `/sys`, so the check does not wait behind the
+ pool it is judging) and compared with a reading taken when the call began. Requests completed
+ meanwhile: the drive is slow; the call is refused and nothing is marked.
+ - **The budget covers a whole unit of work** (L8): a video directory's reads, a page's reads of
+ one video, go through as one call, so a slow drive still answering can be marked by one long
+ unit (unless the counters show it completing, above).
+ - **At most `DRIVE_CALLS_IN_FLIGHT` (4) calls per slot key are in flight.** The rest wait in a
+ queue of its own (not libuv's), so a 64-wide walk that meets a stall puts four calls on the
+ drive, not 64. A slot is released when its call really returns. The queue (M1):
+ - every transition to `stalled` refuses the waiting calls at once, whoever decided it (the
+ pass, the watchdog, a Refresh);
+ - a wait's deadline follows progress (re-review M4): every call that returns on the key, in
+ time or late, restarts the deadline of every call waiting on it, and a waiting call is
+ refused, without marking, only when nothing on the key has returned for the budget plus a
+ grace of a quarter of it (at most 250 ms; the grace lets the calls it waits behind, whose
+ timers start a moment later, time out and mark first). A deep queue on a drive that is busy
+ but answering therefore waits as long as it takes;
+ - calls past their budget are counted per key (`overdue`), each with the counters reading
+ taken when it began; when every slot is held by one, a new call is refused at once, and the
+ location is marked stalled again (even if the pass has since cleared it) unless the disk has
+ completed requests since the oldest of them began: then it is slow, not stalled, and nothing
+ is marked (the same test as a timeout's).
+ - **The slot key** (M2, L7): a configured location's id; a probe of another root under a
+ location's id is keyed by that root and marks nothing; a path on no configured location (a root
+ typed by hand) is keyed by the root it is under (`<root>/<slug>/data` → `<root>`), so it holds
+ at most four threads too, and nothing can mark it. Calls are not nested for one key. The timer
+ is not `unref`'d: it is cleared the moment the call answers.
+- **The cadence** (`common/controller/storageWatch.ts`): `startStorageHealthWatch` arms the 15 s
+ health pass and runs one at once; **`editor/instrumentation.ts` arms it above the idle gate**,
+ beside the storage boot probe (M2): it is in memory and writes nothing, and without it an idle
+ boot has no registered locations and a stall the watchdog marks is never cleared. The five-minute
+ pass (`startStorageWatch`), which may write an auto-pause, stays below the gate. All locations are
+ asked concurrently, each bounded by its own timers, and an overrunning pass is not stacked. A
+ transition is logged (`[storage] "<id>": drive not answering — <cause>; …` / `answering again
+ (ok)`). A CLI process has no pass, but its gated calls still go through the watchdog.
+- **The gate and the watchdog, by caller:**
+
+ | Caller | On a stalled location | Through `onDrive` |
+ |---|---|---|
+ | `inspectChannelMedia` (`lib/channelMedia.ts`) | **The relocation marker is read first** (ruling Q2: it is in the channel dir, on the corpus disk), so a channel mid-move on a stalled drive reads `in-transition`. Then the gate: status `stalled`, detail `drive not answering (location "<label>", since HH:MM)`, before the link and the target. With a config in hand a stalled channel costs one call, the marker read. | The target's `stat` (`:445`). |
+ | `assertChannelMediaReachable` | Refuses it (`ChannelMediaUnreachableError`, status `stalled`), so `runManagedFunction`'s `needsMedia` guard, `generateChannelSnapshot` (also the channel page's **Refresh report**), the operation batch, normalise, keep-videos and the shard action refuse it. | Through inspect. |
+ | The index and stats builds | Hold it: `HELD_REASON.stalled` is "its drive is not answering (a stalled disk)", and IG's `isMediaHeld` holds every status but `ok` and `in-place`. | Through inspect (both of the index build's looks). |
+ | The storage watch's five-minute pass | Counts a stalled location, or a channel whose inspect says `stalled` on no location (L9), as down: two passes auto-pause its channels, and the pass after the drive answers restores them, as for an unmount (ruling Q3). The pause record carries `cause: "not-answering"`, and `autoPauseReasonOf` says "on a drive that is not answering … when the drive answers again" (L4; a record without a cause, all written before, reads as not there). | Through inspect and `probeLocation`. |
+ | `probeLocation` and `probeLocationMemo` (`lib/storageVolumes.ts`) | Probe status `stalled` (`STORAGE_STATUS_LABEL`: "Not answering"), identity unknown, no free space; no `stat`, `statfs` or `findmnt`. The memo is asked after the gate. | The root's `stat` and `statfs` (`:262`, `:279`). |
+ | `volumeFreeBytes` (`controller/storageLocations.ts`) | Unknown ("—"), with no call. | The root's and the mountpoint's `stat`, and the `statfs` (`:288`, `:325`, `:334`). |
+ | `generateChannelSnapshot` (`controller/channelSnapshot.ts`), after every download or sync, sixteen video directories wide (M3) | Refused by its start guard. | Its `data/` listing (a refusal is rethrown, never read as an empty channel), the keep-latest keys' metadata reads (`keyedVideosNewestFirst`, now `mapConcurrent` 16 wide instead of an unbounded `Promise.all`) and each video directory's unit (`:751`, `:844`). A throw keeps the last `snapshot.json`, as on any failed refresh. The sequential reconcile pass before it is not raced. |
+ | `readChannelStat` (`controller/channels.ts`) | `null` (no counts, so no progress bar), with no walk. The one-second job-list poll and the home page ask it for every channel with a job listed. | The `data/` readdir, then each video directory as one call (`:107`); `null` when the drive stops answering mid-walk. The batch jobs' `listChannelStatsFromDisk` passes no drive and is unchanged. |
+ | The recency tail reads (`controller/recencyIndex.ts`) | Skipped, and NOT remembered as misses: layers 3 and 4 until a refresh with the drive answering reads them. | Each tail read of a relocated channel (`:257`). |
+ | `relocationRootPresenceProblem` (`controller/relocateChannelMedia.ts`) | A move onto a stalled location is refused before the root's `stat`. | The root's `stat` (`:307`). |
+ | `inspectSavedVideosStore` (`controller/relocateSavedVideos.ts`) | `unreachable` with the stall's detail; `/storage` skips the store's size walk. | The target's `stat` (`:181`); `/storage`'s store walk too. |
+ | `listSavedVideos` (`controller/savedVideoInventory.ts`), when a page passes `notAnswering` | The channel is skipped and named (its `data/` link is read, not followed). The backup job passes nothing and is unchanged. | Each read of a relocated channel (`:81`). |
+ | The videos list, the video page (and its title), the channel page's Cleanup stage | A notice (`aria-label="media not answering"`, `MediaNotAnswering.tsx`) with links to the channel and `/storage`. | The list's `data/` listing and titles, then the selected video's files; the video page's whole directory read, as one unit; its title; the Cleanup stage's saved-video totals. |
+ | The channel page's Storage stage free space (L1) | "—". | The `statfs` of the relocated target. |
+ | The media file route (`/api/channels/<slug>/videos/<id>/files/<name>`) | 503 with `Retry-After: 15`. | The file's `stat`; the stream after it is not raced. |
+
+- **The memo** (`inspectChannelMedia`): five seconds per channel, keyed by channels dir, slug and
+ the configured `dataDir`, on `globalThis` (`__yttChannelMediaMemo__`). On by default (ruling Q4).
+ A remembered `in-transition` is given as it is; any other remembered answer is gated first; a
+ stall is not remembered. `{ fresh: true }` bypasses it and does not store. **The deciders, every
+ one passing `fresh: true`:**
+
+ | Caller | Where |
+ |---|---|
+ | `assertChannelMediaReachable` (so every guard below) | `lib/channelMedia.ts:514` |
+ | ↳ `runManagedFunction`'s `needsMedia` guard | `jobs/streamCommand.ts:283` |
+ | ↳ the operation batch | `controller/operationBatch.ts:1593` |
+ | ↳ `generateChannelSnapshot` | `controller/channelSnapshot.ts:739` |
+ | ↳ `normalizeAllTranscripts` | `controller/normalizeAll.ts:63` |
+ | ↳ keep-videos | `controller/keepVideosMatching.ts:165` |
+ | ↳ the shard action | `editor/app/channels/[slug]/shardActions.ts:88` |
+ | The channel mover, out and back | `controller/relocateChannelMedia.ts:638`, `:877` |
+ | The index build, before and after the walk | `controller/buildIndex.ts:348`, `:467` |
+ | The stats build | `controller/buildStats.ts:291` |
+ | The storage watch's five-minute pass | `controller/storageWatch.ts:235` |
+ | Clip-window eviction | `controller/evictClipWindows.ts:99` |
+ | The re-point preflight | `controller/storageLocations.ts:564` |
+ | `archilyzer doctor` | `bin/doctor.ts:124` |
+
+ **The memo's readers**, pages and polls: the home page (`editor/app/page.tsx:80`), `/channels`
+ (`editor/app/channels/page.tsx:219`), the channel page (`[slug]/page.tsx:239`), the ops channel
+ route (`api/ops/channel/[slug]/route.ts:70`), `channelsOnLocation`'s rollup for `/storage`
+ (`controller/storageLocations.ts:225`), the runners' `buildChannelWork` (`controller/autoRunner.ts:622`,
+ the tick and the three-second status poll share it), and the bulk actions' skips
+ (`editor/app/channels/actions.ts:562`, `bulkStorageActions.ts:105`; the jobs they queue re-check
+ fresh). `forgetChannelMedia(slug?)` clears it: the channel mover's marker writes and clear (each
+ phase writes its marker after the link and config it changes), `clearRelocationMarker`, a
+ completed re-point, `/storage`'s Refresh and the e2e `invalidate-cache` route.
+- **Headroom:** `UV_THREADPOOL_SIZE=${UV_THREADPOOL_SIZE:-16}` in `editor/package.json`'s `start`
+ (which the rollout restart script runs) and in `docker/entrypoint.sh` before the editor's `exec`.
+ Declared in `envVars.ts` (runtime) and `ENVIRONMENT.md` regenerated. It buys time for calls
+ already in flight and isolates nothing; the doc line says so. `ports.test.ts` reads every
+ `${NAME:-N}` in a script as a port, so the same commit names `UV_THREADPOOL_SIZE` as the one
+ numeric default that is not (ruling Q5).
+- **Where it shows:**
+
+ | Surface | What it says |
+ |---|---|
+ | `/storage` | The row's status badge reads **Not answering**; a line under the details reads `not answering since HH:MM — <cause>. Pages and polls skip this drive until it answers twice in a row. Watched through its disk's request counters.` (or `… with a stat of its root (no disk could be named here).`), `aria-label="location not answering"`. Re-point and Mount are withheld with the status. **Refresh** asks the location's detector first (one answer, counted like the pass's; the counters give none within 10 s of the last sample) and forgets the channels' remembered answers; on a stalled drive its note says so instead of probing. |
+ | `/channels` | The volume chip reads `… · not answering since HH:MM` in place of its free space, with a title saying what it means. Each row's badge reads `on <label> — not answering` (accessible name `media location: Media not answering · on <label>`). |
+ | The channel page | The Storage stage's card is red with the inspector's sentence; its destination list names the location "Not answering"; Move back is withheld for a stalled channel, as for an unreachable one. |
+ | `/saved-videos` | `Not read, because the drive their media is on is not answering: <slugs>.` (`aria-label="saved videos not read"`). |
+ | `/review`, the rack, the channel page (an auto-paused channel) | `Auto-paused — its media is on a drive that is not answering since <date>. It returns to <tier> on its own when the drive answers again.` (L4) |
+ | The runners | `[auto] skipping <slug>: media stalled — drive not answering (…)`, once per state change. |
+ | The logs | The health pass's transition lines, with the cause; the index and stats builds' hold lines. |
+
+ Labels are contracts: no existing accessible name or test id changed.
+
+**Commits**
+
+| Commit | What |
+|---|---|
+| `c0ecc55c` | `common:` `lib/storageHealth.ts`, `probeLocationHealth`, the 15 s health pass, the `stalled` statuses and the gate in inspect, the probe and its memo, `volumeFreeBytes`, `readChannelStat`, the recency reads, the move-root check and the saved-video store; `HELD_REASON.stalled`; the 5 s memo with `fresh` for every decider and `forgetChannelMedia` in the movers; the badge's and the stage card's words. Tests. |
+| `cfe551de` | `editor:` `/storage` (the row's line, Refresh asks the health first), the `/channels` volume chip, the Storage panel's Move back, the videos list and video page notice, the media file route's 503, the e2e `invalidate-cache` route; `refreshLocationHealth`; `views/storage.ts` `notAnswering`. |
+| `480f2556` | `editor:` `UV_THREADPOOL_SIZE=16` in the editor's `start` and `docker/entrypoint.sh`; `envVars.ts` + `ENVIRONMENT.md`; `ports.test.ts` names it as a numeric default that is not a port (folded in from the next commit, ruling Q5; no change to the tree at the tip). |
+| `4e1f7a90` | `editor:` `listSavedVideos({ notAnswering })` for `/saved-videos` and the Cleanup stage's notice. |
+| `63f42ef5` | `plans:` the first version of this section, FACTS, the changelog. |
+| `c69ad41a` | `common:` rulings Q2 and Q1(b): the marker before the gate; `onDrive` (the 3 s watchdog and the four-call cap per location) on every common gated call; `registerLocationHealth`. Tests. |
+| `f6a25cf5` | `editor:` the pages' reads of a relocated drive through `onDrive` (the videos list, the video page and its title, the file route, the Cleanup totals, `/storage`'s store walk). |
+| `b1a30902` | `common:` ruling Q1(a): `detectLocationHealth`, the block-device counters with the child-stat fallback; `detector` in the health state and on `/storage`. Tests. |
+| `981e05a0` | `plans:` this section rewritten for the rulings; FACTS; the changelog. |
+| `33094c29` | `common:` review M1, M2 (slot keys), L2, L5, L6, L7, L8, L10: the queue's deadline, the overdue count, waiters refused on every transition to stalled; the hand-typed-root and candidate-root keys; slow is not stalled; discards and flushes counted; the detector's samples on `globalThis`; findmnt only when needed. Tests. |
+| `bd39579e` | `common, editor:` review M2, L4, L9: the health pass armed above the idle gate; the pause record's `cause`; a `stalled` channel on no location is down. SETTINGS.md. Tests. |
+| `ff235c2f` | `common:` review M3: the snapshot walk through `onDrive`; keep-latest keys bounded. Test. |
+| `19c5842d` | `editor:` review L1, L3: the Storage stage's `statfs` through `onDrive`; the stale comments. |
+| `30df4193` | Merge `main` (`bab894db`: release 14 HS, S1, CF; release 15 UT). Two conflicts, both kept: the changelog's `[Unreleased]` carries UT's bullet then DS's; this file is `main`'s with DS's table row and this section after UT's. |
+| `b34a7613` | `plans:` the review's findings to their commits; FACTS; the changelog. |
+| `c367a3d7` | `editor:` re-review L11: the health pass's block above the boot probe's comment. |
+| `d61bdd93` | `common:` re-review M4: a slot wait's deadline follows progress. Tests. |
+| `ee68d225` | `plans:` the re-review's findings to their commits; FACTS. |
+| `9a12308e` | `common:` the overdue refusal tells slow from stalled by the counters, as a timeout does. Tests. |
+| this commit | `plans:` that ruling as a decision row; FACTS; the report. |
+
+**Tests** (unit; no test stalls a real drive: a stalled call is a promise that never settles or a
+fake `stat` that never answers, and a stalled device is a temp `/sys` whose counters stand still)
+
+| File | What it pins |
+|---|---|
+| `lib/storageHealth.test.ts` (28) | The rules: one miss stalls at once, one clean answer after a stall does not clear it and two in a row do, a miss in between starts over, `absent` is clean, a re-pointed root starts over, the path match is `locationOfDataDir`'s, pruning, `globalThis`, the "since" wording, registering without an answer. `onDrive`: an answer passes through (value or error) and frees its slot; a never-settling call is `stalled` on the timer, marks the location (since now) and keeps its slot until it settles; a stalled location is refused with no call; seven calls at once put four in flight and the three that waited are refused without a call when the location stalls; a freed slot runs a waiter; the 3 s default (answered after 3–4.5 s). After the review: a hand-typed root's four slots are shared by its channels and a fifth call is refused at once, with no entry made; a candidate root has its own slots and leaves the location's calls alone; **M1's case** (four hung calls, two clean answers, a fifth refused at once and the location marked again); a transition to stalled from the pass refuses the waiting call; **L6** (counters completing: four calls refused as slow, the location not marked, a waiter's wait runs out unmarked when nothing returns; counters standing still: marked). After the re-review (**M4**): a healthy 64-wide walk of units at half the budget (the last call waits about fifteen units) has no refusals and marks nothing, on a location and on a hand-typed root; a queue behind four hung calls is still refused within the budget (marked on a location, unmarked on a hand-typed root); a call that returns late hands its slot to the calls waiting behind it. Then: four units slower than the budget on a device whose completions move — the fifth call is refused at once and nothing is marked; with completions unchanged, after the pass cleared the location — refused and marked again. |
+| `lib/storageHealthCounters.test.ts` (7) | The stat line parser (17 and 11 fields, garbage); the verdict over sample pairs (stuck → stalled; moving, idle or drained → ok); device names (partition, `[subvolume]` suffix, non-`/dev` sources); the detector end to end with a fake findmnt and a temp `/sys`: the first sample gives no verdict, stuck → stalled with its cause, drained → ok, busy and moving → ok; a sample sooner than 10 s gives none and keeps the first; every no-device fallback goes to the child stat and says so (tmpfs, findmnt failing, no `/sys` entry, another volume's UUID, no binary). After the review: discards and flushes are counted (a flush alone is not a stall); a known device is read without running findmnt, and findmnt runs again only when its `/sys` entry stops reading; the samples are on `globalThis`. |
+| `lib/storageHealthProbe.test.ts` (5) | The child `stat`: a directory is `ok`, a missing path and a file are `absent`; a fake `stat` asleep for 20 s is `stalled` on the timer without being waited for; the 3 s default; an answer inside the budget is taken; no binary is `ok`. |
+| `controller/storageStall.test.ts` (22) | A spy on every `node:fs` and `node:fs/promises` call, with a hang mode that makes a matching promise-API call never settle. The gate: with the location stalled, inspect reads only the marker (with a config) or config.json and the marker (without); a channel mid-move on a stalled drive reads `in-transition` (and is remembered so); the guard, `probeLocation` and its memo (no findmnt run), `volumeFreeBytes`, `readChannelStat`, the recency layer, the move-root check, a snapshot refresh, the saved-video store and the inventory make no call on the drive. The watchdog: with the location answering and one drive call hung, inspect, `readChannelStat` (at most four video directories asked), `probeLocation`, `volumeFreeBytes`, the recency layer (not remembered as a miss) the move-root check and (M3) the snapshot walk each answer `stalled` within the race and mark the location (the snapshot writes no `snapshot.json`, and at most four video directories reach the drive); after it, inspect, the guard and the walk make no call on the drive, and once cleared the drive is asked again. The memo: two inspects inside 5 s stat the target once, `fresh` and a 5 s age ask again, a fresh answer is not stored, another target is another key, `forgetChannelMedia` and `clearRelocationMarker` clear it, and a stall is seen with an `ok` remembered. |
+| `controller/storageWatch.test.ts` (+9) | The pass stalls a location on one miss and a page then gets `stalled`; the five-minute pass suspects its channel; two clean passes clear it; a location no longer configured is forgotten; a probe that throws is `ok`; the health pass arms on its own, runs once at once and stops, and the five-minute watch arms no health pass; a Refresh counts as one answer; the pass registers every location, records a verdict's detector, and a verdict with no answer changes nothing; a counters verdict records its device and a stat verdict forgets it; a stall auto-pauses after two passes with `cause: "not-answering"` and its wording, the sanitizer keeps the cause, and a record without one reads as not there. |
+| `views/storage.test.ts` (+1) | A stalled row reads "Not answering", carries its line, has no free space, and withholds Re-point and Mount. |
+
+#### Gates (logs `$T/ds-*.log`)
+
+- **tsc** was clean before every commit and on the merged tree (65 s). After the re-review: 53 s,
+ once the worktree's gitignored `export/.next/dev` (a generated `validator.ts` an export dev server
+ had left truncated) was deleted; it is not a source file.
+- **Unit, on the merged tree (`30df4193`):**
+
+ | Suite | Result |
+ |---|---|
+ | common | **2,295/2,295**, 58 s: `main`'s 2,229 plus DS's 66 (60 at the rulings, 6 from the review). After the re-review: **2,299/2,299**, 53 s (4 for M4); after the overdue ruling **2,301/2,301**, 75 s (2 more). |
+ | editor unit | 87/87, and 87/87 after the re-review |
+ | `test:scripts` | 194 passed, 2 skipped (196). The second skip is UT's post-build trace check, which skips a umtool build older than its config (this worktree's `umtool/.next` predates UT); the first is the `LIVE=1` archive check. |
+ | mcp | 271/271 at the first pass; no mcp file has changed since. |
+
+- **Docs:** `docs env --check`, `docs files --check` and `settings example --check` all exit **0**
+ (SETTINGS.md regenerated in `bd39579e` for the pause record's `cause`).
+- **Build:** the editor's `next build` on the merged tree, with the primary's `transcripts/` linked
+ in and capped at 5 GB with no swap: 63 s, max RSS 1,641,444 KB, exit 0. The link was removed after the build, and
+ nothing ran through it. (First pass: 104 s, max RSS 1,643,860 KB.)
+- **e2e** (editor, detached and queued):
+
+ | Run | Specs | Result |
+ |---|---|---|
+ | 1, first pass | the `storage` and `channels` specs, `video-page`, `saved-videos`, `dashboard`, `auto-queue`, and IG's eight (`$T/ds-specs.txt`) | **124 passed, 3 failed, 12 skipped, 16.6 min**: three 30 s timeouts while this slice's own tsc and common suite ran beside the suite. |
+ | 2, first pass | `auto-queue`, `tags`, `saved-videos`, four cleanup specs, `video-titles`, `video-page` (`$T/ds-specs2.txt`) | **70 passed, 1 failed, 5.9 min**: `auto-queue.spec.ts:352`, the Start-button race its own comment describes. |
+ | 3, first pass | `auto-queue` alone | **24 passed, 1.3 min**. |
+ | 4, after the rulings (`b1a30902`) | the 9 `storage` + `channels` specs (`$T/ds-specs4.txt`) | **38 passed, 0 failed, 12 skipped, 2.5 min**. |
+ | 7, after the overdue ruling (`9a12308e`) | the 9 `storage` + `channels` specs (`$T/ds-specs4.txt`) | **38 passed, 0 failed, 12 skipped, 2.5 min**. |
+ | 6, after the re-review (`d61bdd93`) | the 9 `storage` + `channels` specs (`$T/ds-specs4.txt`) | **38 passed, 0 failed, 12 skipped, 2.6 min**. |
+ | 5, after the review, on the merged tree (`30df4193`) | the 9 `storage` + `channels` specs; the snapshot scheduler's (`auto-report-refresh`, `channel-work`, `jobs-batch-tasks-drain`, `incomplete-transcript`); the keep-latest users of the bounded key fan-out (`cleanup-holds`, `saved-videos`); `reconcile`; `review` (the auto-pause wording) — `$T/ds-specs5.txt` | **73 passed, 0 failed, 12 skipped, 5.2 min** (37 min 54 s in the queue behind another session's suite). |
+
+ The 12 skips in each are `channels-rack-audit`, which needs `E2E_RACK_SHOTS`. No e2e fixture has
+ a stalled drive: these confirm nothing changed for drives that answer. The stall paths are the
+ unit tests above.
+- **Numbers tool:** none.
+
+#### Found and left
+
+- **The stated limit.** A call already in flight when the drive stalls holds its thread until the
+ kernel gives up (about 30 s in the observed reset loop). On one drive, at most four threads wait
+ that way for the calls that go through `onDrive`: every page and poll path, and the snapshot
+ walk. A job's own reads that do not go through it are not capped: `measureTree` in the movers,
+ the index build's processing phase (IG recorded it), the snapshot's sequential reconcile pass
+ (one read at a time), and the keep-latest reads of the other callers of `computeKeptVideoIds`
+ (cleanup, persist, prune; now 16 at a time). With 16 threads the editor keeps answering while
+ those four wait.
+- **A root typed by hand** (a channel moved to a root no storage location names): its calls are
+ capped by the root they are under, four at a time, and raced, but nothing can mark it, so every
+ page and poll keeps asking it, each answering after 3 s or at once while its four slots are held
+ by calls the watchdog gave up on. Its channels' `stalled` comes from the watchdog alone, and the
+ five-minute pass counts it as down (L9). Naming the root as a location on `/storage` gives it the
+ full treatment.
+- **The drive's spin-up** (L6, a question for the operator): the budget is 3 s, and a USB disk that
+ spins down when idle can take 3–10 s to answer its first read. The watchdog then refuses that
+ read, and marks the location unless the counters show requests completing meanwhile (spinning
+ up completes none), for 15–30 s, during which the start-of-work guard refuses the channel's jobs.
+ **Does the drive spin down when idle?** If it does, a longer budget for the first call after an
+ idle spell, or a spin-down timer on the drive, would avoid it.
+- **A drive slower than the budget per unit is treated as not answering.** A queue's depth no
+ longer matters (M4): a waiting call is refused only when nothing on the drive has returned for
+ 3 s. But a single unit that takes longer than 3 s is refused: without a mark when the counters
+ show the disk completing other requests, with one otherwise. A snapshot refresh with such a unit
+ throws, so the scheduler keeps the last `snapshot.json` and tries again on its next trigger. Once
+ four such units hold every slot past the budget, the next call is refused at once (a decision
+ below says when that also marks).
+- **The counters need two samples.** A drive already stalled when the editor starts is seen by the
+ counters at the second pass (15–30 s), or at once by the watchdog when a page reaches it.
+ in_flight counts only requests dispatched to the driver: a request requeued during a host reset
+ is not counted, so a sample in that window can read clean; the watchdog covers it.
+- **Ungated request paths**, each a click rather than a page or poll: the channel and video server
+ actions that read a video's directory in the request (`bulkVideoActions`, `digestActions`,
+ `videoActions`, `fixIncompleteTranscript`, `pipelineActions`); most of what they do is enqueue
+ jobs, whose guard is fresh and refuses a stalled channel. `/api/media/fetch-window/<jobId>` (one
+ `stat` of the fetched file, once the job is done). The media file route's stream after its `stat`.
+- **The runners' tick reads the memo.** `buildChannelWork` serves both the tick and the status poll,
+ so a drive unmounted in the last 5 s can have one unit dispatched, which fails at the dangling
+ link. The snapshot regeneration and `runManagedFunction`'s guard are fresh.
+- **umtool's twin of the reachability check** (`checkChannelReachable` in
+ `umtool/report-to-video/cues.mjs`) has no stall gate: umtool is its own process with no health
+ state, and `umtool/**` belongs to another slice.
+- **A CLI process** (`archilyzer index`) has no health pass, so nothing is marked in it; its
+ inspects still go through the watchdog, so a target `stat` that takes over 3 s holds the channel.
+ The corpus disk itself is not watched.
+
+#### Rulings (parent, 2026-09-29) and decisions the operator could overturn
+
+| Question | Ruling | Where |
+|---|---|---|
+| Q1: the child stat of the root can be answered from the cache | Two detectors: the block device's counters every 15 s (child stat only with no device, and the state says which), and a 3 s watchdog on every gated call | `b1a30902`, `c69ad41a`, `f6a25cf5` |
+| Q2: the gate before or after the marker | The marker first; the gate covers everything after it | `c69ad41a` |
+| Q3: a stall auto-pauses | Yes, after two five-minute passes; it clears when the drive answers again, as for an unmount | as built |
+| Q4: the memo on by default | As built; every decider listed above | as built |
+| Q5: a commit that failed `ports.test.ts` alone | The test line folded into the thread-pool commit | `480f2556` |
+
+| What I assumed | The alternative |
+|---|---|
+| At most four gated calls per location in flight; the rest queue in JavaScript and are refused on a stall. **Kept at review.** | No cap: the watchdog alone, and a stall mid-walk fills the pool until the kernel gives up. |
+| A wait for a slot runs out at the budget plus a quarter of it (at most 250 ms), so simultaneous timeouts of the calls it waits behind mark first. | Exactly the budget: a waiter queued in the same tick then gives up a moment before those calls and is refused unmarked, and the mark lands a few ms later. |
+| The keep-latest key reads run 16 at a time for every caller (they were an unbounded `Promise.all`). | Bound them only for the snapshot. |
+| **Ruled after the re-review:** when every slot is held by a call past its budget, the next call is refused at once, and marks the location stalled only when the disk has completed nothing since the oldest of those calls began; four slow units on a disk still completing requests are "drive slow", refused and unmarked. With no device named (the stat detector), it marks. | Mark whenever every slot is overdue (the M1 fix as first built): four slow units on a busy disk then marked it stalled for 15–30 s. |
+| An auto-pause record with no `cause` reads as not there. | Word both cases for it ("not there or not answering"). |
+| Two counter samples closer than 10 s give no verdict (a Refresh just after a pass among them). **Kept at review.** | Compare any two samples (a busy healthy drive can read "in flight, nothing completed" over a few milliseconds). |
+| A findmnt that does not answer reuses the last device named for that root; one naming another volume's UUID names none. | Treat a findmnt that does not answer as a stall. |
+| An `absent` answer counts as clean toward clearing a stall. | Only `ok` clears it. |
+| `/storage`'s Refresh is one answer like the pass's. | Refresh clears a stall outright on one clean answer. |
+| `UV_THREADPOOL_SIZE` defaults to 16 in `start` and the entrypoint; a value already set wins. | A fixed 16, or only in the rollout restart script. |
+| The videos list, the video page and the Cleanup stage show a notice instead of their content. | Render what the snapshot knows and leave out only the drive's files. |
+| `readChannelStat` returns `null` on a stall (the job row draws no progress bar). | Counts marked unknown. |
+
+**What runs which code, for the rollout.** The health pass lives in the editor's process, armed with
+the storage watch, so detection takes effect only when the editor is rebuilt and restarted (and not
+on an idle boot). The restart must go through `pnpm run start` in `editor/` (the rollout restart
+script does) for `UV_THREADPOOL_SIZE` to apply; a process started another way keeps Node's 4 threads
+unless the variable is set. CLI builds have no health state; their inspects are raced.
+
+#### Review
+
+**Verdict: SHIP AFTER FIXES** (`ds-review.md` in the job's scratch). No High. The parent's rulings
+on each finding were applied as below.
+
+| Finding | Ruling | Where |
+|---|---|---|
+| M1: a queued `onDrive` call had no deadline | The review's fix: the `overdue` count, refuse at once and re-mark when every slot is overdue, the slot wait raced against the budget and refused without marking, `refuseWaiters` on every transition to stalled; the named test | `33094c29` |
+| M2: nothing protected an idle boot or a hand-typed root | The 15 s health pass armed above the idle gate, the five-minute pass below it; a path on no entry capped by the root it is under; the limit stated above | `bd39579e`, `33094c29`, this commit |
+| M3: the snapshot walk was uncapped and 16 wide | Its per-video unit (and its listing and keep-latest reads) through `onDrive(config.dataDir, …)`; the three "at most four threads" sentences (this record's Found and left, FACTS, the `onDrive` header) say what is true after it | `ff235c2f`, `33094c29`, this commit |
+| L1: the Storage stage's `statfs` | Through `onDrive`, "—" on a refusal | `19c5842d` |
+| L2: discards and flushes; requeued requests | Fields 12 and 16 added to completed; one FACTS line on requeued requests | `33094c29`, this commit |
+| L3: stale comments calling the child stat the detector | Fixed (the five named, and the health pass's header) | `33094c29`, `bd39579e`, `19c5842d` |
+| L4: "a drive that is not there" for a stalled drive | The pause record carries the cause; both cases worded | `bd39579e` |
+| L5: the detector's samples per module copy; "Refresh asks again at once" | Samples on `globalThis`; the changelog's wording | `33094c29`, this commit |
+| L6: a spin-up longer than the budget | Slow, not stalled, when the counters moved since the call began; the spin-down question above | `33094c29`, this commit |
+| L7: a candidate-root probe shared the location's slots | Keyed by its root | `33094c29` |
+| L8: the budget covers a whole unit | Said in the doc and above | `33094c29` |
+| L9: `stalled` on no location was not down | Counted as down | `bd39579e` |
+| L10: a findmnt every pass | Only on a root change or a failed `/sys` read | `33094c29` |
+| The two questions | Keep the four-call cap; keep the 10 s spacing | as built |
+
+**Re-review: SHIP AFTER FIXES.** Every finding above was confirmed closed (L6 for a busy drive; the
+spin-up case stays the operator's question). Two new ones:
+
+| Finding | Ruling | Where |
+|---|---|---|
+| M4: a slot wait was timed from when the call queued, so a deep queue on a busy but answering drive was refused as "not answering" (units of about 1.1 s at 16 wide, 0.46 s at 32, 0.2 s at 64) | The deadline follows progress: every return on the key re-arms its waiters, and a wait is refused only when nothing on the key has returned for the budget; the overdue and transition refusals unchanged. The limit statement and the keep-latest comment corrected | `d61bdd93`, this commit |
+| L11: the health pass's block sat inside the boot probe's comment, which still called itself the only thing an idle boot runs | Moved above it; the phrase dropped | `c367a3d7` |
+| (the implementer's note) four slow units past the budget on a disk whose counters show completions made the overdue refusal mark the location stalled | The same treatment as L6: the overdue refusal compares the counters with those taken when the oldest overdue call began; moved → refused, not marked; unchanged → marked | `9a12308e`, this commit |
+
## Rollout