Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 9752ed15228a5708a4c7e3cd247699e9f13bf2de
parent 8315b8cb4b83ecb9d89a2e5470b6c80de9f3068b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue, 29 Sep 2026 22:59:42 -0400

common: the health pass reads the block device's request counters; a child stat only when no device can be named

Ruling Q1(a). A child `stat` of a location's root is answered from the
kernel's inode cache whenever the drive was used lately, so it can say "ok"
while every read that reaches the device waits out a reset loop.

detectLocationHealth (lib/storageVolumes.ts), once per location per pass:
- names the root's block device with `findmnt -J -T <root> -o SOURCE,UUID`,
  run as a child raced against 3 s (a findmnt that does not answer reuses the
  device the last pass named for that root); a `[subvolume]` suffix is taken
  off and /dev/mapper links are resolved to their dm-N; a UUID other than the
  location's recorded one names no device;
- reads `/sys/class/block/<dev>/stat` (never the drive): reads completed
  (field 1) + writes completed (5), and requests in flight (9), and compares
  with the previous pass's sample for the location (same device, at least
  10 s earlier): stalled ⇔ in flight at both AND no completion between;
  anything else is a clean answer. The first sample gives no verdict.
- with no device (a container, no findmnt, a tmpfs or network source, no /sys
  entry) falls back to the child `stat` probe.
The verdict names its detector, and the health state records it
(`detector: "counters" | "stat"`); /storage's not-answering line says which.
The pass registers every configured location first (so the watchdog can find
a channel's location before any verdict); /storage's Refresh asks the same
detector and gets no verdict inside the 10 s minimum.

Tests: the stat line parser, the verdict over sample pairs, device names, and
the detector end to end with a fake findmnt and a temp /sys: no verdict on the
first sample, stuck → stalled, drained or moving → ok, a too-soon sample,
every no-device fallback (tmpfs, findmnt failing, no /sys entry, another
volume's UUID, no binary), a findmnt that does not answer; the pass records a
verdict's detector and a verdict with no answer changes nothing.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/storageWatch.test.ts | 30+++++++++++++++++++++++++++++-
Mcommon/controller/storageWatch.ts | 90++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----------------------
Mcommon/lib/storageHealth.ts | 42++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/storageHealthCounters.test.ts | 213+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/storageVolumes.ts | 217+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Meditor/app/storage/buildStorage.ts | 8+++++++-
6 files changed, 569 insertions(+), 31 deletions(-)

diff --git a/common/controller/storageWatch.test.ts b/common/controller/storageWatch.test.ts @@ -413,7 +413,8 @@ test("one missed probe stalls the location; pages then answer 'stalled' without log: (l) => lines.push(l), }); assert.deepEqual(r.answers, { cold: "stalled" }); - assert.deepEqual(r.transitions, [{ id: "cold", from: null, to: "stalled" }]); + // Registered first (answering), so the miss is a transition from ok. + assert.deepEqual(r.transitions, [{ id: "cold", from: "ok", to: "stalled" }]); assert.equal(locationHealth("cold")?.state, "stalled"); assert.match(lines.join("\n"), /"cold": drive not answering/); // The target is there and would answer, but nothing asks it. @@ -517,3 +518,30 @@ test("a Refresh asks one location now: it counts as one answer, and prunes nothi assert.equal(locationHealth("other")?.state, "stalled"); }); }); + +test("the pass registers every location, records a verdict's detector, and a verdict with no answer changes nothing", async () => { + await withTmp(async (h) => { + const verdicts = [ + { answer: null, detector: "counters" as const, device: "sdz1" }, + { + answer: "stalled" as const, + detector: "counters" as const, + device: "sdz1", + cause: "its disk (sdz1) had 1 request(s) in flight and completed none in 15 s", + }, + ]; + let i = 0; + const probe = async () => verdicts[i++]; + const lines: string[] = []; + const first = await runStorageHealthPass({ io: h.io, probe, log: (l) => lines.push(l) }); + // Registered, answering, and the detector named — with no verdict yet. + assert.deepEqual(first.answers, {}); + assert.deepEqual(first.transitions, []); + assert.equal(locationHealth("cold")?.state, "ok"); + assert.equal(locationHealth("cold")?.detector, "counters"); + const second = await runStorageHealthPass({ io: h.io, probe, log: (l) => lines.push(l) }); + assert.deepEqual(second.transitions, [{ id: "cold", from: "ok", to: "stalled" }]); + assert.match(String(locationHealth("cold")?.cause), /its disk \(sdz1\)/); + assert.match(lines.join("\n"), /"cold": drive not answering — its disk \(sdz1\)/); + }); +}); diff --git a/common/controller/storageWatch.ts b/common/controller/storageWatch.ts @@ -24,7 +24,8 @@ import { type StorageLocation, } from "../lib/storageLocations"; import { - probeLocationHealth, + detectLocationHealth, + type HealthVerdict, type LocationHealthProbe, type VolumeBins, } from "../lib/storageVolumes"; @@ -32,8 +33,10 @@ import { HEALTH_PROBE_INTERVAL_MS, HEALTH_PROBE_TIMEOUT_MS, NOT_ANSWERING, + noteLocationDetector, pruneLocationHealth, recordLocationHealth, + registerLocationHealth, type HealthTransition, type LocationHealthState, } from "../lib/storageHealth"; @@ -359,8 +362,10 @@ export type StorageHealthPassOpts = { // Default: the configured locations, read from settings. locations?: readonly StorageLocation[]; io?: { read: () => SiteSettings }; - // Test seam. Default: `probeLocationHealth`, a child `stat` against a 3 s - // timer. + // findmnt, for naming each root's block device. Default: getPaths(). + bins?: Pick<VolumeBins, "findmntBin">; + // Test seam. Default: `detectLocationHealth` — the block device's counters, + // or a child `stat` against a 3 s timer when no device can be named. probe?: LocationHealthProbe; now?: () => number; log?: (line: string) => void; @@ -373,6 +378,32 @@ export type StorageHealthPassResult = { transitions: HealthTransition[]; }; +// A probe's answer as a verdict. A bare state names no detector; a probe that +// threw is "could not ask": `ok`, never `stalled`. +function asVerdict(answer: LocationHealthState | HealthVerdict): HealthVerdict { + return typeof answer === "string" ? { answer } : answer; +} + +const STAT_CAUSE = `a stat of its root did not answer within ${HEALTH_PROBE_TIMEOUT_MS / 1000} s`; + +// Record one verdict. A verdict with no answer records nothing but the +// detector that gave it. +function recordVerdict( + loc: StorageLocation, + verdict: HealthVerdict, + now: number, +): HealthTransition | null { + if (verdict.answer === null) { + if (verdict.detector) noteLocationDetector(loc.id, verdict.detector); + return null; + } + return recordLocationHealth(loc, verdict.answer, { + now, + cause: verdict.cause ?? STAT_CAUSE, + ...(verdict.detector ? { detector: verdict.detector } : {}), + }); +} + export async function runStorageHealthPass( opts: StorageHealthPassOpts = {}, ): Promise<StorageHealthPassResult> { @@ -380,16 +411,21 @@ export async function runStorageHealthPass( const locations = opts.locations ?? (opts.io ?? DEFAULT_IO).read().storage.locations; pruneLocationHealth(locations.map((l) => l.id)); + // Every configured location has an entry before anything is asked, so the + // watchdog (lib/storageHealth.ts `onDrive`) can find a channel's location + // even before the counters have given a first verdict. + registerLocationHealth(locations); + const bins = opts.bins ?? getPaths(); const probe: LocationHealthProbe = - opts.probe ?? ((loc) => probeLocationHealth(loc)); + opts.probe ?? ((loc) => detectLocationHealth(loc, bins)); // Every location at once: each answer is bounded by the probe's own timer, // so the pass is too, and one stalled drive does not delay the others. const answers = await Promise.all( locations.map(async (loc) => { - const answer = await probe(loc).catch( - (): LocationHealthState => "ok", - ); - return [loc, answer] as const; + const verdict = await probe(loc).then(asVerdict, (): HealthVerdict => ({ + answer: "ok", + })); + return [loc, verdict] as const; }), ); const out: StorageHealthPassResult = { @@ -398,19 +434,15 @@ export async function runStorageHealthPass( transitions: [], }; const now = opts.now?.() ?? Date.now(); - for (const [loc, answer] of answers) { - out.answers[loc.id] = answer; - const t = recordLocationHealth(loc, answer, { - now, - cause: `a stat of its root did not answer within ${HEALTH_PROBE_TIMEOUT_MS / 1000} s`, - }); + for (const [loc, verdict] of answers) { + if (verdict.answer !== null) out.answers[loc.id] = verdict.answer; + const t = recordVerdict(loc, verdict, now); if (!t) continue; out.transitions.push(t); if (t.to === "stalled") { log( - `[storage] "${loc.id}": ${NOT_ANSWERING} — a stat of its root did ` + - `not answer within ${HEALTH_PROBE_TIMEOUT_MS / 1000} s; pages and ` + - `polls skip it until two probes in a row answer`, + `[storage] "${loc.id}": ${NOT_ANSWERING} — ${verdict.cause ?? STAT_CAUSE}; ` + + `pages and polls skip it until two passes in a row find it answering`, ); } else if (t.from === "stalled") { log(`[storage] "${loc.id}": answering again (${t.to})`); @@ -422,16 +454,20 @@ export async function runStorageHealthPass( // ONE LOCATION, NOW: what /storage's Refresh asks before its own probe, so the // operator pressing it after doing something about the drive gets an answer // taken afterwards. It counts as one answer like any other — a stalled location -// still needs two clean ones in a row. Nothing is pruned. +// still needs two clean ones in a row — and the counters give none when their +// last sample is under MIN_COUNTER_INTERVAL_MS old. Nothing is pruned. export async function refreshLocationHealth( loc: StorageLocation, - probe: LocationHealthProbe = (l) => probeLocationHealth(l), -): Promise<LocationHealthState> { - const answer = await probe(loc).catch((): LocationHealthState => "ok"); - recordLocationHealth(loc, answer, { - cause: `a stat of its root did not answer within ${HEALTH_PROBE_TIMEOUT_MS / 1000} s`, - }); - return answer; + probe?: LocationHealthProbe, +): Promise<LocationHealthState | null> { + registerLocationHealth([loc]); + const ask: LocationHealthProbe = + probe ?? ((l) => detectLocationHealth(l, getPaths())); + const verdict = await ask(loc).then(asVerdict, (): HealthVerdict => ({ + answer: "ok", + })); + recordVerdict(loc, verdict, Date.now()); + return verdict.answer; } // --------------------------------------------------------------------------- @@ -439,7 +475,8 @@ export async function refreshLocationHealth( // --------------------------------------------------------------------------- // Fifteen seconds (lib/storageHealth.ts says why). Each pass is one short-lived -// `stat` subprocess per location. +// findmnt per location and a read of its device's counters in /sys (or, with no +// device, one short-lived `stat`). export const STORAGE_HEALTH_INTERVAL_MS = HEALTH_PROBE_INTERVAL_MS; // A per-module-copy singleton, deliberately left so: it is not a temp-file @@ -479,6 +516,7 @@ export function startStorageWatch( healthInFlight = true; void runStorageHealthPass({ io: opts.io, + bins: opts.bins ?? opts.paths, probe: opts.healthProbe, log: opts.log, }) diff --git a/common/lib/storageHealth.ts b/common/lib/storageHealth.ts @@ -211,6 +211,13 @@ export function registerLocationHealth( } } +// Which detector decided a location's last answer, when it gave none (the +// counters' first sample has nothing to compare with). +export function noteLocationDetector(id: string, detector: HealthDetector): void { + const h = healthState().byId.get(id); + if (h) h.detector = detector; +} + // Drop every location that is no longer configured. export function pruneLocationHealth(liveIds: Iterable<string>): void { const keep = new Set(liveIds); @@ -468,3 +475,38 @@ export async function onDrive<T>(where: Where, call: () => Promise<T>): Promise< } throw new DriveNotAnsweringError(health); } + +// --------------------------------------------------------------------------- +// The block device's counters +// --------------------------------------------------------------------------- +// +// `/sys/class/block/<dev>/stat` is the kernel's own count of a device's +// requests (Documentation/block/stat.rst): field 1 reads completed, 5 writes +// completed, 9 requests in flight now. Reading it never touches the drive. A +// drive that is merely slow, even one grinding through a long write, keeps +// completing requests; one in a reset loop has requests in flight and +// completes none. So, between two samples a pass apart: +// +// stalled ⇔ in flight at both samples AND no read or write completed between +// ok ⇔ anything else (nothing in flight at one of them, or completions +// moved) +// +// and the health rules above turn one `stalled` into a stall and two `ok`s in +// a row into its end. + +export type BlockStatSample = { completed: number; inFlight: number }; + +export function parseBlockStat(line: string): BlockStatSample | null { + const f = line.trim().split(/\s+/).map(Number); + if (f.length < 9 || f.slice(0, 9).some((n) => !Number.isFinite(n))) return null; + return { completed: f[0] + f[4], inFlight: f[8] }; +} + +export function countersVerdict( + prev: BlockStatSample, + cur: BlockStatSample, +): "ok" | "stalled" { + return prev.inFlight > 0 && cur.inFlight > 0 && cur.completed === prev.completed + ? "stalled" + : "ok"; +} diff --git a/common/lib/storageHealthCounters.test.ts b/common/lib/storageHealthCounters.test.ts @@ -0,0 +1,213 @@ +import { beforeEach, test } from "node:test"; +import assert from "node:assert/strict"; +import path from "node:path"; +import { tmpdir } from "node:os"; +import { chmod, mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; +import { + blockDeviceName, + detectLocationHealth, + MIN_COUNTER_INTERVAL_MS, + resetHealthDetector, +} from "./storageVolumes"; +import { countersVerdict, parseBlockStat } from "./storageHealth"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/storageHealthCounters.test.ts +// +// THE COUNTERS DETECTOR. No test stalls a real drive: findmnt is a fake that +// names a device, and /sys/class/block is a temp directory whose `stat` files +// the test writes — a stalled device is one whose in-flight count stays up +// while its completions stand still. + +beforeEach(() => resetHealthDetector()); + +// A real line from this machine's /sys/class/block/<dev>/stat (17 fields). +const LINE = + "368126412 135952867 36463191157 80435430 29670281 46781260 4001255365 167469727 0 44151394 253218297 877589 0 861936568 4100450 578669 1212688"; + +test("the stat line: reads + writes completed, and requests in flight", () => { + assert.deepEqual(parseBlockStat(LINE), { + completed: 368126412 + 29670281, + inFlight: 0, + }); + // An 11-field line from an older kernel parses the same way. + assert.deepEqual(parseBlockStat("10 0 0 0 5 0 0 0 3 0 0\n"), { completed: 15, inFlight: 3 }); + assert.equal(parseBlockStat(""), null); + assert.equal(parseBlockStat("1 2 3"), null); + assert.equal(parseBlockStat("a b c d e f g h i"), null); +}); + +test("the verdict over a pair of samples", () => { + const s = (completed: number, inFlight: number) => ({ completed, inFlight }); + // In flight at both, nothing completed between: stalled. + assert.equal(countersVerdict(s(100, 2), s(100, 5)), "stalled"); + // Completions moved: slow, perhaps, but answering. + assert.equal(countersVerdict(s(100, 2), s(101, 2)), "ok"); + // Nothing in flight at either end: idle. + assert.equal(countersVerdict(s(100, 0), s(100, 0)), "ok"); + assert.equal(countersVerdict(s(100, 0), s(100, 4)), "ok"); + assert.equal(countersVerdict(s(100, 4), s(100, 0)), "ok"); +}); + +test("a mount source's device name: a partition, a bind or subvolume suffix, nothing that is not /dev", async () => { + assert.equal(await blockDeviceName("/dev/no-such-disk1"), "no-such-disk1"); + assert.equal(await blockDeviceName("/dev/no-such-disk2[/@home]"), "no-such-disk2"); + assert.equal(await blockDeviceName("tmpfs"), null); + assert.equal(await blockDeviceName("server:/export"), null); + assert.equal(await blockDeviceName(""), null); +}); + +// ── the detector end to end, with a fake findmnt and a fake /sys ──────────── + +const FAKE_FINDMNT = `#!/usr/bin/env node +import { readFileSync } from "node:fs"; +import path from "node:path"; +const c = JSON.parse(readFileSync(path.join(import.meta.dirname, "control.json"), "utf8")); +if (c.sleepMs) Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, c.sleepMs); +if (c.exit) process.exit(c.exit); +process.stdout.write(JSON.stringify({ filesystems: [{ source: c.source, uuid: c.uuid ?? null }] }) + "\\n"); +`; + +type H = { + root: string; + bins: { findmntBin: string }; + sys: string; + control: (c: Record<string, unknown>) => Promise<void>; + counters: (device: string, completed: number, inFlight: number) => Promise<void>; +}; + +async function withHarness(fn: (h: H) => Promise<void>): Promise<void> { + const dir = await mkdtemp(path.join(tmpdir(), "ttb-counters-")); + try { + const bin = path.join(dir, "fake-findmnt.mjs"); + await writeFile(bin, FAKE_FINDMNT); + await chmod(bin, 0o755); + const root = path.join(dir, "media"); + await mkdir(root); + const sys = path.join(dir, "sys-block"); + await fn({ + root, + bins: { findmntBin: bin }, + sys, + control: (c) => writeFile(path.join(dir, "control.json"), JSON.stringify(c)), + counters: async (device, completed, inFlight) => { + await mkdir(path.join(sys, device), { recursive: true }); + // Reads `completed`, writes 0, in flight `inFlight`, the rest zero. + await writeFile( + path.join(sys, device, "stat"), + `${completed} 0 0 0 0 0 0 0 ${inFlight} 0 0 0 0 0 0 0 0\n`, + ); + }, + }); + } finally { + await rm(dir, { recursive: true, force: true }); + } +} + +const loc = (root: string, uuid?: string) => ({ + id: "usb", + root, + ...(uuid ? { volume: { uuid, mountpoint: root, relPath: "" } } : {}), +}); + +test("counters: first sample no verdict; stuck → stalled; two clean samples; each pass apart", async () => { + await withHarness(async (h) => { + await h.control({ source: "/dev/fakedisk1", uuid: "u-1" }); + const at = (i: number) => ({ sysBlockDir: h.sys, now: i * 15_000 }); + await h.counters("fakedisk1", 100, 2); + assert.deepEqual(await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(1)), { + answer: null, + detector: "counters", + device: "fakedisk1", + }); + // Still two in flight, nothing completed in 15 s. + await h.counters("fakedisk1", 100, 2); + const stuck = await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(2)); + assert.equal(stuck.answer, "stalled"); + assert.equal(stuck.detector, "counters"); + assert.match(String(stuck.cause), /^its disk \(fakedisk1\) had 2 request\(s\) in flight and completed none in 15 s$/); + // It drains: nothing in flight. + await h.counters("fakedisk1", 140, 0); + assert.equal((await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(3))).answer, "ok"); + // Busy and moving: still ok. + await h.counters("fakedisk1", 190, 3); + assert.equal((await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(4))).answer, "ok"); + }); +}); + +test("counters: a second sample sooner than the minimum interval gives no verdict and keeps the first", async () => { + await withHarness(async (h) => { + await h.control({ source: "/dev/fakedisk1" }); + await h.counters("fakedisk1", 100, 2); + await detectLocationHealth(loc(h.root), h.bins, { sysBlockDir: h.sys, now: 0 }); + const soon = await detectLocationHealth(loc(h.root), h.bins, { + sysBlockDir: h.sys, + now: MIN_COUNTER_INTERVAL_MS - 1, + }); + assert.equal(soon.answer, null); + // Compared with the FIRST sample, not the refused one. + const later = await detectLocationHealth(loc(h.root), h.bins, { + sysBlockDir: h.sys, + now: 15_000, + }); + assert.equal(later.answer, "stalled"); + }); +}); + +test("no device → the child stat, and the verdict says so", async () => { + await withHarness(async (h) => { + const opts = { sysBlockDir: h.sys, now: 0 }; + // A tmpfs or network source names no block device. + await h.control({ source: "tmpfs" }); + assert.deepEqual(await detectLocationHealth(loc(h.root), h.bins, opts), { + answer: "ok", + detector: "stat", + }); + // findmnt cannot say (no such path, a container without it). + await h.control({ exit: 1 }); + assert.equal((await detectLocationHealth(loc(h.root), h.bins, opts)).detector, "stat"); + assert.equal( + (await detectLocationHealth(loc(path.join(h.root, "gone")), h.bins, opts)).answer, + "absent", + ); + // A device with no /sys entry. + await h.control({ source: "/dev/nodisk9" }); + assert.equal((await detectLocationHealth(loc(h.root), h.bins, opts)).detector, "stat"); + // A root whose filesystem is not the location's recorded volume. + await h.counters("fakedisk1", 1, 0); + await h.control({ source: "/dev/fakedisk1", uuid: "someone-else" }); + assert.equal( + (await detectLocationHealth(loc(h.root, "u-1"), h.bins, opts)).detector, + "stat", + ); + // No findmnt binary at all. + assert.equal( + ( + await detectLocationHealth( + loc(h.root), + { findmntBin: path.join(h.root, "no-findmnt") }, + opts, + ) + ).detector, + "stat", + ); + }); +}); + +test("a findmnt that does not answer: the last device named for that root is read", async () => { + await withHarness(async (h) => { + await h.control({ source: "/dev/fakedisk1" }); + await h.counters("fakedisk1", 100, 2); + await detectLocationHealth(loc(h.root), h.bins, { sysBlockDir: h.sys, now: 0 }); + await h.control({ sleepMs: 5_000, source: "/dev/fakedisk1" }); + const started = Date.now(); + const v = await detectLocationHealth(loc(h.root), h.bins, { + sysBlockDir: h.sys, + now: 15_000, + timeoutMs: 300, + }); + assert.ok(Date.now() - started < 2_000, "not waited for"); + assert.equal(v.detector, "counters"); + assert.equal(v.answer, "stalled"); + }); +}); diff --git a/common/lib/storageVolumes.ts b/common/lib/storageVolumes.ts @@ -1,14 +1,18 @@ import path from "node:path"; -import { lstat, stat } from "node:fs/promises"; +import { lstat, readFile, realpath, stat } from "node:fs/promises"; import { execa } from "execa"; import type { Paths } from "./paths"; import { getFreeBytes } from "./diskSpace"; import type { StorageLocation, StorageVolume } from "./storageLocations"; import { HEALTH_PROBE_TIMEOUT_MS, + countersVerdict, isDriveNotAnswering, onDrive, + parseBlockStat, stalledLocation, + type BlockStatSample, + type HealthDetector, type LocationHealthState, } from "./storageHealth"; @@ -360,9 +364,22 @@ export type HealthProbeOptions = { statBin?: string; }; +// One pass's answer about one location. `answer: null` is no verdict (the +// counters' first sample, or a second one taken too soon after the last). +export type HealthVerdict = { + answer: LocationHealthState | null; + detector?: HealthDetector; + // For a `stalled` answer: what did not answer, in words with no path. + cause?: string; + // The block device the counters were read from. + device?: string; +}; + +// What the health pass asks about a location. A bare state is a verdict with +// no detector named (the tests' scripted probes). export type LocationHealthProbe = ( - loc: Pick<StorageLocation, "id" | "root">, -) => Promise<LocationHealthState>; + loc: Pick<StorageLocation, "id" | "label" | "root" | "volume">, +) => Promise<LocationHealthState | HealthVerdict>; export async function probeLocationHealth( loc: Pick<StorageLocation, "root">, @@ -409,6 +426,200 @@ export async function probeLocationHealth( return answer; } +// --------------------------------------------------------------------------- +// The counters detector: the block device's own request counters +// --------------------------------------------------------------------------- +// +// THE STAT PROBE ABOVE CAN BE ANSWERED FROM THE KERNEL'S CACHE. A root's inode +// is cached whenever anything has used the drive lately, so a child `stat` of +// it answers in microseconds while the reads that actually reach the device +// wait out a reset loop. The block device's counters (lib/storageHealth.ts, +// `countersVerdict`) are the device's own account of what it has done, and +// reading them touches only /sys. So the health pass asks this first: +// +// 1. the root's device, once per pass: `findmnt -J -T <root> -o SOURCE,UUID` +// as a child raced against 3 s (a findmnt stuck resolving the root holds +// nothing of ours), the `[subvolume]` suffix a bind or btrfs mount adds +// taken off, `/dev/mapper/<x>` resolved to its `dm-N`, then the basename. +// A partition and a mapper device both have `/sys/class/block/<name>/stat`. +// A findmnt that timed out reuses the device the last pass found for that +// root; one whose UUID is not the location's recorded one names none (the +// root is then a directory on some other filesystem, not the drive). +// 2. that device's `stat` line, compared with the previous pass's sample for +// the location (same device, at least MIN_COUNTER_INTERVAL_MS earlier). +// +// NO DEVICE (a container, no findmnt, a network or tmpfs mount, no /sys entry) +// FALLS BACK TO THE CHILD `stat`, and the verdict says which detector answered. + +export type DetectorOptions = HealthProbeOptions & { + // Test seams: where /sys/class/block is, and the clock. + sysBlockDir?: string; + now?: number; +}; + +export const SYS_BLOCK_DIR = "/sys/class/block"; +// Two samples closer than this are not compared: a healthy drive can have a +// request in flight at two instants a moment apart without completing one. +// The pass is 15 s apart; a /storage Refresh just after a pass gives no verdict. +export const MIN_COUNTER_INTERVAL_MS = 10_000; + +type CounterSample = BlockStatSample & { device: string; at: number }; + +type DetectorState = { + samples: Map<string, CounterSample>; + deviceByRoot: Map<string, string>; +}; + +// Module state, not globalThis: only the health pass (one module copy, armed +// from instrumentation) and /storage's Refresh read it, and a Refresh reaching +// another copy just starts that copy's comparison over. +const detector: DetectorState = { samples: new Map(), deviceByRoot: new Map() }; + +export function resetHealthDetector(): void { + detector.samples.clear(); + detector.deviceByRoot.clear(); +} + +type Raced = { answered: false } | ({ answered: true } & Run); + +// `run`, but raced against a timer that nobody waits past: a child stuck in +// the kernel is sent SIGKILL and left to exit when it can. +async function runRaced(bin: string, args: string[], timeoutMs: number): Promise<Raced> { + let child: ReturnType<typeof execa>; + try { + child = execa(bin, args, { buffer: true, reject: false, stdin: "ignore" }); + } catch { + return { answered: true, ok: false, exitCode: undefined, stdout: "", stderr: "" }; + } + const answered: Promise<Raced> = child.then( + (res) => { + const exitCode = typeof res.exitCode === "number" ? res.exitCode : undefined; + return { + answered: true as const, + ok: exitCode === 0, + exitCode, + stdout: typeof res.stdout === "string" ? res.stdout : "", + stderr: typeof res.stderr === "string" ? res.stderr : "", + }; + }, + () => ({ answered: true as const, ok: false, exitCode: undefined, stdout: "", stderr: "" }), + ); + let timer: ReturnType<typeof setTimeout> | undefined; + const timedOut = new Promise<Raced>((resolve) => { + timer = setTimeout(() => resolve({ answered: false }), timeoutMs); + }); + const out = await Promise.race([answered, timedOut]); + if (timer) clearTimeout(timer); + if (!out.answered) { + try { + child.kill("SIGKILL"); + } catch { + /* already gone */ + } + } + return out; +} + +// The block device name under /sys/class/block for a mount SOURCE, or null. +export async function blockDeviceName(source: string): Promise<string | null> { + const bare = source.replace(/\[.*\]$/, "").trim(); + if (!bare.startsWith("/dev/")) return null; + // /dev/mapper/<x> and /dev/disk/by-*/<x> are links to the kernel's name. + // Resolving them reads /dev, never the drive. + const resolved = await realpath(bare).catch(() => bare); + const name = path.basename(resolved); + return name && name !== "dev" ? name : null; +} + +async function blockDeviceOfRoot( + loc: Pick<StorageLocation, "root" | "volume">, + bins: Pick<VolumeBins, "findmntBin">, + timeoutMs: number, +): Promise<{ device: string | null; timedOut: boolean }> { + const res = await runRaced( + bins.findmntBin, + ["-J", "-T", loc.root, "-o", "SOURCE,UUID"], + timeoutMs, + ); + if (!res.answered) return { device: null, timedOut: true }; + if (!res.ok) return { device: null, timedOut: false }; + let fs0: Record<string, unknown> | undefined; + try { + fs0 = (JSON.parse(res.stdout) as { filesystems?: Record<string, unknown>[] }) + ?.filesystems?.[0]; + } catch { + return { device: null, timedOut: false }; + } + const source = typeof fs0?.source === "string" ? fs0.source : ""; + const uuid = typeof fs0?.uuid === "string" ? fs0.uuid : ""; + const recorded = loc.volume?.uuid?.trim() ?? ""; + if (recorded && uuid && uuid !== recorded) return { device: null, timedOut: false }; + return { device: await blockDeviceName(source), timedOut: false }; +} + +async function readBlockStat( + device: string, + sysBlockDir: string, +): Promise<BlockStatSample | null> { + try { + return parseBlockStat(await readFile(path.join(sysBlockDir, device, "stat"), "utf8")); + } catch { + return null; + } +} + +// One location's verdict for the health pass: the counters when its device can +// be named and read, the child `stat` otherwise. +export async function detectLocationHealth( + loc: Pick<StorageLocation, "id" | "root" | "volume">, + bins: Pick<VolumeBins, "findmntBin">, + opts: DetectorOptions = {}, +): Promise<HealthVerdict> { + const now = opts.now ?? Date.now(); + const timeoutMs = opts.timeoutMs ?? HEALTH_PROBE_TIMEOUT_MS; + const mapped = loc.root.trim() + ? await blockDeviceOfRoot(loc, bins, timeoutMs) + : { device: null, timedOut: false }; + const device = + mapped.device ?? (mapped.timedOut ? (detector.deviceByRoot.get(loc.root) ?? null) : null); + if (device) { + const sample = await readBlockStat(device, opts.sysBlockDir ?? SYS_BLOCK_DIR); + if (sample) { + detector.deviceByRoot.set(loc.root, device); + const prev = detector.samples.get(loc.id); + if (prev && prev.device === device && now - prev.at < MIN_COUNTER_INTERVAL_MS) { + return { answer: null, detector: "counters", device }; + } + detector.samples.set(loc.id, { ...sample, device, at: now }); + if (!prev || prev.device !== device) { + return { answer: null, detector: "counters", device }; + } + const answer = countersVerdict(prev, sample); + return { + answer, + detector: "counters", + device, + ...(answer === "stalled" + ? { + cause: + `its disk (${device}) had ${sample.inFlight} request(s) in flight and ` + + `completed none in ${Math.round((now - prev.at) / 1000)} s`, + } + : {}), + }; + } + } + detector.samples.delete(loc.id); + const answer = await probeLocationHealth(loc, opts); + return { + answer, + detector: "stat", + ...(answer === "stalled" + ? { cause: `a stat of its root did not answer within ${timeoutMs / 1000} s` } + : {}), + }; +} + // udisksctl availability, memoised per binary path. Same shape as a digest // app's probe (`digestApps.ts` claudeCode.probe): `--version`, reject:false, // short timeout. Memoised because /storage asks once per render and the answer diff --git a/editor/app/storage/buildStorage.ts b/editor/app/storage/buildStorage.ts @@ -94,9 +94,15 @@ export async function buildStorage(): Promise<StorageRowsPayload> { for (const loc of locations) { const stall = stalledLocation(loc); if (stall) { + const watched = + stall.detector === "counters" + ? " Watched through its disk's request counters." + : stall.detector === "stat" + ? " Watched with a stat of its root (no disk could be named here)." + : ""; notAnswering[loc.id] = `${notAnsweringText(stall, now)} — ${stall.cause ?? "its root did not answer"}. ` + - `Pages and polls skip this drive until it answers twice in a row.`; + `Pages and polls skip this drive until it answers twice in a row.${watched}`; } } const store = await inspectSavedVideosStore(paths, settings);