Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d70e34ef60c382e5ae0a3e20caf74b9e18f1b101
parent cbe8cf07ee7056160908c231612dd03b943c6cc0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Wed, 30 Sep 2026 00:52:06 -0400

Merge main (release 15 DS, CF, UT; release 14 S1, HS) into r15/site-scope — release-15.md kept main's sections whole (IG, UT, DS) and put SS's after them, before the Rollout, with SS's row after UT's in the slices table; FACTS and the editor changelog merged cleanly

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
MENVIRONMENT.md | 1+
MSETTINGS.md | 1+
Mcommon/bin/doctor.ts | 4+++-
Mcommon/controller/buildIndex.ts | 8++++++--
Mcommon/controller/buildStats.ts | 4+++-
Mcommon/controller/channelSnapshot.ts | 26+++++++++++++++++++++++---
Mcommon/controller/channels.ts | 85++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----------------------
Mcommon/controller/evictClipWindows.ts | 4+++-
Mcommon/controller/keptVideos.ts | 31+++++++++++++++++++++++--------
Mcommon/controller/recencyIndex.ts | 60+++++++++++++++++++++++++++++++++++++++++++++++++++---------
Mcommon/controller/relocateChannelMedia.ts | 46++++++++++++++++++++++++++++++++++++++++++----
Mcommon/controller/relocateSavedVideos.ts | 35++++++++++++++++++++++++++++++++++-
Mcommon/controller/savedVideoInventory.ts | 75+++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------------
Mcommon/controller/storageLocations.ts | 135+++++++++++++++++++++++++++++++++++++++++++++++++++----------------------------
Acommon/controller/storageStall.test.ts | 544+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/storageWatch.test.ts | 218++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mcommon/controller/storageWatch.ts | 264++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----
Mcommon/lib/channelMedia.ts | 194++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Mcommon/lib/channelMediaHold.ts | 1+
Mcommon/lib/channelPriority.ts | 26+++++++++++++++++++++++++-
Mcommon/lib/envVars.ts | 1+
Mcommon/lib/paths.ts | 7++++---
Mcommon/lib/ports.test.ts | 6++++++
Acommon/lib/storageHealth.test.ts | 544+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/storageHealth.ts | 749+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/storageHealthCounters.test.ts | 231+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/storageHealthProbe.test.ts | 112+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/storageVolumes.ts | 368+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mcommon/styles/tokens.css | 17+++++++++++++++++
Mcommon/views/pipeline/stageStatus.ts | 3++-
Mcommon/views/storage.test.ts | 33+++++++++++++++++++++++++++++++++
Mcommon/views/storage.ts | 11+++++++++++
Mdocker/entrypoint.sh | 6++++++
Meditor/CHANGELOG.md | 2++
Meditor/app/api/channels/[slug]/videos/[id]/files/[name]/route.ts | 25+++++++++++++++++++++++--
Meditor/app/api/test/invalidate-cache/route.ts | 8++++++++
Aeditor/app/channels/[slug]/components/MediaNotAnswering.tsx | 67+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/components/stages/StorageStage.tsx | 8+++++---
Meditor/app/channels/[slug]/page.tsx | 48+++++++++++++++++++++++++++++++++++++++++++++---
Meditor/app/channels/[slug]/videos/[id]/page.tsx | 181++++++++++++++++++++++++++++++++++++++++++++++++++++++-------------------------
Meditor/app/channels/[slug]/videos/page.tsx | 58+++++++++++++++++++++++++++++++++++++++++++++++-----------
Meditor/app/channels/components/ChannelVolumeBar.tsx | 11+++++++++--
Meditor/app/channels/page.tsx | 15+++++++++++++++
Meditor/app/components/MediaLocationBadge.tsx | 11+++++++----
Meditor/app/saved-videos/page.tsx | 21++++++++++++++++++++-
Meditor/app/storage/actions.ts | 24++++++++++++++++++++++++
Meditor/app/storage/buildStorage.ts | 46+++++++++++++++++++++++++++++++++++++++++-----
Meditor/app/storage/components/StorageLocationsTable.tsx | 9+++++++++
Meditor/instrumentation.ts | 36++++++++++++++++++++++++++++--------
Meditor/package.json | 2+-
Mhomepage/CHANGELOG.md | 1+
Mhomepage/app/components/ArchiveGrowthChart.tsx | 80++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------
Mhomepage/app/lib/growthGaps.test.ts | 167+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mhomepage/app/lib/growthGaps.ts | 129++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-------------
Mhomepage/e2e/fixture-summary.ts | 5++++-
Mhomepage/e2e/growth-chart.spec.ts | 130+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----
Mhomepage/e2e/instance-colours.spec.ts | 56+++++++++++++++++++++++++++++++++++++++++++++-----------
Mhomepage/e2e/marketing.spec.ts | 4+++-
Mplans/FACTS.md | 176++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-------
Mplans/release-14.md | 254+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mplans/release-15.md | 600++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mscripts/next-build-trace.test.mjs | 236+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Mumtool/app/api/clip/[key]/audio/route.ts | 12++++++------
Mumtool/app/api/clip/[key]/video/route.ts | 8++++----
Mumtool/app/api/face/frame/route.ts | 8++++----
Mumtool/lib/paths.mjs | 21+++++++++++++++++++--
Mumtool/lib/paths.ts | 1+
Mumtool/next.config.ts | 32+++++++++++++++++++++++++-------
68 files changed, 5974 insertions(+), 368 deletions(-)

diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md @@ -77,6 +77,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not | `AUDIO_CHECK_BACKOFF_FACTOR` | the built-in factor | The audio check's interval backoff factor, in (0, 1], for a one-off run. | common/ytdlp/audioCheckedDownload.ts | | `ARCHILYZER_STATS_ALLOW_DOWNGRADE` | off | `1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions. | common/controller/buildStats.ts | | `ARCHILYZER_INDEX_ALLOW_HELD` | off | `1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel. | common/controller/buildIndex.ts | +| `UV_THREADPOOL_SIZE` | `16` for the editor (`4` is Node's own) | Threads in Node's pool for filesystem calls. A call on a stalled drive holds one until the drive answers, so the editor starts with 16. It buys time for calls already in flight and isolates nothing: the storage health probe and its gate keep new calls off a stalled drive. | Node's libuv (set by editor/package.json `start` and docker/entrypoint.sh) | | `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts | | `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool | | `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs | diff --git a/SETTINGS.md b/SETTINGS.md @@ -449,6 +449,7 @@ Per entry — each entry spells its own values. | `reason` | One reason today. A union so a second one has somewhere to go, and so a surface can say WHICH machine decided rather than "automatic". | | `since` | ISO, for "auto-paused — media unreachable since <date>". | | `previousTier` | The base tier the channel had before the machine paused it; what a restore puts back. Never `paused` (that would restore to paused — a no-op dressed as a restore). | +| `cause` | `not-there` (the drive is unmounted or unplugged) or `not-answering` (it is there and does not answer: a stalled disk). Only what the words on /review, the rack and the channel page say. Optional: a record written before it existed reads as `not-there`. | Default: diff --git a/common/bin/doctor.ts b/common/bin/doctor.ts @@ -121,7 +121,9 @@ export async function collectDoctorReport(deps: DoctorDeps): Promise<DoctorRepor const unreachable: string[] = []; const { inspectChannelMedia } = await import("../lib/channelMedia"); for (const slug of channelSlugs) { - const loc = await inspectChannelMedia(paths, slug); + const loc = await inspectChannelMedia(paths, slug, undefined, { + fresh: true, + }); if (loc.status !== "ok" && loc.status !== "in-place") { unreachable.push(`${slug}: ${loc.status} — ${loc.detail ?? ""}`.trim()); } diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -345,7 +345,9 @@ async function scanSource( // An unmounted drive is not an empty channel (lib/channelMedia.ts): the // readdir below would fail, and every record the channel has would be // removed as gone. - const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg); + const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg, { + fresh: true, + }); if (isMediaHeld(media.status)) { held.set(ch.name, heldReason(media, cfg.dataDir, locations)); continue; @@ -462,7 +464,9 @@ async function scanSource( // Asked again after the walk: a drive that went away DURING it leaves the // videos after that point missing from this scan, which would remove them. // Three syscalls a channel. - const after = await inspectChannelMedia({ channelsDir }, ch.name, cfg); + const after = await inspectChannelMedia({ channelsDir }, ch.name, cfg, { + fresh: true, + }); if (isMediaHeld(after.status)) { held.set(ch.name, heldReason(after, cfg.dataDir, locations)); continue; diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts @@ -288,7 +288,9 @@ async function scanSource( channels.set(ch.name, cfg); // An unmounted drive is not an empty channel (lib/channelMedia.ts): the // readdir below would fail and every one of its stats would be removed. - const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg); + const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg, { + fresh: true, + }); if (isMediaHeld(media.status)) { held.set(ch.name, heldReason(media, cfg.dataDir, locations)); continue; diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -22,6 +22,7 @@ import { } from "../lib/availability"; import { resolveCookiePolicy } from "../lib/cookiePolicy"; import { assertChannelMediaReachable } from "../lib/channelMedia"; +import { isDriveNotAnswering, onDrive } from "../lib/storageHealth"; import { CLIPS_DIR_NAME } from "../lib/clipWindow"; import { getSettings } from "../lib/settings"; import { @@ -737,6 +738,19 @@ export async function generateChannelSnapshot( const config = await readChannelConfig(paths, slug); await assertChannelMediaReachable(paths, slug, config); + // A CHANNEL ON ANOTHER DRIVE IS WALKED THROUGH THE WATCHDOG + // (lib/storageHealth.ts `onDrive`): the data/ listing, the keep-latest keys and + // each video directory's unit below. At most four of them wait on that drive + // at once — this walk runs in the editor's own process after every download + // or sync of the channel, sixteen wide, which is exactly while a long write is + // stressing the drive — and one that does not answer in 3 s throws, so the + // scheduler keeps the last good snapshot.json, as on any failed refresh. + // The reconcile pass just below is sequential (one read at a time) and is + // not raced. + const drive = config?.dataDir?.trim() || undefined; + const through = <T>(read: () => Promise<T>): Promise<T> => + drive ? onDrive(drive, read) : read(); + // Heal any video dir that drifted from the canonical id layout before we read // data/* (best-effort; never fail snapshot generation on a reconcile error). try { @@ -753,7 +767,11 @@ export async function generateChannelSnapshot( maybeMissingRecord, roster, ] = await Promise.all([ - readdir(dataDir, { withFileTypes: true }).catch(() => [] as Dirent[]), + through(() => readdir(dataDir, { withFileTypes: true })).catch((err) => { + // A drive that did not answer is not an empty channel: rethrown. + if (isDriveNotAnswering(err)) throw err; + return [] as Dirent[]; + }), readPlaylistUrls(playlistPath), readArchive(archivePath), loadFailedTranscriptions(paths, slug), @@ -769,6 +787,7 @@ export async function generateChannelSnapshot( paths, channelSlug: slug, keepLatest: config?.keepLatest ?? 0, + through, }); const videoDirNames = dirEntries .filter((d) => d.isDirectory()) @@ -821,7 +840,8 @@ export async function generateChannelSnapshot( const limit = pLimit(SNAPSHOT_VIDEO_CONCURRENCY); const perVideo = await Promise.all( videoDirNames.map((id) => - limit(async () => { + // One video directory's reads are one unit through the watchdog. + limit(() => through(async () => { const dir = path.join(dataDir, id); const files = await readVideoFiles(dir, { checkUntranscribable: true }); // EVERY FILE IN THE DIR, STATTED ONCE, feeding two numbers. @@ -952,7 +972,7 @@ export async function generateChannelSnapshot( vttProvenance, digest, }; - }), + })), ), ); diff --git a/common/controller/channels.ts b/common/controller/channels.ts @@ -16,7 +16,8 @@ import { readVideoFiles, } from "../lib/videoStatus"; import { loadDigest } from "../lib/digest-server"; -import { readRelocationMarker } from "../lib/channelMedia"; +import { channelMediaStall, readRelocationMarker } from "../lib/channelMedia"; +import { isDriveNotAnswering, onDrive } from "../lib/storageHealth"; // TYPE-ONLY, and it must stay that way: ./channelSnapshot imports // readChannelConfig from this module, and it drags in the snapshot generator's // whole dependency graph (lmdb, the archive reader, the digest layer). A value @@ -87,39 +88,56 @@ async function hasDigestWithItems(videoDir: string): Promise<boolean> { ); } -async function countDataFiles(dataDir: string): Promise<{ +// `drive` is the channel's configured target (`config.dataDir`) when its media +// is on another drive. Then every read goes through `onDrive`: none while that +// location is stalled, at most four in flight on it, and one that has not +// answered in 3 s marks it stalled — and the walk answers null ("the drive did +// not answer") instead of counts. The rest of the walk is refused without a +// call. An in-place channel's walk is on the corpus disk and is not wrapped. +async function countDataFiles( + dataDir: string, + drive?: string, +): Promise<{ videos: number; transcripts: number; downloads: number; digests: number; -}> { +} | null> { + const through = <T>(call: () => Promise<T>): Promise<T> => + drive ? onDrive(drive, call) : call(); let dirs: Dirent[]; try { - dirs = await readdir(dataDir, { withFileTypes: true }); - } catch { + dirs = await through(() => readdir(dataDir, { withFileTypes: true })); + } catch (err) { + if (isDriveNotAnswering(err)) return null; return { videos: 0, transcripts: 0, downloads: 0, digests: 0 }; } const videoDirs = dirs.filter((d) => d.isDirectory()); // Bounded: the largest channel has 11,224 video dirs and this used to open // them all at once. - const flags = await mapConcurrent( - videoDirs, - VIDEO_READ_CONCURRENCY, - async (d) => { - const dir = path.join(dataDir, d.name); - const files = await readVideoFiles(dir); - return { - transcript: isVideoTranscribed(files), - download: isVideoDownloaded(files), - // Only transcribed videos can carry a digest, so the sidecar read is - // skipped for the rest — the same conditional per-video sidecar-read - // pattern channelSnapshot.ts uses for coverage and VTT provenance. - digest: isVideoTranscribed(files) - ? await hasDigestWithItems(dir) - : false, - }; - }, - ); + let flags: Array<{ transcript: boolean; download: boolean; digest: boolean }>; + try { + flags = await mapConcurrent(videoDirs, VIDEO_READ_CONCURRENCY, (d) => + // One video directory's few reads are one call through the watchdog. + through(async () => { + const dir = path.join(dataDir, d.name); + const files = await readVideoFiles(dir); + return { + transcript: isVideoTranscribed(files), + download: isVideoDownloaded(files), + // Only transcribed videos can carry a digest, so the sidecar read is + // skipped for the rest — the same conditional per-video sidecar-read + // pattern channelSnapshot.ts uses for coverage and VTT provenance. + digest: isVideoTranscribed(files) + ? await hasDigestWithItems(dir) + : false, + }; + }), + ); + } catch (err) { + if (isDriveNotAnswering(err)) return null; + throw err; + } let transcripts = 0; let downloads = 0; let digests = 0; @@ -189,8 +207,18 @@ export async function readChannelStat( ): Promise<ChannelStat | null> { const config = await readChannelConfig(paths, slug); if (!config) return null; + // A WALK OF `data/` ON A DRIVE THAT IS NOT ANSWERING IS NOT STARTED. The + // one-second job-list poll asks this for every channel with a job listed, and + // on a stalled drive each readdir and stat in the walk would hold an I/O + // thread until the drive came back. No counts is what a caller already + // handles (the row draws no progress bar). + if (channelMediaStall(config)) return null; const channelDir = path.join(paths.channelsDir, slug); - const counts = await countDataFiles(path.join(channelDir, "data")); + const counts = await countDataFiles( + path.join(channelDir, "data"), + config.dataDir?.trim() || undefined, + ); + if (!counts) return null; return { slug, config, @@ -265,7 +293,14 @@ export async function listChannelStatsFromDisk( if (!config) continue; const channelDir = path.join(paths.channelsDir, slug); const dataDir = path.join(channelDir, "data"); - const counts = await countDataFiles(dataDir); + // A batch job's ground truth: no drive passed, so nothing is raced and the + // answer is never null. + const counts = (await countDataFiles(dataDir)) ?? { + videos: 0, + transcripts: 0, + downloads: 0, + digests: 0, + }; out.push({ slug, config, diff --git a/common/controller/evictClipWindows.ts b/common/controller/evictClipWindows.ts @@ -96,7 +96,9 @@ async function evictChannel( // fine. (The JOB also declares `needsMedia: true`, which covers a // single-channel run before it starts; this covers the corpus-wide one, // where there is no slug for that guard to check.) - const media = await inspectChannelMedia(opts.paths, slug); + const media = await inspectChannelMedia(opts.paths, slug, undefined, { + fresh: true, + }); if (media.status !== "ok" && media.status !== "in-place") { out.skipped.push( `${slug}: media ${media.status}${media.detail ? ` (${media.detail})` : ""} — nothing was touched`, diff --git a/common/controller/keptVideos.ts b/common/controller/keptVideos.ts @@ -2,6 +2,10 @@ import path from "node:path"; import { readdir } from "node:fs/promises"; import type { Paths } from "../lib/paths"; import { loadRawMetadataFromDir } from "../lib/transcripts-server"; +import { mapConcurrent } from "../lib/concurrency"; + +// Metadata reads in flight while keying a channel's videos by upload date. +const KEY_READ_CONCURRENCY = 16; // Rolling "keep-latest" window computation. Given a channel's keepLatest config, // returns the ids of the newest N videos (by upload date). Used by both the @@ -13,6 +17,11 @@ export type ComputeKeptOptions = { paths: Paths; channelSlug: string; keepLatest: number; + // Runs each video's metadata read. The snapshot passes `onDrive` for a + // channel on another drive (lib/storageHealth.ts), which caps the reads in + // flight on that drive and gives up on one that does not answer. Default: + // the read itself. + through?: <T>(read: () => Promise<T>) => Promise<T>; }; // List a channel's data-dir video ids (directories, skipping dotfiles). Mirrors @@ -58,16 +67,21 @@ async function uploadKey(videoDir: string, id: string): Promise<string> { async function keyedVideosNewestFirst( paths: Paths, channelSlug: string, + through: <T>(read: () => Promise<T>) => Promise<T> = (read) => read(), ): Promise<Array<{ id: string; key: string }>> { - const ids = await listChannelVideoIds(paths, channelSlug); + const ids = await through(() => listChannelVideoIds(paths, channelSlug)); if (ids.length === 0) return []; const dataDir = path.join(paths.channelsDir, channelSlug, "data"); - const keyed = await Promise.all( - ids.map(async (id) => ({ - id, - key: await uploadKey(path.join(dataDir, id), id), - })), - ); + // BOUNDED, like every other corpus-shaped fan-out (lib/concurrency.ts): one + // metadata read per video, and the largest channel has eleven thousand. With + // a `through` of `onDrive`, at most four of these are on the drive at once + // and the rest wait in its queue, which refuses a waiting read only when + // nothing on the drive has returned for the watchdog's budget — never for + // the queue's depth alone. + const keyed = await mapConcurrent(ids, KEY_READ_CONCURRENCY, async (id) => ({ + id, + key: await through(() => uploadKey(path.join(dataDir, id), id)), + })); keyed.sort((a, b) => a.key === b.key ? b.id.localeCompare(a.id) : b.key.localeCompare(a.key), ); @@ -82,9 +96,10 @@ export async function computeKeptVideoIds({ paths, channelSlug, keepLatest, + through, }: ComputeKeptOptions): Promise<Set<string>> { if (!Number.isFinite(keepLatest) || keepLatest <= 0) return new Set(); - const keyed = await keyedVideosNewestFirst(paths, channelSlug); + const keyed = await keyedVideosNewestFirst(paths, channelSlug, through); return new Set(keyed.slice(0, Math.floor(keepLatest)).map((k) => k.id)); } diff --git a/common/controller/recencyIndex.ts b/common/controller/recencyIndex.ts @@ -7,6 +7,12 @@ import { mapConcurrent } from "../lib/concurrency"; import { extractVideoId } from "../lib/videoId"; import { uploadKeyFor } from "./keptVideos"; import type { AutoQueueOrder } from "../jobs/autoQueuePolicy"; +import type { ChannelConfig } from "../lib/channelConfig"; +import { + isDriveNotAnswering, + onDrive, + stalledLocationForPath, +} from "../lib/storageHealth"; // Upload-date lookup for the auto-queue's "newest first" ordering. // @@ -189,11 +195,22 @@ async function readTailUploadDate(file: string): Promise<string | null> { // Date the ids in `wanted` from their on-disk metadata. Removes each id it // keys from `wanted`, like interpolateFromPlaylist. +// +// `drives` maps a channel whose media is on another drive to its configured +// target. Its reads go through `onDrive`: none while that drive's location is +// stalled (lib/storageHealth.ts) — each would hold an I/O thread until the drive +// came back, 32 at a time — at most four in flight on it, and one that has not +// answered in 3 s marks it stalled. An id not read for that reason is NOT +// memoized as a miss: it falls through to layers 3 and 4 for now, and a later +// refresh with the drive answering reads it. +const NOT_READ = Symbol("not read: the drive is not answering"); + async function datesFromMetadata( paths: Paths, owner: ReadonlyMap<string, string>, wanted: Set<string>, out: Map<string, RecencyKey>, + drives: ReadonlyMap<string, string> = new Map(), ): Promise<void> { const todo: string[] = []; for (const id of wanted) { @@ -205,7 +222,10 @@ async function datesFromMetadata( } continue; } - if (!owner.has(id)) continue; + const slug = owner.get(id); + if (slug === undefined) continue; + const drive = drives.get(slug); + if (drive && stalledLocationForPath(drive)) continue; todo.push(id); if (todo.length >= TAIL_READS_PER_BUILD) { if (!tailCapLogged) { @@ -219,19 +239,31 @@ async function datesFromMetadata( } } if (todo.length === 0) return; - const dates = await mapConcurrent(todo, TAIL_READ_CONCURRENCY, (id) => - readTailUploadDate( - path.join( + const dates = await mapConcurrent( + todo, + TAIL_READ_CONCURRENCY, + async (id): Promise<string | null | typeof NOT_READ> => { + const slug = owner.get(id) as string; + const file = path.join( paths.channelsDir, - owner.get(id) as string, + slug, "data", id, "metadata.info.json", - ), - ), + ); + const drive = drives.get(slug); + if (!drive) return readTailUploadDate(file); + try { + return await onDrive(drive, () => readTailUploadDate(file)); + } catch (err) { + if (isDriveNotAnswering(err)) return NOT_READ; + throw err; + } + }, ); for (const [i, id] of todo.entries()) { const date = dates[i]; + if (date === NOT_READ) continue; if (tailMemo.size < TAIL_MEMO_CAP) tailMemo.set(id, date); if (!date) continue; out.set(id, { key: date, estimated: false }); @@ -312,7 +344,12 @@ export function interpolateFromPlaylist( export type BuildRecencyKeysArgs = { paths: Paths; // The channels whose work is in play — the runner's own channel-meta list. - meta: ReadonlyArray<{ slug: string }>; + // The config, when the caller holds it, is what tells layer 2 which channels + // are on another drive (their reads go through the stall watchdog). + meta: ReadonlyArray<{ + slug: string; + config?: Pick<ChannelConfig, "dataDir"> | null; + }>; // Every video id the caller might sort. Ids outside this set are not keyed. candidateIds: ReadonlySet<string>; // videoId -> owning channel slug. Required for the tail-read layer, which has @@ -468,7 +505,12 @@ export async function buildRecencyKeys({ // auto-transcribe, whose entire candidate set is by definition absent from the // transcript index. if (owner && missing.size > 0) { - await datesFromMetadata(paths, owner, missing, out); + const drives = new Map<string, string>(); + for (const m of meta) { + const dir = m.config?.dataDir?.trim(); + if (dir) drives.set(m.slug, dir); + } + await datesFromMetadata(paths, owner, missing, out, drives); } // Layer 3: playlist interpolation, for videos with nothing on disk at all. diff --git a/common/controller/relocateChannelMedia.ts b/common/controller/relocateChannelMedia.ts @@ -38,9 +38,17 @@ import { type VolumeBins, } from "../lib/storageVolumes"; import { getFreeBytes } from "../lib/diskSpace"; +import { + NOT_ANSWERING, + isDriveNotAnswering, + onDrive, + sinceText, + stalledLocation, +} from "../lib/storageHealth"; import { getSettings, type SiteSettings } from "../lib/settings"; import { formatBytes } from "../lib/format"; import { + forgetChannelMedia, inspectChannelMedia, relocatedDataDir, relocationMarkerPath, @@ -282,10 +290,29 @@ export async function relocationRootPresenceProblem( const named = locationForRoot(r, storage.locations); const where = named ? ` (location "${named.id}")` : ""; + // A destination whose drive is not answering is refused WITHOUT the stat + // below: that stat would wait on the drive, and a move onto it would too. + const stall = named ? stalledLocation(named) : null; + if (stall) { + return ( + `The destination root ${r}${where}: ${NOT_ANSWERING} ` + + `(${sinceText(stall.since)}). Wait for it to answer, or check the drive.` + ); + } + + // Through the watchdog: a stat that has not answered in 3 s marks the + // location stalled and refuses the same way. let isDir = false; try { - isDir = (await stat(r)).isDirectory(); - } catch { + isDir = (await onDrive(named ?? r, () => stat(r))).isDirectory(); + } catch (err) { + if (isDriveNotAnswering(err)) { + return ( + `The destination root ${r}${where}: ${NOT_ANSWERING} ` + + `(${err.health ? sinceText(err.health.since) : "a stat of it did not answer"}). ` + + `Wait for it to answer, or check the drive.` + ); + } isDir = false; } if (!isDir) { @@ -373,16 +400,23 @@ async function sweepParked( // The channel's marker, at `channels/<slug>/.relocating.json`. The FILE is the // contract — see relocateDir.ts — and these three are the channel's name for it. +// +// Each write and the clear also drop the channel from `inspectChannelMedia`'s +// five-second page memo. Every phase change (copy → swap → reclaim) writes the +// marker AFTER the link and the config it changes, so a page asks the disk +// again the moment the move has done something it would see. async function writeMarker( paths: Paths, slug: string, marker: RelocationMarker, ): Promise<void> { await writeDirMarker(relocationMarkerPath(paths, slug), marker); + forgetChannelMedia(slug); } async function clearMarker(paths: Paths, slug: string): Promise<void> { await clearDirMarker(relocationMarkerPath(paths, slug)); + forgetChannelMedia(slug); } async function readMarkerRaw( @@ -601,7 +635,9 @@ async function moveOut(args: { // every guard and exactly wrong here: the rerun that finishes an interrupted // move is the one caller allowed to see it. A marker for a DIFFERENT target // was already refused above. - const location = await inspectChannelMedia(paths, slug, args.config); + const location = await inspectChannelMedia(paths, slug, args.config, { + fresh: true, + }); if (!args.resumed && location.status !== "in-place") { throw new Error( `Channel "${slug}" is not in a movable state: ${ @@ -838,7 +874,9 @@ async function moveBack(args: { // // `in-transition` is allowed because a marker is what a resume carries, and a // rerun is the caller this precondition must not refuse. - const location = await inspectChannelMedia(paths, slug, args.config); + const location = await inspectChannelMedia(paths, slug, args.config, { + fresh: true, + }); if ( !args.resumed && location.status !== "ok" && diff --git a/common/controller/relocateSavedVideos.ts b/common/controller/relocateSavedVideos.ts @@ -20,6 +20,13 @@ import { } from "../lib/settings"; import type { RelocationMarker, RelocationPhase } from "../lib/channelMedia"; import { + NOT_ANSWERING, + isDriveNotAnswering, + onDrive, + sinceText, + stalledLocation, +} from "../lib/storageHealth"; +import { relocatedSavedVideosDir, savedVideosMarkerPath, } from "../lib/savedVideoStore"; @@ -156,7 +163,33 @@ export async function inspectSavedVideosStore( detail: `the store links to ${target}, but no storage location is recorded for it`, }; } - if (!(await isDirectory(target))) { + // The store's drive is not answering: reported without the stat below, + // which would wait on it (the /storage page asks on every render). The + // page skips the store's size walk for an unreachable store, too. + const stall = loc ? stalledLocation(loc) : null; + if (stall) { + return { + dir, + locationId, + target, + status: "unreachable", + detail: `${NOT_ANSWERING} (location "${stall.label}", ${sinceText(stall.since)})`, + }; + } + let reachable: boolean; + try { + reachable = await onDrive(loc ?? target, () => isDirectory(target)); + } catch (err) { + if (!isDriveNotAnswering(err)) throw err; + return { + dir, + locationId, + target, + status: "unreachable", + detail: err.message, + }; + } + if (!reachable) { return { dir, locationId, diff --git a/common/controller/savedVideoInventory.ts b/common/controller/savedVideoInventory.ts @@ -4,8 +4,13 @@ import type { Paths } from "../lib/paths"; import { mapConcurrent } from "../lib/concurrency"; import { loadSavedVideo } from "../lib/savedVideo-server"; import { savedVideoPath, type SavedVideoPointer } from "../lib/savedVideo"; +import { + isDriveNotAnswering, + onDrive, + stalledLocationForPath, +} from "../lib/storageHealth"; -const { pathExists, readdir } = fs; +const { pathExists, readdir, readlink } = fs; // Channels are read a few at a time so the per-video fan-out inside each one // still dominates; the product is the real ceiling on open descriptors. @@ -37,12 +42,23 @@ async function listChannelSlugs(paths: Paths): Promise<string[]> { // channel when channelSlug is given), by following the saved-video.json pointers // in each data dir. The pointer's `dir` is absolute, so per-channel store // overrides resolve correctly without consulting channel config here. +// +// `notAnswering`, for a PAGE: pass an array and every channel whose `data/` +// links onto a storage location whose drive is not answering +// (lib/storageHealth.ts) is skipped — its pointers are one read per video dir, +// each of which would wait on the drive — and its slug is pushed there, so the +// page can say which channels it did not read. The link is read, not followed: +// it is on the corpus disk. A relocated channel's reads then go through +// `onDrive`'s watchdog, so a drive that stops answering mid-list is skipped and +// named the same way. Omitted (the backup job), nothing is skipped or raced. export async function listSavedVideos({ paths, channelSlug, + notAnswering, }: { paths: Paths; channelSlug?: string; + notAnswering?: string[]; }): Promise<SavedVideoEntry[]> { const slugs = channelSlug ? [channelSlug] : await listChannelSlugs(paths); // One pointer read per video dir — 78,350 of them across the corpus. Done one @@ -53,25 +69,43 @@ export async function listSavedVideos({ CHANNEL_CONCURRENCY, async (slug): Promise<SavedVideoEntry[]> => { const dataDir = path.join(paths.channelsDir, slug, "data"); - if (!(await pathExists(dataDir))) return []; - const ids = await readdir(dataDir).catch(() => [] as string[]); - const entries = await mapConcurrent( - ids, - VIDEO_CONCURRENCY, - async (videoId): Promise<SavedVideoEntry | null> => { - const videoDir = path.join(dataDir, videoId); - const pointer = await loadSavedVideo(videoDir); - if (!pointer) return null; - return { - slug, - videoId, - videoDir, - storedPath: savedVideoPath(pointer), - pointer, - }; - }, - ); - return entries.filter((e): e is SavedVideoEntry => e !== null); + let drive = ""; + if (notAnswering) { + drive = await readlink(dataDir).catch(() => ""); + if (drive && stalledLocationForPath(drive)) { + notAnswering.push(slug); + return []; + } + } + const through = <T>(call: () => Promise<T>): Promise<T> => + drive ? onDrive(drive, call) : call(); + try { + if (!(await through(() => pathExists(dataDir)))) return []; + const ids = await through(() => + readdir(dataDir).catch(() => [] as string[]), + ); + const entries = await mapConcurrent( + ids, + VIDEO_CONCURRENCY, + async (videoId): Promise<SavedVideoEntry | null> => { + const videoDir = path.join(dataDir, videoId); + const pointer = await through(() => loadSavedVideo(videoDir)); + if (!pointer) return null; + return { + slug, + videoId, + videoDir, + storedPath: savedVideoPath(pointer), + pointer, + }; + }, + ); + return entries.filter((e): e is SavedVideoEntry => e !== null); + } catch (err) { + if (!notAnswering || !isDriveNotAnswering(err)) throw err; + notAnswering.push(slug); + return []; + } }, ); const out = perChannel.flat(); @@ -91,6 +125,7 @@ export type SavedVideoTotals = { export async function savedVideoTotals(opts: { paths: Paths; channelSlug?: string; + notAnswering?: string[]; }): Promise<SavedVideoTotals> { const entries = await listSavedVideos(opts); let bytes = 0; diff --git a/common/controller/storageLocations.ts b/common/controller/storageLocations.ts @@ -24,7 +24,16 @@ import { type StorageLocationProbe, type VolumeBins, } from "../lib/storageVolumes"; -import { inspectChannelMedia, relocatedDataDir } from "../lib/channelMedia"; +import { + forgetChannelMedia, + inspectChannelMedia, + relocatedDataDir, +} from "../lib/channelMedia"; +import { + isDriveNotAnswering, + onDrive, + stalledLocation, +} from "../lib/storageHealth"; import { relocationQueueKey } from "../lib/queueKeys"; import { runManagedFunction, @@ -247,61 +256,85 @@ export async function volumeFreeBytes(opts: { out[INTERNAL_LOCATION_ID] = Number.isFinite(corpus) ? corpus : undefined; await Promise.all( opts.locations.map(async (loc) => { - const st = await stat(loc.root).catch(() => null); - if (!st?.isDirectory()) { + // NOT ASKED WHILE ITS DRIVE IS NOT ANSWERING: each stat and the statfs + // below would hold an I/O thread until it did. "—", like unmounted. The + // calls that reach the drive go through `onDrive`'s watchdog; one that + // has not answered in 3 s marks the location stalled, and reads "—" too. + if (stalledLocation(loc)) { out[loc.id] = undefined; return; } - // THE DIRECTORY EXISTING IS NOT THE DRIVE BEING THERE, and this is the - // half the stat above could not catch. An unmounted mountpoint is a real, - // empty directory ON ITS PARENT'S FILESYSTEM — so `getFreeBytes` succeeds - // and reports the parent volume's free space, which on this machine is - // the disk the operator is trying to empty. The location would then read - // "233 GB free" about a platter that is not plugged in. - // - // THE BOUNDARY IS TESTED AT THE MOUNTPOINT, NEVER AT THE ROOT, and the - // first version of this got that wrong in the one way that matters on - // this machine. A location's root is `join(mountpoint, relPath)` — the - // production `platter` is `/run/media/<user>/<uuid>/archilyzer-media`, a - // SUBDIRECTORY of the mountpoint — so the root and its parent are on the - // same filesystem BY CONSTRUCTION whenever `relPath` is non-empty, and - // comparing those two devices reported "free space unknown" for a - // correctly mounted drive. A bind mount reads the same way. - // - // Crossing a mount changes the device number, so `stat(mountpoint).dev` - // against `stat(dirname(mountpoint)).dev` is the honest question, and it - // is two syscalls — TABLES NEVER PROBE (see the header) rules out asking - // `probeLocation`, which is up to three subprocesses. - // - // Only asked of a location that has learned a `volume.uuid`: that field - // is the assertion that the root is supposed to be on its own volume. A - // location on a plain directory (never probed, a container, a - // subdirectory of the system disk by design) shares its parent's device - // legitimately, and withholding its free space would be wrong. - const mountpoint = loc.volume?.uuid - ? (loc.volume.mountpoint ?? "").trim() - : ""; - // `/` is its own parent, so a volume mounted at the root has no boundary - // to test and is trivially there — the process is reading from it. - if (mountpoint && mountpoint !== path.dirname(mountpoint)) { - const [atMount, aboveMount] = await Promise.all([ - stat(mountpoint).catch(() => null), - stat(path.dirname(mountpoint)).catch(() => null), - ]); - // Gone entirely, or present as an ordinary directory on the parent - // filesystem: either way nothing is mounted there. - if (!atMount || (aboveMount && aboveMount.dev === atMount.dev)) { - out[loc.id] = undefined; - return; - } + try { + out[loc.id] = await freeOnLocation(loc); + } catch (err) { + if (!isDriveNotAnswering(err)) throw err; + out[loc.id] = undefined; } - const free = await getFreeBytes(loc.root); - out[loc.id] = Number.isFinite(free) ? free : undefined; }), ); return out; } +// One location's free space, for volumeFreeBytes. Throws DriveNotAnsweringError +// when its drive does not answer. +async function freeOnLocation( + loc: StorageLocation, +): Promise<number | undefined> { + const orNull = <T>(p: Promise<T>): Promise<T | null> => + p.catch((err) => { + if (isDriveNotAnswering(err)) throw err; + return null; + }); + const st = await orNull(onDrive(loc, () => stat(loc.root))); + if (!st?.isDirectory()) return undefined; + // THE DIRECTORY EXISTING IS NOT THE DRIVE BEING THERE, and this is the + // half the stat above could not catch. An unmounted mountpoint is a real, + // empty directory ON ITS PARENT'S FILESYSTEM — so `getFreeBytes` succeeds + // and reports the parent volume's free space, which on this machine is + // the disk the operator is trying to empty. The location would then read + // "233 GB free" about a platter that is not plugged in. + // + // THE BOUNDARY IS TESTED AT THE MOUNTPOINT, NEVER AT THE ROOT, and the + // first version of this got that wrong in the one way that matters on + // this machine. A location's root is `join(mountpoint, relPath)` — the + // production `platter` is `/run/media/<user>/<uuid>/archilyzer-media`, a + // SUBDIRECTORY of the mountpoint — so the root and its parent are on the + // same filesystem BY CONSTRUCTION whenever `relPath` is non-empty, and + // comparing those two devices reported "free space unknown" for a + // correctly mounted drive. A bind mount reads the same way. + // + // Crossing a mount changes the device number, so `stat(mountpoint).dev` + // against `stat(dirname(mountpoint)).dev` is the honest question, and it + // is two syscalls — TABLES NEVER PROBE (see the header) rules out asking + // `probeLocation`, which is up to three subprocesses. + // + // Only asked of a location that has learned a `volume.uuid`: that field + // is the assertion that the root is supposed to be on its own volume. A + // location on a plain directory (never probed, a container, a + // subdirectory of the system disk by design) shares its parent's device + // legitimately, and withholding its free space would be wrong. + const mountpoint = loc.volume?.uuid + ? (loc.volume.mountpoint ?? "").trim() + : ""; + // `/` is its own parent, so a volume mounted at the root has no boundary + // to test and is trivially there — the process is reading from it. + if (mountpoint && mountpoint !== path.dirname(mountpoint)) { + // The mountpoint is the drive's own top directory; its parent is on the + // filesystem above it, so only the first goes through the watchdog. + const [atMount, aboveMount] = await Promise.all([ + orNull(onDrive(loc, () => stat(mountpoint))), + stat(path.dirname(mountpoint)).catch(() => null), + ]); + // Gone entirely, or present as an ordinary directory on the parent + // filesystem: either way nothing is mounted there. + if (!atMount || (aboveMount && aboveMount.dev === atMount.dev)) { + return undefined; + } + } + const free = await onDrive(loc, () => getFreeBytes(loc.root)); + return Number.isFinite(free) ? free : undefined; +} + // --------------------------------------------------------------------------- // The probe memo // --------------------------------------------------------------------------- @@ -528,7 +561,9 @@ export async function preflightRepoint(opts: { const missing: string[] = []; for (const slug of slugs) { const config = await readChannelConfig(opts.paths, slug); - const media = await inspectChannelMedia(opts.paths, slug, config); + const media = await inspectChannelMedia(opts.paths, slug, config, { + fresh: true, + }); if (media.status === "in-transition") { base.problems.push( `${slug}: a media relocation is in flight or was interrupted ` + @@ -795,6 +830,10 @@ export async function repointStorageLocation(opts: { } resetStorageProbeMemo(); + // Every channel on it has a new link and a new dataDir: the page memo's keys + // already differ, and this drops the old answers rather than letting them + // age out. + forgetChannelMedia(); log( `Done. "${loc.label}" is at ${pre.newRoot}; ${pre.channels.length} ` + `channel(s) re-pointed.`, diff --git a/common/controller/storageStall.test.ts b/common/controller/storageStall.test.ts @@ -0,0 +1,544 @@ +// A STALLED DRIVE IS NOT ASKED: every page-and-poll path that would touch a +// storage location's drive in-process asks the health state first +// (lib/storageHealth.ts) and, on a stalled location, answers WITHOUT the call. +// +// No test stalls a real drive. The drive here is an ordinary temp directory +// that answers every call at once; the health state is TOLD it is stalled. +// Every node:fs and node:fs/promises call this file's code makes is recorded +// (the buildIndex.test.ts spy, reads included), so a gate that let one call +// through shows up as a recorded path under the drive — it cannot hide behind +// a call that happened to answer quickly. +// +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/storageStall.test.ts + +import { after, beforeEach, test } from "node:test"; +import assert from "node:assert/strict"; +import { createRequire, syncBuiltinESMExports } from "node:module"; +import { + existsSync, + mkdirSync, + mkdtempSync, + rmSync, + symlinkSync, + writeFileSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import type { Paths } from "../lib/paths"; +import type { StorageLocation } from "../lib/storageLocations"; +import { + assertChannelMediaReachable, + ChannelMediaUnreachableError, + CHANNEL_MEDIA_MEMO_MS, + clearRelocationMarker, + forgetChannelMedia, + inspectChannelMedia, +} from "../lib/channelMedia"; +import { HELD_REASON, isMediaHeld } from "../lib/channelMediaHold"; +import { + locationHealth, + recordLocationHealth, + registerLocationHealth, + resetStorageHealth, + setDriveCallBudget, +} from "../lib/storageHealth"; +import { + probeLocation, + probeLocationMemo, + resetStorageProbeMemo, +} from "../lib/storageVolumes"; +import { volumeFreeBytes } from "./storageLocations"; +import { readChannelStat } from "./channels"; +import { buildRecencyKeys, clearRecencyCache } from "./recencyIndex"; +import { relocationRootPresenceProblem } from "./relocateChannelMedia"; +import { generateChannelSnapshot } from "./channelSnapshot"; +import { inspectSavedVideosStore } from "./relocateSavedVideos"; +import { listSavedVideos } from "./savedVideoInventory"; +import type { SiteSettings } from "../lib/settings"; + +// ── the spy ──────────────────────────────────────────────────────────────── +type Call = { fn: string; path: string }; +let calls: Call[] = []; +// THE HANG: a promise-API call whose (function, path) matches never settles — +// what a read blocked on a stalled drive looks like from here. Recorded like +// any other call. No drive is involved. +let hang: ((c: Call) => boolean) | null = null; +{ + const req = createRequire(import.meta.url); + const fsCjs = req("node:fs") as Record<string, unknown>; + const fspCjs = req("node:fs/promises") as Record<string, unknown>; + const NAMES = [ + "access", "appendFile", "chmod", "copyFile", "cp", "lstat", "mkdir", + "open", "opendir", "readdir", "readFile", "readlink", "realpath", "rename", + "rm", "rmdir", "stat", "statfs", "symlink", "unlink", "utimes", "writeFile", + ]; + const asPath = (v: unknown) => + typeof v === "string" + ? v + : v instanceof URL + ? fileURLToPath(v) + : Buffer.isBuffer(v) + ? v.toString() + : null; + const wrap = (mod: Record<string, unknown>, name: string, promises = false) => { + const fn = mod[name]; + if (typeof fn !== "function") return; + mod[name] = function (this: unknown, ...args: unknown[]) { + const p = asPath(args[0]); + if (p !== null) { + const call = { fn: name, path: path.resolve(p) }; + calls.push(call); + if (promises && hang?.(call)) return new Promise(() => {}); + } + return (fn as (...a: unknown[]) => unknown).apply(this, args); + }; + }; + for (const n of NAMES) { + wrap(fspCjs, n, true); + wrap(fsCjs, n); + wrap(fsCjs, `${n}Sync`); + } + wrap(fsCjs, "existsSync"); + wrap(fsCjs, "createReadStream"); + syncBuiltinESMExports(); +} + +// ── the corpus and the drive ─────────────────────────────────────────────── +const ROOT = mkdtempSync(path.join(tmpdir(), "ttb-stall-")); +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const CORPUS = path.join(ROOT, "corpus"); +const DRIVE = path.join(ROOT, "drive"); +const SLUG = "on-drive"; +const paths = { + transcriptsDir: CORPUS, + channelsDir: path.join(CORPUS, "channels"), + lmdbPath: path.join(CORPUS, "index.mdb"), + savedVideosDir: path.join(CORPUS, "saved-videos"), + jobsDir: path.join(CORPUS, "jobs"), + // A findmnt that leaves a mark if anything runs it. + findmntBin: path.join(ROOT, "findmnt-ran.sh"), + udisksctlBin: path.join(ROOT, "no-udisksctl"), +} as unknown as Paths; +const LOC: StorageLocation = { + id: "usb", + label: "USB drive", + root: DRIVE, + autoRepoint: false, +}; +const TARGET = path.join(DRIVE, SLUG, "data"); +const LINK = path.join(paths.channelsDir, SLUG, "data"); +const CONFIG = { dataDir: TARGET }; +const FINDMNT_MARK = path.join(ROOT, "findmnt-ran"); + +function seed(): void { + rmSync(CORPUS, { recursive: true, force: true }); + rmSync(DRIVE, { recursive: true, force: true }); + mkdirSync(path.join(TARGET, "vid1"), { recursive: true }); + writeFileSync( + path.join(TARGET, "vid1", "metadata.info.json"), + JSON.stringify({ id: "vid1", upload_date: "20260601" }), + ); + writeFileSync(path.join(TARGET, "vid1", "transcript.en.vtt"), "WEBVTT\n"); + mkdirSync(path.join(paths.channelsDir, SLUG), { recursive: true }); + writeFileSync( + path.join(paths.channelsDir, SLUG, "config.json"), + JSON.stringify({ + handling: "youtube", + name: SLUG, + url: `https://www.youtube.com/@${SLUG}/videos`, + dataDir: TARGET, + }), + ); + symlinkSync(TARGET, LINK); + writeFileSync(paths.findmntBin, `#!/bin/sh\ntouch ${FINDMNT_MARK}\nexit 1\n`, { + mode: 0o755, + }); + rmSync(FINDMNT_MARK, { force: true }); +} + +// Anything that reaches the drive: a path under its root, or through the +// channel's link (`data/` itself, followed, or anything below it). +function onDrive(c: Call): boolean { + const under = (p: string, base: string) => + p === base || p.startsWith(base + path.sep); + return under(c.path, DRIVE) || under(c.path, LINK); +} + +function stall(): void { + recordLocationHealth(LOC, "stalled", { now: Date.now() }); +} + +beforeEach(() => { + hang = null; + setDriveCallBudget(); + resetStorageHealth(); + forgetChannelMedia(); + resetStorageProbeMemo(); + clearRecencyCache(); + seed(); + calls = []; +}); + +// ── inspectChannelMedia: the gate and the memo ───────────────────────────── + +const MARKER = path.join(paths.channelsDir, SLUG, ".relocating.json"); + +test("inspect with the config in hand: on a stalled drive only the marker is read, on the corpus disk", async () => { + stall(); + calls = []; + const media = await inspectChannelMedia(paths, SLUG, CONFIG); + assert.deepEqual(calls, [{ fn: "readFile", path: MARKER }]); + assert.deepEqual(calls.filter(onDrive), []); + assert.equal(media.status, "stalled"); + assert.equal(media.target, TARGET); + assert.match(String(media.detail), /^drive not answering \(location "USB drive", since /); + // Held by both pool-wide builds, with a reason that names no path. + assert.equal(isMediaHeld(media.status), true); + assert.doesNotMatch(HELD_REASON.stalled, /\//); +}); + +test("inspect without the config reads config.json and nothing on the drive", async () => { + stall(); + calls = []; + const media = await inspectChannelMedia(paths, SLUG); + assert.equal(media.status, "stalled"); + assert.deepEqual( + calls.map((c) => [c.fn, path.relative(ROOT, c.path)]), + [ + ["readFile", path.join("corpus", "channels", SLUG, "config.json")], + ["readFile", path.join("corpus", "channels", SLUG, ".relocating.json")], + ], + ); +}); + +test("the marker first: a channel mid-move on a stalled drive reads in-transition", async () => { + writeFileSync( + MARKER, + JSON.stringify({ target: TARGET, direction: "back", startedAt: "", phase: "copy" }), + ); + stall(); + calls = []; + const media = await inspectChannelMedia(paths, SLUG, CONFIG); + assert.equal(media.status, "in-transition"); + assert.deepEqual(calls.filter(onDrive), []); + // Remembered as a move: a second look within five seconds gives the move + // back, not the stall. + const again = await inspectChannelMedia(paths, SLUG, CONFIG); + assert.equal(again.status, "in-transition"); +}); + +test("the start-of-work guard refuses a stalled channel without a call", async () => { + stall(); + calls = []; + await assert.rejects( + () => assertChannelMediaReachable(paths, SLUG, CONFIG), + (err: unknown) => + err instanceof ChannelMediaUnreachableError && + err.status === "stalled" && + /drive not answering/.test(err.message), + ); + assert.deepEqual(calls.filter(onDrive), []); +}); + +test("the memo: two inspects within 5 s stat the drive once; fresh and age each ask again", async () => { + const t0 = 1_000_000; + const targetStats = () => + calls.filter((c) => c.fn === "stat" && c.path === TARGET).length; + const a = await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 }); + assert.equal(a.status, "ok"); + assert.equal(targetStats(), 1); + const b = await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 + 4_999 }); + assert.equal(b.status, "ok"); + assert.equal(targetStats(), 1, "a second inspect inside five seconds is remembered"); + await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 + 4_999, fresh: true }); + assert.equal(targetStats(), 2, "fresh bypasses the memo"); + await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 + CHANNEL_MEDIA_MEMO_MS }); + assert.equal(targetStats(), 3, "an answer five seconds old is asked again"); + // A fresh answer is not remembered: the memo still holds the one from t0+5000. + await inspectChannelMedia(paths, SLUG, CONFIG, { now: t0 + CHANNEL_MEDIA_MEMO_MS + 1 }); + assert.equal(targetStats(), 3); +}); + +test("the memo is keyed by the configured target, and a mover's forget clears it", async () => { + const now = 2_000_000; + const targetStats = () => + calls.filter((c) => c.fn === "stat" && c.path === TARGET).length; + await inspectChannelMedia(paths, SLUG, CONFIG, { now }); + // Another configured target is another key. + const other = await inspectChannelMedia(paths, SLUG, { dataDir: path.join(DRIVE, "x", "data") }, { now }); + assert.equal(other.status, "inconsistent"); + forgetChannelMedia(SLUG); + await inspectChannelMedia(paths, SLUG, CONFIG, { now }); + assert.equal(targetStats(), 2); + // clearRelocationMarker (the operator's last resort) forgets too. + await clearRelocationMarker(paths, SLUG); + await inspectChannelMedia(paths, SLUG, CONFIG, { now }); + assert.equal(targetStats(), 3); +}); + +test("the gate is asked before the memo: a stall is seen with an ok remembered", async () => { + const now = 3_000_000; + assert.equal((await inspectChannelMedia(paths, SLUG, CONFIG, { now })).status, "ok"); + stall(); + calls = []; + const media = await inspectChannelMedia(paths, SLUG, CONFIG, { now: now + 1 }); + assert.equal(media.status, "stalled"); + assert.deepEqual(calls, []); +}); + +// ── the other gated callers ──────────────────────────────────────────────── + +test("probeLocation and its memo: 'stalled', with no stat, no statfs and no findmnt", async () => { + // A remembered "available" first, so the memo's gate is the thing tested. + await probeLocationMemo(LOC, paths); + rmSync(FINDMNT_MARK, { force: true }); + stall(); + calls = []; + const direct = await probeLocation(LOC, paths); + const memo = await probeLocationMemo(LOC, paths); + assert.equal(direct.status, "stalled"); + assert.equal(memo.status, "stalled"); + assert.deepEqual(direct.identity, { known: false }); + assert.equal(direct.freeBytes, undefined); + assert.deepEqual(calls.filter(onDrive), []); + assert.equal(existsSync(FINDMNT_MARK), false, "findmnt was not run"); +}); + +test("volumeFreeBytes: a stalled location reads unknown, with no call on it", async () => { + stall(); + calls = []; + const out = await volumeFreeBytes({ paths, locations: [LOC] }); + assert.equal(out.usb, undefined); + assert.equal(typeof out.internal, "number"); + assert.deepEqual(calls.filter(onDrive), []); +}); + +test("readChannelStat: no walk of data/ on a stalled drive", async () => { + const before = await readChannelStat(paths, SLUG); + assert.equal(before?.videoCount, 1); + stall(); + calls = []; + assert.equal(await readChannelStat(paths, SLUG), null); + assert.deepEqual(calls.filter(onDrive), []); +}); + +test("recency: no tail read on a stalled drive, and no miss remembered for it", async () => { + const owner = new Map([["vid1", SLUG]]); + const meta = [{ slug: SLUG, config: CONFIG }]; + const args = { + paths, + meta, + candidateIds: new Set(["vid1"]), + owner, + interpolate: false, + fresh: true, + }; + stall(); + calls = []; + const stalledKeys = await buildRecencyKeys(args); + assert.deepEqual(calls.filter(onDrive), []); + // Layer 4: an undatable id sorts oldest. + assert.deepEqual(stalledKeys.get("vid1"), { key: "", estimated: false }); + // The drive answers again: the same id is read now, because the stall was + // not remembered as a miss. + resetStorageHealth(); + const keys = await buildRecencyKeys(args); + assert.deepEqual(keys.get("vid1"), { key: "20260601", estimated: false }); + assert.ok(calls.some((c) => c.fn === "open" && onDrive(c))); +}); + +test("a move onto a stalled location is refused without a stat of its root", async () => { + stall(); + calls = []; + const problem = await relocationRootPresenceProblem( + DRIVE, + { locations: [LOC], defaultLocationId: "" }, + paths, + ); + assert.match(String(problem), /drive not answering/); + assert.match(String(problem), /location "usb"/); + assert.deepEqual(calls.filter(onDrive), []); +}); + +test("a snapshot refresh of a stalled channel throws before its walk", async () => { + stall(); + calls = []; + await assert.rejects( + () => generateChannelSnapshot(paths, SLUG), + (err: unknown) => + err instanceof ChannelMediaUnreachableError && err.status === "stalled", + ); + assert.deepEqual(calls.filter(onDrive), []); +}); + +test("the saved-video store on a stalled drive reads unreachable, without a stat through its link", async () => { + const storeTarget = path.join(DRIVE, "saved-videos"); + mkdirSync(storeTarget, { recursive: true }); + symlinkSync(storeTarget, paths.savedVideosDir); + const settings = { + storage: { locations: [LOC], defaultLocationId: "", savedVideosLocationId: "usb" }, + } as unknown as SiteSettings; + assert.equal((await inspectSavedVideosStore(paths, settings)).status, "ok"); + stall(); + calls = []; + const store = await inspectSavedVideosStore(paths, settings); + assert.equal(store.status, "unreachable"); + assert.match(String(store.detail), /^drive not answering \(location "USB drive"/); + const followed = calls.filter( + (c) => + onDrive(c) || + (c.path.startsWith(paths.savedVideosDir) && !["lstat", "readlink"].includes(c.fn)), + ); + assert.deepEqual(followed, []); +}); + +test("the saved-video inventory, for a page: a stalled channel is named, not read", async () => { + // A saved video on the drive, so a channel that WAS read has an entry. + // (savedVideoInventory reads through fs-extra, whose functions graceful-fs + // captured before this file's spy was installed, so this case proves the skip + // by what comes back, not by the spy.) + writeFileSync( + path.join(TARGET, "vid1", "saved-video.json"), + JSON.stringify({ storedAt: "", dir: "/store", file: "v.mp4", bytes: 7 }), + ); + assert.equal((await listSavedVideos({ paths, notAnswering: [] })).length, 1); + stall(); + const notAnswering: string[] = []; + assert.deepEqual(await listSavedVideos({ paths, notAnswering }), []); + assert.deepEqual(notAnswering, [SLUG]); + // Without the array (the backup job) nothing is skipped: it reads the drive. + assert.equal((await listSavedVideos({ paths })).length, 1); +}); + +// ── the watchdog: a call that does not answer in the budget ──────────────── +// The location is registered (the health pass does that in the editor) but +// answering; the hang makes one call on the drive never settle. The budget is +// shortened to 100 ms; lib/storageHealth.test.ts holds the 3 s default. + +function hangOnDrive(fns: string[]): void { + hang = (c) => fns.includes(c.fn) && onDrive(c); +} + +async function watchdogCase( + fns: string[], + run: () => Promise<unknown>, +): Promise<unknown> { + registerLocationHealth([LOC]); + setDriveCallBudget(100); + hangOnDrive(fns); + calls = []; + const started = Date.now(); + const out = await run(); + assert.ok(Date.now() - started < 2_000, "answered on the watchdog, not on the drive"); + assert.equal(locationHealth("usb")?.state, "stalled", "the location is marked at once"); + assert.match(String(locationHealth("usb")?.cause), /a read in the editor did not answer/); + return out; +} + +test("watchdog: inspect's target stat never answers → stalled, marked, and nothing more is asked of the drive", async () => { + const media = (await watchdogCase(["stat"], () => + inspectChannelMedia(paths, SLUG, CONFIG), + )) as { status: string; detail?: string }; + assert.equal(media.status, "stalled"); + assert.match(String(media.detail), /^drive not answering \(location "USB drive"/); + // The next inspect, the guard and the walk make no call on the drive. + calls = []; + assert.equal((await inspectChannelMedia(paths, SLUG, CONFIG)).status, "stalled"); + await assert.rejects(() => assertChannelMediaReachable(paths, SLUG, CONFIG)); + assert.equal(await readChannelStat(paths, SLUG), null); + assert.deepEqual(calls.filter(onDrive), []); + // Cleared (two clean answers), the drive is asked again. + recordLocationHealth(LOC, "ok"); + recordLocationHealth(LOC, "ok"); + hang = null; + assert.equal( + (await inspectChannelMedia(paths, SLUG, CONFIG, { fresh: true })).status, + "ok", + ); +}); + +test("watchdog: a video directory's read never answers mid-walk → readChannelStat answers null", async () => { + for (const id of ["vid2", "vid3", "vid4", "vid5", "vid6"]) { + mkdirSync(path.join(TARGET, id), { recursive: true }); + } + const out = await watchdogCase(["readdir"], async () => { + // The walk's first readdir (of data/ itself, through the link) answers; + // the video directories' do not. + hang = (c) => + c.fn === "readdir" && c.path.startsWith(path.join(LINK, "vid")); + return readChannelStat(paths, SLUG); + }); + assert.equal(out, null); + // At most four video directories were asked before the stall refused the rest. + const asked = calls.filter((c) => c.fn === "readdir" && c.path.startsWith(path.join(LINK, "vid"))); + assert.ok(asked.length <= 4, `${asked.length} video dirs asked`); +}); + +test("watchdog: probeLocation's stat never answers → 'stalled'", async () => { + const probe = (await watchdogCase(["stat"], () => probeLocation(LOC, paths))) as { + status: string; + }; + assert.equal(probe.status, "stalled"); +}); + +test("watchdog: volumeFreeBytes' stat never answers → unknown", async () => { + const out = (await watchdogCase(["stat"], () => + volumeFreeBytes({ paths, locations: [LOC] }), + )) as Record<string, number | undefined>; + assert.equal(out.usb, undefined); +}); + +test("watchdog: a recency tail read never answers → not dated, not remembered as a miss", async () => { + const args = { + paths, + meta: [{ slug: SLUG, config: CONFIG }], + candidateIds: new Set(["vid1"]), + owner: new Map([["vid1", SLUG]]), + interpolate: false, + fresh: true, + }; + const keys = (await watchdogCase(["open"], () => buildRecencyKeys(args))) as Map< + string, + { key: string } + >; + assert.equal(keys.get("vid1")?.key, ""); + resetStorageHealth(); + hang = null; + assert.equal((await buildRecencyKeys(args)).get("vid1")?.key, "20260601"); +}); + +test("watchdog: a move onto a root whose stat never answers is refused", async () => { + const problem = await watchdogCase(["stat"], () => + relocationRootPresenceProblem( + DRIVE, + { locations: [LOC], defaultLocationId: "" }, + paths, + ), + ); + assert.match(String(problem), /drive not answering/); +}); + +test("watchdog (M3): the snapshot walk's video unit never answers → the refresh throws and writes no snapshot", async () => { + for (const id of ["vid2", "vid3", "vid4", "vid5", "vid6", "vid7"]) { + mkdirSync(path.join(TARGET, id), { recursive: true }); + } + const snapshotFile = path.join(paths.channelsDir, SLUG, "snapshot.json"); + await watchdogCase(["readdir"], async () => { + // data/ itself answers (the listing, the reconcile pass); the video + // directories do not. + hang = (c) => c.fn === "readdir" && c.path.startsWith(path.join(LINK, "vid")); + await assert.rejects( + () => generateChannelSnapshot(paths, SLUG), + (err: unknown) => err instanceof Error && err.name === "DriveNotAnsweringError", + ); + }); + assert.equal(existsSync(snapshotFile), false, "the last snapshot.json stands"); + // At most four video directories reached the drive. + const asked = calls.filter( + (c) => c.fn === "readdir" && c.path.startsWith(path.join(LINK, "vid")), + ); + assert.ok(asked.length <= 4, `${asked.length} video dirs asked`); +}); diff --git a/common/controller/storageWatch.test.ts b/common/controller/storageWatch.test.ts @@ -6,19 +6,36 @@ import path from "node:path"; import type { Paths } from "../lib/paths"; import type { SiteSettings } from "../lib/settings"; import { + autoPauseReasonOf, compileLanes, sanitizeChannelPriority, } from "../lib/channelPriority"; import { LANES } from "../lib/autoQueueTypes"; import { + refreshLocationHealth, resetStorageWatchSuspicion, + runStorageHealthPass, runStorageWatchPass, + startStorageHealthWatch, + startStorageWatch, + stopStorageHealthWatch, + stopStorageWatch, } from "./storageWatch"; +import { inspectChannelMedia } from "../lib/channelMedia"; +import { + locationHealth, + resetStorageHealth, + type LocationHealthState, +} from "../lib/storageHealth"; // THE CONFIRMATION COUNT IS MODULE STATE (see storageWatch.ts rule 3), so each // case starts from a clean one — otherwise the second test inherits the first -// test's suspicions and pauses on what should be its first pass. -beforeEach(() => resetStorageWatchSuspicion()); +// test's suspicions and pauses on what should be its first pass. The health +// state (lib/storageHealth.ts) is process state for the same reason. +beforeEach(() => { + resetStorageWatchSuspicion(); + resetStorageHealth(); +}); // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test controller/storageWatch.test.ts @@ -375,3 +392,200 @@ test("the restore needs only one good pass", async () => { assert.equal(h.writes, 1); }); }); + +// --------------------------------------------------------------------------- +// The health pass (15 s): is the drive ANSWERING +// --------------------------------------------------------------------------- +// +// The probe is injected: `probeLocationHealth` (a child `stat` raced against a +// 3 s timer) has its own tests in lib/storageHealthProbe.test.ts. Here the +// answers are scripted, one per pass. + +function scripted(answers: LocationHealthState[]) { + let i = 0; + return async () => answers[Math.min(i++, answers.length - 1)]; +} + +test("one missed probe stalls the location; pages then answer 'stalled' without asking", async () => { + await withTmp(async (h) => { + await seedRelocated(h, "slow", { targetExists: true }); + const lines: string[] = []; + const r = await runStorageHealthPass({ + io: h.io, + probe: scripted(["stalled"]), + log: (l) => lines.push(l), + }); + assert.deepEqual(r.answers, { cold: "stalled" }); + // Registered first (answering), so the miss is a transition from ok. + assert.deepEqual(r.transitions, [{ id: "cold", from: "ok", to: "stalled" }]); + assert.equal(locationHealth("cold")?.state, "stalled"); + assert.match(lines.join("\n"), /"cold": drive not answering/); + // The target is there and would answer, but nothing asks it. + const media = await inspectChannelMedia(h.paths, "slow"); + assert.equal(media.status, "stalled"); + // The five-minute pass sees the location as down (its probe answers + // "stalled" without a stat) and suspects the channel, as for any outage. + const w = await runStorageWatchPass({ paths: h.paths, io: h.io, bins: h.paths }); + assert.deepEqual(w.suspected, ["slow"]); + assert.equal(h.writes, 0); + }); +}); + +test("the stall clears only after two clean probes in a row", async () => { + await withTmp(async (h) => { + await seedRelocated(h, "slow", { targetExists: true }); + const probe = scripted(["stalled", "ok", "stalled", "ok", "ok"]); + const lines: string[] = []; + const pass = () => + runStorageHealthPass({ io: h.io, probe, log: (l) => lines.push(l) }); + await pass(); + await pass(); // one clean answer + assert.equal(locationHealth("cold")?.state, "stalled"); + await pass(); // missed again: the count starts over + await pass(); // one clean + assert.equal(locationHealth("cold")?.state, "stalled"); + assert.equal((await inspectChannelMedia(h.paths, "slow")).status, "stalled"); + const last = await pass(); // two clean in a row + assert.deepEqual(last.transitions, [{ id: "cold", from: "stalled", to: "ok" }]); + assert.match(lines.at(-1) ?? "", /"cold": answering again/); + assert.equal((await inspectChannelMedia(h.paths, "slow")).status, "ok"); + }); +}); + +test("a location no longer configured is forgotten", async () => { + await withTmp(async (h) => { + await runStorageHealthPass({ io: h.io, probe: scripted(["stalled"]) }); + assert.equal(locationHealth("cold")?.state, "stalled"); + await runStorageHealthPass({ locations: [], probe: scripted(["ok"]) }); + assert.equal(locationHealth("cold"), undefined); + }); +}); + +test("a probe that throws is 'could not ask': ok, never stalled", async () => { + await withTmp(async (h) => { + const r = await runStorageHealthPass({ + io: h.io, + probe: async () => { + throw new Error("spawn failed"); + }, + }); + assert.deepEqual(r.answers, { cold: "ok" }); + }); +}); + +test("the health pass is armed on its own, runs once at once, and stops; the five-minute watch arms no health pass", async () => { + await withTmp(async (h) => { + let asked = 0; + const armed = startStorageHealthWatch({ + io: h.io, + probe: async () => { + asked += 1; + return "stalled"; + }, + log: () => {}, + }); + try { + assert.equal(armed, true); + // Armed once per process. + assert.equal(startStorageHealthWatch({ io: h.io }), false); + for (let i = 0; i < 50 && locationHealth("cold")?.state !== "stalled"; i++) { + await new Promise((r) => setTimeout(r, 10)); + } + assert.equal(asked, 1); + assert.equal(locationHealth("cold")?.state, "stalled"); + } finally { + stopStorageHealthWatch(); + } + assert.equal(startStorageHealthWatch({ io: h.io, probe: async () => "ok" }), true); + stopStorageHealthWatch(); + // The watch (below the idle gate) runs nothing at arm time and asks no drive. + resetStorageHealth(); + assert.equal(startStorageWatch({ paths: h.paths, io: h.io, bins: h.paths, write: false }), true); + stopStorageWatch(); + assert.equal(locationHealth("cold"), undefined); + }); +}); + +test("a Refresh asks one location now: it counts as one answer, and prunes nothing", async () => { + await withTmp(async (h) => { + const other = { id: "other", label: "Other", root: "/elsewhere", autoRepoint: false }; + await runStorageHealthPass({ + locations: [h.io.read().storage.locations[0], other], + probe: scripted(["stalled"]), + }); + const cold = h.io.read().storage.locations[0]; + assert.equal(await refreshLocationHealth(cold, async () => "ok"), "ok"); + // One clean answer is not two. + assert.equal(locationHealth("cold")?.state, "stalled"); + assert.equal(locationHealth("other")?.state, "stalled"); + await refreshLocationHealth(cold, async () => "ok"); + assert.equal(locationHealth("cold")?.state, "ok"); + assert.equal(locationHealth("other")?.state, "stalled"); + }); +}); + +test("the pass registers every location, records a verdict's detector, and a verdict with no answer changes nothing", async () => { + await withTmp(async (h) => { + const verdicts = [ + { answer: null, detector: "counters" as const, device: "sdz1" }, + { + answer: "stalled" as const, + detector: "counters" as const, + device: "sdz1", + cause: "its disk (sdz1) had 1 request(s) in flight and completed none in 15 s", + }, + ]; + let i = 0; + const probe = async () => verdicts[i++]; + const lines: string[] = []; + const first = await runStorageHealthPass({ io: h.io, probe, log: (l) => lines.push(l) }); + // Registered, answering, and the detector named — with no verdict yet. + assert.deepEqual(first.answers, {}); + assert.deepEqual(first.transitions, []); + assert.equal(locationHealth("cold")?.state, "ok"); + assert.equal(locationHealth("cold")?.detector, "counters"); + const second = await runStorageHealthPass({ io: h.io, probe, log: (l) => lines.push(l) }); + assert.deepEqual(second.transitions, [{ id: "cold", from: "ok", to: "stalled" }]); + assert.match(String(locationHealth("cold")?.cause), /its disk \(sdz1\)/); + assert.match(lines.join("\n"), /"cold": drive not answering — its disk \(sdz1\)/); + }); +}); + +test("a counters verdict records its device (for the watchdog); a stat verdict forgets it", async () => { + await withTmp(async (h) => { + const verdicts = [ + { answer: null, detector: "counters" as const, device: "sdz1" }, + { answer: "ok" as const, detector: "stat" as const }, + ]; + let i = 0; + const probe = async () => verdicts[i++]; + await runStorageHealthPass({ io: h.io, probe }); + assert.equal(locationHealth("cold")?.device, "sdz1"); + await runStorageHealthPass({ io: h.io, probe }); + assert.equal(locationHealth("cold")?.device, undefined); + assert.equal(locationHealth("cold")?.detector, "stat"); + }); +}); + +test("a stall auto-pauses after two passes, and says the drive is not answering (not that it is not there)", async () => { + await withTmp(async (h) => { + await seedRelocated(h, "slow", { targetExists: true }); + await runStorageHealthPass({ io: h.io, probe: async () => "stalled" }); + await twoPasses(h); + const entry = h.io.read().channelPriority.channels.slow; + assert.equal(entry?.tier, "paused"); + assert.equal(entry?.autoPaused?.cause, "not-answering"); + const reason = autoPauseReasonOf(h.io.read().channelPriority, "slow"); + assert.match(String(reason), /drive that is not answering/); + assert.match(String(reason), /when the drive answers again/); + // The sanitizer keeps the cause; a record without one reads as not there. + const kept = sanitizeChannelPriority(h.io.read().channelPriority); + assert.equal(kept.channels.slow?.autoPaused?.cause, "not-answering"); + const old = sanitizeChannelPriority({ + channels: { + a: { tier: "paused", autoPaused: { reason: "storage", since: "", previousTier: "low" } }, + }, + }); + assert.match(String(autoPauseReasonOf(old, "a")), /drive that is not there/); + }); +}); diff --git a/common/controller/storageWatch.ts b/common/controller/storageWatch.ts @@ -14,13 +14,33 @@ import { resolveFocusSlugs, restoreAfterMedia, sanitizeChannelPriority, + type AutoPauseCause, type ChannelPriority, } from "../lib/channelPriority"; import { LANES } from "../lib/autoQueueTypes"; import { siteChannelIndex } from "../lib/site"; import { inspectChannelMedia } from "../lib/channelMedia"; -import { locationOfDataDir } from "../lib/storageLocations"; -import type { VolumeBins } from "../lib/storageVolumes"; +import { + locationOfDataDir, + type StorageLocation, +} from "../lib/storageLocations"; +import { + detectLocationHealth, + type HealthVerdict, + type LocationHealthProbe, + type VolumeBins, +} from "../lib/storageVolumes"; +import { + HEALTH_PROBE_INTERVAL_MS, + HEALTH_PROBE_TIMEOUT_MS, + NOT_ANSWERING, + noteLocationDetector, + pruneLocationHealth, + recordLocationHealth, + registerLocationHealth, + type HealthTransition, + type LocationHealthState, +} from "../lib/storageHealth"; import { listChannelConfigs } from "./channels"; import { maybeAutoRepoint, probeAllLocations } from "./storageLocations"; @@ -185,6 +205,7 @@ export async function runStorageWatchPass( const configs = await listChannelConfigs(paths); let model: ChannelPriority = settings.channelPriority; + const pauseCauses = new Map<string, AutoPauseCause>(); for (const { slug, config } of configs) { const wasAutoPaused = Boolean(model.channels[slug]?.autoPaused); @@ -208,14 +229,28 @@ export async function runStorageWatchPass( const loc = locationOfDataDir(dataDir, locations); const probe = loc ? probes[loc.id] : undefined; const locationDown = Boolean(loc) && probe?.status !== "available"; - const media = await inspectChannelMedia(paths, slug, config); + // FRESH: this pass is the detector, and the page memo is not what it asks. + // A location the health probe found not answering reads `stalled` here + // without a call, and a stall is `down` like any other. + const media = await inspectChannelMedia(paths, slug, config, { + fresh: true, + }); // `in-transition` is NEVER a reason to pause: a marker means a move is // running or was interrupted, and the relocate job is precisely the thing // that would then be refused by the state it created. + // A stall is down too, on a location or not (a root typed by hand gets its + // `stalled` from the watchdog alone). const down = media.status === "in-transition" ? false - : locationDown || media.status === "unreachable"; + : locationDown || + media.status === "unreachable" || + media.status === "stalled"; + // WHICH down it is, for the pause record's words (autoPauseReasonOf). + const cause: AutoPauseCause = + media.status === "stalled" || probe?.status === "stalled" + ? "not-answering" + : "not-there"; if (down && !wasAutoPaused) { // ONE BAD READ IS A SUSPICION, TWO IN A ROW IS A FACT. See rule 3. @@ -230,7 +265,8 @@ export async function runStorageWatchPass( continue; } const before = model; - model = autoPauseForMedia(model, slug); + model = autoPauseForMedia(model, slug, new Date(), cause); + pauseCauses.set(slug, cause); // autoPauseForMedia no-ops on a channel the OPERATOR already paused — // which is right, and means "nothing changed" is a normal outcome here. if (model !== before) { @@ -286,7 +322,9 @@ export async function runStorageWatchPass( ) : stored; let merged: ChannelPriority = base; - for (const slug of out.paused) merged = autoPauseForMedia(merged, slug); + for (const slug of out.paused) { + merged = autoPauseForMedia(merged, slug, new Date(), pauseCauses.get(slug)); + } for (const slug of out.restored) merged = restoreAfterMedia(merged, slug); merged = sanitizeChannelPriority(merged); // AND THE TREES, in the same write. Two writes to one settings file race each @@ -308,23 +346,177 @@ export async function runStorageWatchPass( return out; } -// --------------------------------------------------------------------------- -// The cadence -// --------------------------------------------------------------------------- - // Five minutes. A drive does not come and go on a timescale a person would // notice faster than that, and every pass is one findmnt per location plus two // stats per relocated channel — cheap, but not free, and this runs for the life // of the process. export const STORAGE_WATCH_INTERVAL_MS = 5 * 60_000; +// --------------------------------------------------------------------------- +// The health pass: is each location's drive ANSWERING +// --------------------------------------------------------------------------- +// +// A SECOND CADENCE, AND A MUCH SHORTER ONE. The pass above asks "is the disk +// here" every five minutes and pauses on two misses, which is right for a +// cable pulled out. It is no help for a drive that is here and stalled — an +// SMR disk in a USB enclosure resetting under a long write — because every +// in-process call on that drive waits for it, and four waits stop the editor +// answering at all. So every 15 s this reads each location's block device +// counters in /sys (`detectLocationHealth`; a child `stat` of the root only +// where no device can be named) and records the answer in `lib/storageHealth.ts`, +// which every page and poll consults before it touches a drive. One `stalled` +// answer marks a location at once; two clean answers in a row clear it (the +// rules are that module's). +// +// READ-ONLY AND IN MEMORY. It writes no settings and pauses nothing: the pass +// above sees a stalled location as down (its probe answers `stalled` without +// asking) and pauses on its own cadence. So, like the boot probe, it is armed +// ABOVE the idle gate (`startStorageHealthWatch`, from instrumentation): an idle +// boot has no five-minute pass, but it has the health pass — without it nothing +// registers the locations, and a stall the watchdog marks is never cleared. + +export type StorageHealthPassOpts = { + // Default: the configured locations, read from settings. + locations?: readonly StorageLocation[]; + io?: { read: () => SiteSettings }; + // findmnt, for naming each root's block device. Default: getPaths(). + bins?: Pick<VolumeBins, "findmntBin">; + // Test seam. Default: `detectLocationHealth` — the block device's counters, + // or a child `stat` against a 3 s timer when no device can be named. + probe?: LocationHealthProbe; + now?: () => number; + log?: (line: string) => void; +}; + +export type StorageHealthPassResult = { + probed: number; + answers: Record<string, LocationHealthState>; + // Only the locations whose state changed. + transitions: HealthTransition[]; +}; + +// A probe's answer as a verdict. A bare state names no detector; a probe that +// threw is "could not ask": `ok`, never `stalled`. +function asVerdict(answer: LocationHealthState | HealthVerdict): HealthVerdict { + return typeof answer === "string" ? { answer } : answer; +} + +const STAT_CAUSE = `a stat of its root did not answer within ${HEALTH_PROBE_TIMEOUT_MS / 1000} s`; + +// Record one verdict. A verdict with no answer records nothing but the +// detector that gave it. +function recordVerdict( + loc: StorageLocation, + verdict: HealthVerdict, + now: number, +): HealthTransition | null { + // The counters' device, for the watchdog's slow-or-stalled check; a stat + // verdict forgets it. + const device = + verdict.detector === "counters" + ? verdict.device + : verdict.detector === "stat" + ? null + : undefined; + if (verdict.answer === null) { + if (verdict.detector) noteLocationDetector(loc.id, verdict.detector, device); + return null; + } + return recordLocationHealth(loc, verdict.answer, { + now, + cause: verdict.cause ?? STAT_CAUSE, + ...(verdict.detector ? { detector: verdict.detector } : {}), + ...(device !== undefined ? { device } : {}), + }); +} + +export async function runStorageHealthPass( + opts: StorageHealthPassOpts = {}, +): Promise<StorageHealthPassResult> { + const log = opts.log ?? (() => {}); + const locations = + opts.locations ?? (opts.io ?? DEFAULT_IO).read().storage.locations; + pruneLocationHealth(locations.map((l) => l.id)); + // Every configured location has an entry before anything is asked, so the + // watchdog (lib/storageHealth.ts `onDrive`) can find a channel's location + // even before the counters have given a first verdict. + registerLocationHealth(locations); + const bins = opts.bins ?? getPaths(); + const probe: LocationHealthProbe = + opts.probe ?? ((loc) => detectLocationHealth(loc, bins)); + // Every location at once: each answer is bounded by the probe's own timer, + // so the pass is too, and one stalled drive does not delay the others. + const answers = await Promise.all( + locations.map(async (loc) => { + const verdict = await probe(loc).then(asVerdict, (): HealthVerdict => ({ + answer: "ok", + })); + return [loc, verdict] as const; + }), + ); + const out: StorageHealthPassResult = { + probed: answers.length, + answers: {}, + transitions: [], + }; + const now = opts.now?.() ?? Date.now(); + for (const [loc, verdict] of answers) { + if (verdict.answer !== null) out.answers[loc.id] = verdict.answer; + const t = recordVerdict(loc, verdict, now); + if (!t) continue; + out.transitions.push(t); + if (t.to === "stalled") { + log( + `[storage] "${loc.id}": ${NOT_ANSWERING} — ${verdict.cause ?? STAT_CAUSE}; ` + + `pages and polls skip it until two passes in a row find it answering`, + ); + } else if (t.from === "stalled") { + log(`[storage] "${loc.id}": answering again (${t.to})`); + } + } + return out; +} + +// ONE LOCATION, NOW: what /storage's Refresh asks before its own probe, so the +// operator pressing it after doing something about the drive gets an answer +// taken afterwards. It counts as one answer like any other — a stalled location +// still needs two clean ones in a row — and the counters give none when their +// last sample is under MIN_COUNTER_INTERVAL_MS old. Nothing is pruned. +export async function refreshLocationHealth( + loc: StorageLocation, + probe?: LocationHealthProbe, +): Promise<LocationHealthState | null> { + registerLocationHealth([loc]); + const ask: LocationHealthProbe = + probe ?? ((l) => detectLocationHealth(l, getPaths())); + const verdict = await ask(loc).then(asVerdict, (): HealthVerdict => ({ + answer: "ok", + })); + recordVerdict(loc, verdict, Date.now()); + return verdict.answer; +} + +// --------------------------------------------------------------------------- +// The cadence +// --------------------------------------------------------------------------- + +// Fifteen seconds (lib/storageHealth.ts says why). Each pass reads each +// location's device counters in /sys (a findmnt only when the device is not +// known yet or its /sys entry stopped reading; with no device, one short-lived +// `stat`). +export const STORAGE_HEALTH_INTERVAL_MS = HEALTH_PROBE_INTERVAL_MS; + // A per-module-copy singleton, deliberately left so: it is not a temp-file // name (slice W folded every tmp + rename onto lib/jsonFile-server.ts, whose // state is on globalThis), and its one caller is editor/instrumentation.ts, -// so only one copy ever arms it. +// so only one copy ever arms it. (The health STATE the second timer writes is +// on globalThis — pages in another module copy read it.) let timer: ReturnType<typeof setInterval> | null = null; +let healthTimer: ReturnType<typeof setInterval> | null = null; +let healthInFlight = false; -// ARMED ONCE PER PROCESS. `unref()` so it never holds the event loop open — a +// THE FIVE-MINUTE PASS, ARMED ONCE PER PROCESS, below the idle gate: it writes +// settings (an auto-pause). `unref()` so it never holds the event loop open — a // CLI that imports a controller must still exit. export function startStorageWatch( opts: StorageWatchOpts & { intervalMs?: number } = {}, @@ -343,7 +535,51 @@ export function startStorageWatch( } export function stopStorageWatch(): void { - if (!timer) return; - clearInterval(timer); + if (timer) clearInterval(timer); timer = null; } + +// THE FIFTEEN-SECOND HEALTH PASS, ARMED ONCE PER PROCESS, above the idle gate: +// it is in memory and writes nothing (see the section header). It also runs +// once at once, so the locations are registered and a drive that is already +// stalled is on its way to being known before the first page. +export function startStorageHealthWatch( + opts: { + io?: { read: () => SiteSettings }; + bins?: Pick<VolumeBins, "findmntBin">; + probe?: LocationHealthProbe; + intervalMs?: number; + log?: (line: string) => void; + } = {}, +): boolean { + if (healthTimer) return false; + const health = () => { + // One at a time: a pass is bounded by its timers, but a pass that overran + // the interval must not stack a second one on top of it. + if (healthInFlight) return; + healthInFlight = true; + void runStorageHealthPass({ + io: opts.io, + bins: opts.bins, + probe: opts.probe, + log: opts.log, + }) + .catch((err) => { + (opts.log ?? console.warn)( + `[storage] health pass failed: ${(err as Error).message}`, + ); + }) + .finally(() => { + healthInFlight = false; + }); + }; + healthTimer = setInterval(health, opts.intervalMs ?? STORAGE_HEALTH_INTERVAL_MS); + healthTimer.unref?.(); + health(); + return true; +} + +export function stopStorageHealthWatch(): void { + if (healthTimer) clearInterval(healthTimer); + healthTimer = null; +} diff --git a/common/lib/channelMedia.ts b/common/lib/channelMedia.ts @@ -2,6 +2,15 @@ import path from "node:path"; import { lstat, readFile, readlink, rm, stat } from "node:fs/promises"; import type { Paths } from "./paths"; import type { ChannelConfig } from "./channelConfig"; +import { + DRIVE_CALL_BUDGET_MS, + NOT_ANSWERING, + isDriveNotAnswering, + onDrive, + sinceText, + stalledLocationForPath, + type LocationHealth, +} from "./storageHealth"; // WHERE A CHANNEL'S MEDIA ACTUALLY IS, and whether it can be reached. // @@ -60,7 +69,12 @@ export type ChannelMediaStatus = // A relocation is in flight (or was interrupted): the marker is present. | "in-transition" // Disk and config disagree, in either direction. Never guessed past. - | "inconsistent"; + | "inconsistent" + // Relocated onto a storage location whose drive is not answering + // (`lib/storageHealth.ts`). Answered from memory, WITHOUT a filesystem call: + // a call there would block one of the process's few I/O threads for as long + // as the drive takes to come back. Held and refused like `unreachable`. + | "stalled"; export type ChannelMediaLocation = { // Always channelDir/data — the path every reader uses, relocated or not. @@ -162,6 +176,7 @@ export async function clearRelocationMarker( slug: string, ): Promise<void> { await rm(relocationMarkerPath(paths, slug), { force: true }); + forgetChannelMedia(slug); } // The `dataDir` field alone, read straight off config.json. Deliberately NOT @@ -185,13 +200,124 @@ async function readConfiguredDataDir( } } +// The configured target's drive is not answering: the answer, from memory. The +// detail names the location and when it stopped, never a path — the badge's +// title shows it, and /storage has the paths. +export function stalledMediaLocation( + dataDir: string, + configured: string, + // Null when the drive is on no location the health state knows (a root typed + // by hand) and the watchdog found it not answering, or when the watchdog + // found it slow rather than stalled. + health: LocationHealth | null, + // The watchdog's own words, for a refusal that marked no location. + detail?: string, +): ChannelMediaLocation { + return { + dataDir, + relocated: true, + target: configured, + status: "stalled", + detail: health + ? `${NOT_ANSWERING} (location "${health.label}", ${sinceText(health.since)})` + : (detail ?? + `${NOT_ANSWERING} (a read did not answer within ${DRIVE_CALL_BUDGET_MS / 1000} s)`), + }; +} + +// THE STALL, for a caller holding a parsed config: the stalled location the +// channel's media is on, or null. No I/O — the question every page and poll +// that reads a channel's `data/` asks before it does. +export function channelMediaStall( + config: Pick<ChannelConfig, "dataDir"> | null | undefined, +): LocationHealth | null { + const dir = config?.dataDir?.trim(); + return dir ? stalledLocationForPath(dir) : null; +} + +// --------------------------------------------------------------------------- +// The memo +// --------------------------------------------------------------------------- +// +// FIVE SECONDS, PER CHANNEL, KEYED BY SLUG AND THE CONFIGURED TARGET. The home +// page, /channels and the auto-queue status poll (every three seconds, four +// lanes) each inspect every channel; without this each of them costs three +// syscalls a relocated channel, every time, on a drive that may be the slow +// one. The key carries the configured `dataDir`, so a move that rewrites it is +// a new key at once, and the movers clear the memo outright +// (`forgetChannelMedia`) whenever a marker is written or removed. +// +// THE STALL GATE IS ASKED BEFORE A REMEMBERED ANSWER IS GIVEN, so a drive that +// stops answering is seen on the next call even when an `ok` from four seconds +// ago is remembered. A remembered `in-transition` is given as it is: the +// marker is asked before the gate (below), and the movers forget the channel +// whenever they write or remove it. +// +// `fresh: true` BYPASSES IT, and every caller that decides something from the +// answer passes it: the start-of-work guard (`assertChannelMediaReachable`), +// the movers, the index and stats builds, the storage watch, eviction and the +// re-point preflight. The memo is for pages and polls. +// +// ONE MAP PER PROCESS (on `globalThis`), for the reason `storageHealth.ts` +// gives: the movers that clear it run in one bundle layer and the pages that +// read it in another. + +export const CHANNEL_MEDIA_MEMO_MS = 5_000; + +export type InspectOptions = { + // Skip the memo: take a fresh answer, and do not remember it. + fresh?: boolean; + // Test seam for the clock. + now?: number; +}; + +type MediaMemo = Map<string, { at: number; location: ChannelMediaLocation }>; + +declare global { + // eslint-disable-next-line no-var + var __yttChannelMediaMemo__: MediaMemo | undefined; +} + +function mediaMemo(): MediaMemo { + if (!globalThis.__yttChannelMediaMemo__) { + globalThis.__yttChannelMediaMemo__ = new Map(); + } + return globalThis.__yttChannelMediaMemo__; +} + +// Forget what the memo holds: for one channel, or for every channel. The +// movers call it whenever the disk changes under a channel (a marker written or +// removed, a link swapped, a location re-pointed), so the next page sees it. +export function forgetChannelMedia(slug?: string): void { + const memo = mediaMemo(); + if (slug === undefined) { + memo.clear(); + return; + } + for (const key of [...memo.keys()]) { + if (key.split("\u0000")[1] === slug) memo.delete(key); + } +} + +function memoKey(paths: ChannelMediaPaths, slug: string, configured?: string): string { + return `${paths.channelsDir}\u0000${slug}\u0000${configured ?? ""}`; +} + +// Past this many entries, expired ones are swept on insert. A corpus has tens +// of channels; this only matters to a process that inspects many corpora. +const MEMO_SWEEP_AT = 512; + // Two stats and (at most) one small JSON read. Render-safe: nothing here walks a // directory, so calling it per channel on a listing page costs three syscalls a -// row. +// row — or none, for five seconds after the last answer (see the memo above). +// On a stalled location the one call that reaches the drive (the target's +// `stat`) is not made; the marker, the link and config.json are on the corpus +// disk and are read as usual. export async function inspectChannelMedia( paths: ChannelMediaPaths, slug: string, config?: Pick<ChannelConfig, "dataDir"> | null, + opts: InspectOptions = {}, ): Promise<ChannelMediaLocation> { const dataDir = channelMediaDir(paths, slug); const configured = @@ -201,6 +327,42 @@ export async function inspectChannelMedia( ? config.dataDir.trim() : undefined; + const now = opts.now ?? Date.now(); + const key = memoKey(paths, slug, configured); + const memo = mediaMemo(); + if (!opts.fresh) { + const hit = memo.get(key); + if (hit && now - hit.at < CHANNEL_MEDIA_MEMO_MS) { + if (hit.location.status !== "in-transition" && configured) { + const stall = stalledLocationForPath(configured); + if (stall) return stalledMediaLocation(dataDir, configured, stall); + } + return { ...hit.location }; + } + } + const location = await inspectOnDisk(paths, slug, dataDir, configured); + // A stall is not remembered: the health state is already its memory. + if (!opts.fresh && location.status !== "stalled") { + if (memo.size >= MEMO_SWEEP_AT) { + for (const [k, v] of memo) { + if (now - v.at >= CHANNEL_MEDIA_MEMO_MS) memo.delete(k); + } + } + memo.set(key, { at: now, location }); + } + return { ...location }; +} + +async function inspectOnDisk( + paths: ChannelMediaPaths, + slug: string, + dataDir: string, + configured: string | undefined, +): Promise<ChannelMediaLocation> { + // THE MARKER FIRST. It is in the channel dir, on the corpus disk, so reading + // it costs the drive nothing — and a channel mid-move reads `in-transition` + // whatever its drive is doing, which is what a resumed move and every guard + // key off. const marker = await readRelocationMarker(paths, slug); if (marker) { return { @@ -215,6 +377,15 @@ export async function inspectChannelMedia( }; } + // THE GATE: a channel whose configured target is on a location whose drive + // is not answering is answered from memory, before the link is looked at and + // before the target's stat, which would hold an I/O thread for as long as the + // drive takes. + if (configured) { + const stall = stalledLocationForPath(configured); + if (stall) return stalledMediaLocation(dataDir, configured, stall); + } + let link: Awaited<ReturnType<typeof lstat>> | null = null; try { link = await lstat(dataDir); @@ -266,8 +437,12 @@ export async function inspectChannelMedia( // The link points at a DEEP path (<root>/<slug>/data), so an unmounted root // gives ENOENT here. An empty mountpoint can never be mistaken for the // media, which is the whole reason the suffix is fixed. + // + // THE ONE CALL HERE THAT REACHES THE DRIVE, so it goes through the + // watchdog: not made while the location is stalled, and a stat that has + // not answered in 3 s marks it stalled and answers `stalled` now. try { - const st = await stat(configured); + const st = await onDrive(configured, () => stat(configured)); if (!st.isDirectory()) { return { dataDir, @@ -277,7 +452,10 @@ export async function inspectChannelMedia( detail: `${configured} exists but is not a directory`, }; } - } catch { + } catch (err) { + if (isDriveNotAnswering(err)) { + return stalledMediaLocation(dataDir, configured, err.health, err.message); + } return { dataDir, relocated: true, @@ -316,6 +494,10 @@ export async function inspectChannelMedia( // "ok" and "in-place" pass; everything else throws. An in-transition or // inconsistent channel is refused for the same reason an unreachable one is: // the caller would otherwise read a half-populated or empty dir as the truth. +// A stalled one is refused because the work would block on the drive. +// +// ALWAYS FRESH: this is the start-of-work guard, and a remembered "ok" from a +// few seconds ago is not what a job about to read `data/` should be told. // // THIS CHECK HAS A TWIN. `checkChannelReachable` in // `umtool/report-to-video/cues.mjs` repeats the same statuses in plain `.mjs`, @@ -329,7 +511,9 @@ export async function assertChannelMediaReachable( slug: string, config?: Pick<ChannelConfig, "dataDir"> | null, ): Promise<ChannelMediaLocation> { - const location = await inspectChannelMedia(paths, slug, config); + const location = await inspectChannelMedia(paths, slug, config, { + fresh: true, + }); if (location.status === "ok" || location.status === "in-place") { return location; } diff --git a/common/lib/channelMediaHold.ts b/common/lib/channelMediaHold.ts @@ -33,6 +33,7 @@ export const HELD_REASON: Record<ChannelMediaStatus, string> = { unreachable: "its media is not reachable (drive not mounted?)", "in-transition": "a move of its media is in progress or was interrupted", inconsistent: "its data link and its config disagree", + stalled: "its drive is not answering (a stalled disk)", ok: "reachable", "in-place": "reachable", }; diff --git a/common/lib/channelPriority.ts b/common/lib/channelPriority.ts @@ -183,8 +183,15 @@ export type ChannelAutoPause = { reason: "storage"; since: string; previousTier: StoredChannelTier; + cause?: AutoPauseCause; }; +// Which storage trouble paused it: the drive is not there (unmounted, +// unplugged), or it is there and not answering (lib/storageHealth.ts). A record +// with none was written before the second existed, when the first was the only +// one. +export type AutoPauseCause = "not-there" | "not-answering"; + export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = { reason: "One reason today. A union so a second one has somewhere to go, and so " + @@ -195,6 +202,11 @@ export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = { "The base tier the channel had before the machine paused it; what a " + "restore puts back. Never `paused` (that would restore to paused — a " + "no-op dressed as a restore).", + cause: + "`not-there` (the drive is unmounted or unplugged) or `not-answering` (it " + + "is there and does not answer: a stalled disk). Only what the words on " + + "/review, the rack and the channel page say. Optional: a record written " + + "before it existed reads as `not-there`.", }; // Each field is documented in CHANNEL_PRIORITY_FIELD_DOCS below (rendered into SETTINGS.md). @@ -387,12 +399,16 @@ function sanitizeAutoPause( reason: "storage", since: typeof r.since === "string" ? r.since : "", previousTier: previousTier === "paused" ? DEFAULT_CHANNEL_TIER : previousTier, + ...(r.cause === "not-there" || r.cause === "not-answering" + ? { cause: r.cause } + : {}), }; } // --- Auto-pause: the machine's own pause, and its undo ---------------------- -// PAUSE A CHANNEL BECAUSE ITS DRIVE IS NOT THERE, recording what to put back. +// PAUSE A CHANNEL BECAUSE ITS DRIVE IS NOT THERE (OR NOT ANSWERING), recording +// what to put back, and which of the two it was. // // A NO-OP IN TWO CASES, and both matter. Already auto-paused: a flapping drive // must not overwrite `previousTier` with the `paused` it wrote last time, which @@ -403,6 +419,7 @@ export function autoPauseForMedia( model: ChannelPriority, slug: string, now: Date = new Date(), + cause?: AutoPauseCause, ): ChannelPriority { const key = slug.trim(); if (!key) return model; @@ -421,6 +438,7 @@ export function autoPauseForMedia( reason: "storage", since: now.toISOString(), previousTier, + ...(cause ? { cause } : {}), }, }, }, @@ -466,6 +484,12 @@ export function autoPauseReasonOf( const auto = model.channels[slug]?.autoPaused; if (!auto) return null; const since = auto.since ? ` since ${auto.since.slice(0, 10)}` : ""; + if (auto.cause === "not-answering") { + return ( + `Auto-paused — its media is on a drive that is not answering${since}. ` + + `It returns to ${auto.previousTier} on its own when the drive answers again.` + ); + } return ( `Auto-paused — its media is on a drive that is not there${since}. ` + `It returns to ${auto.previousTier} on its own when the drive is back.` diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts @@ -113,6 +113,7 @@ const DECLARED: EnvVarDecl[] = [ { name: "AUDIO_CHECK_BACKOFF_FACTOR", audience: "runtime", default: "the built-in factor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "The audio check's interval backoff factor, in (0, 1], for a one-off run." }, { name: "ARCHILYZER_STATS_ALLOW_DOWNGRADE", audience: "runtime", default: "off", readBy: "common/controller/buildStats.ts", doc: "`1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions." }, { name: "ARCHILYZER_INDEX_ALLOW_HELD", audience: "runtime", default: "off", readBy: "common/controller/buildIndex.ts", doc: "`1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel." }, + { name: "UV_THREADPOOL_SIZE", audience: "runtime", default: "`16` for the editor (`4` is Node's own)", readBy: "Node's libuv (set by editor/package.json `start` and docker/entrypoint.sh)", doc: "Threads in Node's pool for filesystem calls. A call on a stalled drive holds one until the drive answers, so the editor starts with 16. It buys time for calls already in flight and isolates nothing: the storage health probe and its gate keep new calls off a stalled drive." }, { name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." }, { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." }, { name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." }, diff --git a/common/lib/paths.ts b/common/lib/paths.ts @@ -178,9 +178,10 @@ let cached: Paths | null = null; // reference — to every file under it when the path is a directory // (plans/FACTS.md, "A path joined from `process.cwd()` …"). So every join on // such a value goes through this one opted-out call. Nothing changes at run -// time. -function under(...parts: string[]): string { - return path.join(/* turbopackIgnore: true */ ...parts); +// time. The comment sits before a named first argument, not a spread: that is +// the form Turbopack's own advice shows. +function under(first: string, ...rest: string[]): string { + return path.join(/* turbopackIgnore: true */ first, ...rest); } export function getPaths(): Paths { diff --git a/common/lib/ports.test.ts b/common/lib/ports.test.ts @@ -23,6 +23,11 @@ import { const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); const PACKAGES = ["", "common", "editor", "export", "homepage", "mcp", "umtool", "umtool/report-to-video"]; +// A script's `${NAME:-N}` that is a number and NOT a port, named one by one so +// every other spelled default is still read as a port: the editor's `start` +// gives Node's thread pool 16 threads (envVars.ts, UV_THREADPOOL_SIZE). +const NUMERIC_NOT_PORTS = new Set(["UV_THREADPOOL_SIZE"]); + function scripts(dir: string): Record<string, string> { const file = path.join(ROOT, dir, "package.json"); return (JSON.parse(readFileSync(file, "utf8")).scripts ?? {}) as Record<string, string>; @@ -63,6 +68,7 @@ function found(): Found[] { for (const [script, line] of Object.entries(scripts(pkg))) { const where = `${pkg || "."}/package.json "${script}"`; for (const m of line.matchAll(/\$\{([A-Z0-9_]+):-(\d+)\}/g)) { + if (NUMERIC_NOT_PORTS.has(m[1])) continue; hits.push({ where, name: m[1], port: Number(m[2]) }); } for (const m of line.matchAll(/--ports\s+(\S+)/g)) { diff --git a/common/lib/storageHealth.test.ts b/common/lib/storageHealth.test.ts @@ -0,0 +1,544 @@ +import { beforeEach, test } from "node:test"; +import assert from "node:assert/strict"; +import { + DRIVE_CALLS_IN_FLIGHT, + DRIVE_CALL_BUDGET_MS, + DriveNotAnsweringError, + HEALTH_CLEAN_TO_CLEAR, + allLocationHealth, + driveCallsInFlight, + isDriveNotAnswering, + noteLocationDetector, + onDrive, + registerLocationHealth, + setCounterReader, + setDriveCallBudget, + locationHealth, + notAnsweringText, + pruneLocationHealth, + recordLocationHealth, + resetStorageHealth, + sinceText, + stalledLocation, + stalledLocationForPath, +} from "./storageHealth"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/storageHealth.test.ts +// +// The rules the health state keeps (storageHealth.ts's header): one miss stalls +// at once, two clean answers in a row clear it, a miss in between starts the +// count again, and a new root starts the location over. + +beforeEach(() => { + resetStorageHealth(); + setDriveCallBudget(); + setCounterReader(undefined); +}); + +const USB = { id: "usb", label: "USB drive", root: "/mnt/usb/media" }; + +test("one missed probe marks the location stalled at once", () => { + assert.equal(recordLocationHealth(USB, "ok", { now: 1_000 })?.to, "ok"); + const t = recordLocationHealth(USB, "stalled", { now: 2_000, cause: "no answer" }); + assert.deepEqual(t, { id: "usb", from: "ok", to: "stalled" }); + const h = locationHealth("usb"); + assert.equal(h?.state, "stalled"); + assert.equal(h?.since, 2_000); + assert.equal(h?.cause, "no answer"); +}); + +test("a location first seen stalled is stalled", () => { + const t = recordLocationHealth(USB, "stalled", { now: 5 }); + assert.deepEqual(t, { id: "usb", from: null, to: "stalled" }); + assert.equal(stalledLocation(USB)?.since, 5); +}); + +test("one clean probe after a stall does not clear it; two in a row do", () => { + assert.equal(HEALTH_CLEAN_TO_CLEAR, 2); + recordLocationHealth(USB, "ok", { now: 0 }); + recordLocationHealth(USB, "stalled", { now: 10 }); + assert.equal(recordLocationHealth(USB, "ok", { now: 20 }), null); + assert.equal(locationHealth("usb")?.state, "stalled"); + // `since` stays the stall's start while it lasts. + assert.equal(locationHealth("usb")?.since, 10); + const t = recordLocationHealth(USB, "ok", { now: 30 }); + assert.deepEqual(t, { id: "usb", from: "stalled", to: "ok" }); + const h = locationHealth("usb"); + assert.equal(h?.state, "ok"); + assert.equal(h?.since, 30); + assert.equal(h?.cause, undefined); +}); + +test("a miss between two clean probes starts the count again", () => { + recordLocationHealth(USB, "stalled", { now: 0 }); + recordLocationHealth(USB, "ok", { now: 1 }); + assert.equal(recordLocationHealth(USB, "stalled", { now: 2 }), null); + // The stall's start is not moved by a second miss. + assert.equal(locationHealth("usb")?.since, 0); + recordLocationHealth(USB, "ok", { now: 3 }); + assert.equal(locationHealth("usb")?.state, "stalled"); + recordLocationHealth(USB, "ok", { now: 4 }); + assert.equal(locationHealth("usb")?.state, "ok"); +}); + +test("an absent answer is a clean one: an unplugged drive answers at once", () => { + recordLocationHealth(USB, "stalled", { now: 0 }); + recordLocationHealth(USB, "absent", { now: 1 }); + const t = recordLocationHealth(USB, "absent", { now: 2 }); + assert.deepEqual(t, { id: "usb", from: "stalled", to: "absent" }); + assert.equal(stalledLocation(USB), null); +}); + +test("a re-pointed root starts the location over", () => { + recordLocationHealth(USB, "stalled", { now: 0 }); + const moved = { ...USB, root: "/mnt/elsewhere/media" }; + // The stall belonged to the old root: the new root's lookup is not stalled, + // and neither is the old one's (the entry now describes the new root). + assert.equal(stalledLocation(moved), null); + const t = recordLocationHealth(moved, "ok", { now: 1 }); + assert.deepEqual(t, { id: "usb", from: null, to: "ok" }); + assert.equal(locationHealth("usb")?.root, "/mnt/elsewhere/media"); + assert.equal(stalledLocation(USB), null); +}); + +test("the path gate matches like locationOfDataDir: longest root, strictly under", () => { + const parent = { id: "parent", label: "Parent", root: "/mnt/p" }; + const child = { id: "child", label: "Child", root: "/mnt/p/archive" }; + recordLocationHealth(parent, "stalled", { now: 0 }); + recordLocationHealth(child, "ok", { now: 0 }); + // A channel on the nested location answers for that location, not its parent. + assert.equal(stalledLocationForPath("/mnt/p/archive/chan/data"), null); + assert.equal(stalledLocationForPath("/mnt/p/chan/data")?.id, "parent"); + // The root itself is not "under" it, and an unrelated path is on nothing. + assert.equal(stalledLocationForPath("/mnt/p"), null); + assert.equal(stalledLocationForPath("/elsewhere/chan/data"), null); + assert.equal(stalledLocationForPath(""), null); +}); + +test("pruning drops locations no longer configured", () => { + recordLocationHealth(USB, "stalled", { now: 0 }); + recordLocationHealth({ id: "b", label: "B", root: "/b" }, "ok", { now: 0 }); + pruneLocationHealth(["b"]); + assert.deepEqual(Object.keys(allLocationHealth()), ["b"]); + assert.equal(stalledLocationForPath("/mnt/usb/media/x/data"), null); +}); + +test("the state is one map per process, on globalThis", () => { + recordLocationHealth(USB, "stalled", { now: 0 }); + // A second module copy reads the same object (the house pattern). + assert.equal( + globalThis.__yttStorageHealth__?.byId.get("usb")?.state, + "stalled", + ); +}); + +test("the words: since a time today, or a date and time", () => { + const now = new Date(2026, 8, 29, 12, 0).getTime(); + const today = new Date(2026, 8, 29, 11, 35).getTime(); + const yesterday = new Date(2026, 8, 28, 23, 5).getTime(); + assert.equal(sinceText(today, now), "since 11:35"); + assert.equal(sinceText(yesterday, now), "since 2026-09-28 23:05"); + assert.equal(notAnsweringText({ since: today }, now), "not answering since 11:35"); +}); + +// ── the watchdog (`onDrive`) ──────────────────────────────────────────────── +// A call that never answers is a promise that never settles: exactly what a +// read blocked on a stalled drive looks like from here. No drive is involved. + +const never = () => new Promise<never>(() => {}); + +function deferred<T>() { + let resolve!: (v: T) => void; + let reject!: (e: Error) => void; + const promise = new Promise<T>((res, rej) => { + resolve = res; + reject = rej; + }); + return { promise, resolve, reject }; +} + +test("a call that answers in time passes through, value or error, and frees its slot", async () => { + registerLocationHealth([USB]); + assert.equal(await onDrive("/mnt/usb/media/ch/data", async () => 42), 42); + await assert.rejects( + () => onDrive(USB, async () => { + throw new Error("ENOENT"); + }), + /ENOENT/, + ); + assert.equal(driveCallsInFlight("usb"), 0); + assert.equal(locationHealth("usb")?.state, "ok"); +}); + +test("a call that never answers: stalled on the timer, the location marked at once, the call left to settle", async () => { + registerLocationHealth([USB], 0); + setDriveCallBudget(80); + const late = deferred<string>(); + const started = Date.now(); + await assert.rejects( + () => onDrive("/mnt/usb/media/ch/data", () => late.promise), + (err: unknown) => + err instanceof DriveNotAnsweringError && + isDriveNotAnswering(err) && + err.health?.id === "usb" && + /^drive not answering \(location "USB drive", since /.test(err.message), + ); + assert.ok(Date.now() - started < 1_000); + const h = locationHealth("usb"); + assert.equal(h?.state, "stalled"); + assert.ok((h?.since ?? 0) >= started, "since is now, not the entry's first sighting"); + assert.match(String(h?.cause), /a read in the editor did not answer within 0.08 s/); + // The slot is held until the call really returns. + assert.equal(driveCallsInFlight("usb"), 1); + late.resolve("finally"); + await new Promise((r) => setImmediate(r)); + assert.equal(driveCallsInFlight("usb"), 0); +}); + +test("a stalled location is refused without the call being made", async () => { + registerLocationHealth([USB]); + recordLocationHealth(USB, "stalled"); + let made = 0; + await assert.rejects( + () => onDrive("/mnt/usb/media/ch/data", async () => ++made), + DriveNotAnsweringError, + ); + await assert.rejects(() => onDrive(USB, async () => ++made), DriveNotAnsweringError); + assert.equal(made, 0); +}); + +test("at most four calls in flight on a location; the rest wait, and are refused without a call when it stalls", async () => { + assert.equal(DRIVE_CALLS_IN_FLIGHT, 4); + registerLocationHealth([USB]); + setDriveCallBudget(80); + let made = 0; + const calls = Array.from({ length: 7 }, () => + onDrive(USB, () => { + made += 1; + return never(); + }).then( + () => "answered", + (err: Error) => err.name, + ), + ); + await new Promise((r) => setImmediate(r)); + assert.equal(made, 4, "four in flight, three waiting"); + const outcomes = await Promise.all(calls); + assert.deepEqual(outcomes, Array(7).fill("DriveNotAnsweringError")); + assert.equal(made, 4, "the three that waited were refused without a call"); + assert.equal(locationHealth("usb")?.state, "stalled"); +}); + +test("a waiting call runs when a slot frees, if the location is still answering", async () => { + registerLocationHealth([USB]); + const gates = Array.from({ length: 4 }, () => deferred<number>()); + const first = gates.map((g) => onDrive(USB, () => g.promise)); + let fifth = false; + const waiting = onDrive(USB, async () => { + fifth = true; + return 5; + }); + await new Promise((r) => setImmediate(r)); + assert.equal(fifth, false); + gates[0].resolve(1); + assert.equal(await waiting, 5); + for (const g of gates.slice(1)) g.resolve(0); + await Promise.all(first); + assert.equal(driveCallsInFlight("usb"), 0); +}); + +test("a path on no known location is raced and capped by its root, and names no location", async () => { + setDriveCallBudget(80); + let made = 0; + // Two channels under one hand-typed root share its four slots. + const calls = [ + ...Array.from({ length: 3 }, () => "/hand/typed/chan-a/data"), + ...Array.from({ length: 3 }, () => "/hand/typed/chan-b/data"), + ].map((p) => + onDrive(p, () => { + made += 1; + return never(); + }).then( + () => "answered", + (err: DriveNotAnsweringError) => (err.health === null ? "refused" : "marked"), + ), + ); + await new Promise((r) => setImmediate(r)); + assert.equal(made, 4, "one root, four slots"); + assert.deepEqual(await Promise.all(calls), Array(6).fill("refused")); + assert.equal(made, 4); + assert.deepEqual(allLocationHealth(), {}); + // Every slot is held by a call given up on: the next is refused at once. + const started = Date.now(); + await assert.rejects(() => onDrive("/hand/typed/chan-c/data", async () => ++made)); + assert.ok(Date.now() - started < 50); + assert.equal(made, 4); +}); + +test("a probe of another root under a location's id has its own slots and does not rewrite that location", async () => { + registerLocationHealth([USB]); + setDriveCallBudget(50); + // Four candidate probes that never answer: they hold the CANDIDATE root's + // four slots, not the location's. + for (let i = 0; i < 4; i++) { + await assert.rejects( + () => onDrive({ ...USB, root: "/mnt/candidate" }, never), + DriveNotAnsweringError, + ); + } + assert.equal(locationHealth("usb")?.root, "/mnt/usb/media"); + assert.equal(locationHealth("usb")?.state, "ok"); + assert.equal(driveCallsInFlight("usb"), 0); + // The real location's calls run as if nothing happened. + assert.deepEqual( + await Promise.all([1, 2, 3, 4, 5].map((n) => onDrive(USB, async () => n))), + [1, 2, 3, 4, 5], + ); +}); + +test("M1: every slot held by a call given up on — a new call is refused within the budget, after the pass cleared the location", async () => { + registerLocationHealth([USB]); + setDriveCallBudget(80); + const hung = Array.from({ length: 4 }, () => onDrive(USB, never).catch(() => "gave up")); + assert.deepEqual(await Promise.all(hung), Array(4).fill("gave up")); + assert.equal(locationHealth("usb")?.state, "stalled"); + // Two clean answers from the pass (in a reset loop's good moment). + recordLocationHealth(USB, "ok"); + recordLocationHealth(USB, "ok"); + assert.equal(locationHealth("usb")?.state, "ok"); + let made = false; + const started = Date.now(); + await assert.rejects( + () => onDrive(USB, async () => { + made = true; + }), + (err: unknown) => err instanceof DriveNotAnsweringError && err.health?.id === "usb", + ); + assert.ok(Date.now() - started < 80, "refused at once, not after a wait"); + assert.equal(made, false); + // And the location is marked again: none of the four has returned. + assert.equal(locationHealth("usb")?.state, "stalled"); + assert.match(String(locationHealth("usb")?.cause), /4 reads on it have not answered/); +}); + +test("M1: every transition to stalled refuses the waiting calls at once, whoever decided it", async () => { + registerLocationHealth([USB]); + setDriveCallBudget(5_000); + const gates = Array.from({ length: 4 }, () => deferred<number>()); + const inFlight = gates.map((g) => onDrive(USB, () => g.promise)); + const waiting = onDrive(USB, async () => 5).then( + () => "ran", + (err: Error) => err.name, + ); + await new Promise((r) => setImmediate(r)); + const started = Date.now(); + // The pass (not the watchdog) finds the drive stalled. + recordLocationHealth(USB, "stalled"); + assert.equal(await waiting, "DriveNotAnsweringError"); + assert.ok(Date.now() - started < 1_000); + for (const g of gates) g.resolve(0); + await Promise.all(inFlight); +}); + +test("L6: a call that times out while its disk is still completing requests is slow — refused, the location not marked", async () => { + registerLocationHealth([USB]); + noteLocationDetector("usb", "counters", "sdz1"); + let completed = 100; + setCounterReader((device) => { + assert.equal(device, "sdz1"); + completed += 7; + return { completed, inFlight: 2 }; + }); + setDriveCallBudget(60); + // Four slow calls and one waiting behind them. + const slow = Array.from({ length: 4 }, () => + onDrive(USB, never).then( + () => "answered", + (err: Error) => err.message, + ), + ); + const waiter = onDrive(USB, async () => 1).then( + () => "ran", + (err: Error) => err.message, + ); + const outcomes = await Promise.all(slow); + for (const o of outcomes) assert.match(o, /^drive slow/); + assert.equal(locationHealth("usb")?.state, "ok", "slow is not stalled"); + // Nothing on the drive returns: the waiter's wait runs out — refused, still + // without marking. + assert.match(await waiter, /nothing on it answered for 0.06 s while a read waited/); + assert.equal(locationHealth("usb")?.state, "ok"); + // With the counters standing still, the same timeout marks it. + setCounterReader(() => ({ completed: 500, inFlight: 2 })); + resetStorageHealth(); + registerLocationHealth([USB]); + noteLocationDetector("usb", "counters", "sdz1"); + setCounterReader(() => ({ completed: 500, inFlight: 2 })); + await assert.rejects(() => onDrive(USB, never), DriveNotAnsweringError); + assert.equal(locationHealth("usb")?.state, "stalled"); +}); + +test("the default budget is 3 s", async () => { + assert.equal(DRIVE_CALL_BUDGET_MS, 3_000); + registerLocationHealth([USB]); + const started = Date.now(); + await assert.rejects(() => onDrive(USB, never), DriveNotAnsweringError); + const took = Date.now() - started; + assert.ok(took >= 3_000 && took < 4_500, `answered after ${took} ms`); +}); + +test("registering locations creates entries without an answer, and a moved root starts over", () => { + registerLocationHealth([USB], 5); + assert.equal(locationHealth("usb")?.state, "ok"); + recordLocationHealth(USB, "stalled", { now: 6 }); + registerLocationHealth([USB], 7); + assert.equal(locationHealth("usb")?.state, "stalled", "registering is not an answer"); + registerLocationHealth([{ ...USB, root: "/mnt/new" }], 8); + assert.equal(locationHealth("usb")?.state, "ok"); + assert.equal(locationHealth("usb")?.root, "/mnt/new"); +}); + +// ── M4: the wait's deadline follows progress ─────────────────────────────── + +const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms)); + +test("M4: a healthy 64-wide walk of units at half the budget — no refusals, nothing marked", async () => { + registerLocationHealth([USB]); + setDriveCallBudget(100); + let answered = 0; + const started = Date.now(); + const outcomes = await Promise.all( + Array.from({ length: 64 }, () => + onDrive("/mnt/usb/media/ch/data", async () => { + await sleep(50); + answered += 1; + return "ok"; + }).catch((err: Error) => err.message), + ), + ); + // Sixteen rounds of 50 ms: the last call waited about 750 ms, far past the + // 100 ms budget, and was not refused — the drive kept answering. + assert.ok(Date.now() - started >= 700, `took ${Date.now() - started} ms`); + assert.deepEqual(outcomes, Array(64).fill("ok")); + assert.equal(answered, 64); + assert.equal(locationHealth("usb")?.state, "ok"); + assert.equal(driveCallsInFlight("usb"), 0); +}); + +test("M4: the same deep queue on a hand-typed root — no refusals either", async () => { + setDriveCallBudget(100); + const outcomes = await Promise.all( + Array.from({ length: 32 }, () => + onDrive("/hand/typed/ch/data", async () => { + await sleep(50); + return "ok"; + }).catch((err: Error) => err.message), + ), + ); + assert.deepEqual(outcomes, Array(32).fill("ok")); +}); + +test("M4: a queue behind four hung calls is still refused within the budget", async () => { + setDriveCallBudget(100); + // A hand-typed root: nothing marks it, so the refusal is the timeouts'. + const hung = Array.from({ length: 4 }, () => + onDrive("/hand/typed/ch/data", never).catch(() => "gave up"), + ); + const started = Date.now(); + const waiters = Array.from({ length: 6 }, () => + onDrive("/hand/typed/ch/data", async () => "ran").catch((err: Error) => err.name), + ); + assert.deepEqual(await Promise.all(waiters), Array(6).fill("DriveNotAnsweringError")); + assert.ok(Date.now() - started < 400, `refused after ${Date.now() - started} ms`); + await Promise.all(hung); + // And on a configured location, the timeouts mark it and refuse the queue. + registerLocationHealth([USB]); + const hungHere = Array.from({ length: 4 }, () => onDrive(USB, never).catch(() => "gave up")); + const t0 = Date.now(); + const waitersHere = Array.from({ length: 6 }, () => + onDrive(USB, async () => "ran").catch((err: Error) => err.name), + ); + assert.deepEqual(await Promise.all(waitersHere), Array(6).fill("DriveNotAnsweringError")); + assert.ok(Date.now() - t0 < 400); + assert.equal(locationHealth("usb")?.state, "stalled"); + await Promise.all(hungHere); +}); + +test("M4: a call that returns late frees its slot for the calls waiting behind it", async () => { + registerLocationHealth([USB]); + noteLocationDetector("usb", "counters", "sdz1"); + // The counters keep completing, so the late calls are slow, not stalled. + let completed = 0; + setCounterReader(() => ({ completed: (completed += 5), inFlight: 3 })); + setDriveCallBudget(80); + // Four calls that take 150 ms: refused as slow at 80 ms, returning at 150. + const slow = Array.from({ length: 4 }, () => + onDrive(USB, () => sleep(150)).then( + () => "answered", + (err: Error) => err.message, + ), + ); + // Four more queue at 70 ms, deadline 70 + 80 + 20: the returns at 150 ms + // come first and hand them the slots. + await sleep(70); + const behind = Array.from({ length: 4 }, () => + onDrive(USB, async () => "ran").catch((err: Error) => err.message), + ); + for (const o of await Promise.all(slow)) assert.match(o, /^drive slow/); + assert.deepEqual(await Promise.all(behind), Array(4).fill("ran")); + assert.equal(locationHealth("usb")?.state, "ok"); + // The first return can feed all four waiters; let the other three slow + // calls return too, so their releases do not land in the next test's state. + await sleep(120); + assert.equal(driveCallsInFlight("usb"), 0); +}); + +// ── the overdue refusal: slow or stalled, by the counters ────────────────── + +test("every slot held by a slow unit on a disk still completing: the next call is refused, nothing marked", async () => { + registerLocationHealth([USB]); + noteLocationDetector("usb", "counters", "sdz1"); + let completed = 0; + setCounterReader(() => ({ completed: (completed += 3), inFlight: 4 })); + setDriveCallBudget(60); + // Four units slower than the budget (they return long after): each is + // refused as slow at its timeout, and holds its slot, overdue. + const slow = Array.from({ length: 4 }, () => + onDrive(USB, never).catch((err: Error) => err.message), + ); + for (const o of await Promise.all(slow)) assert.match(o, /^drive slow/); + let made = false; + const started = Date.now(); + await assert.rejects( + () => onDrive(USB, async () => { + made = true; + }), + (err: unknown) => + err instanceof DriveNotAnsweringError && + err.health === null && + /^drive slow \(4 reads on it are past 0.06 s/.test(err.message), + ); + assert.ok(Date.now() - started < 60, "refused at once"); + assert.equal(made, false); + assert.equal(locationHealth("usb")?.state, "ok", "slow, not stalled"); +}); + +test("every slot held by an overdue call on a disk completing nothing: the next call is refused and marks it", async () => { + registerLocationHealth([USB]); + noteLocationDetector("usb", "counters", "sdz1"); + setCounterReader(() => ({ completed: 500, inFlight: 4 })); + setDriveCallBudget(60); + const hung = Array.from({ length: 4 }, () => onDrive(USB, never).catch(() => "gave up")); + await Promise.all(hung); + assert.equal(locationHealth("usb")?.state, "stalled"); + // The pass clears it; the four have still not returned. + recordLocationHealth(USB, "ok"); + recordLocationHealth(USB, "ok"); + await assert.rejects( + () => onDrive(USB, async () => 1), + (err: unknown) => err instanceof DriveNotAnsweringError && err.health?.id === "usb", + ); + assert.equal(locationHealth("usb")?.state, "stalled"); + assert.match(String(locationHealth("usb")?.cause), /4 reads on it have not answered/); +}); diff --git a/common/lib/storageHealth.ts b/common/lib/storageHealth.ts @@ -0,0 +1,749 @@ +import { locationOfDataDir, type StorageLocation } from "./storageLocations"; + +// IS A STORAGE LOCATION'S DRIVE ANSWERING RIGHT NOW — the in-memory answer every +// page and poll asks before it touches the drive. +// +// A drive can be mounted and still not answer. An SMR disk in a USB enclosure +// under a long write stalls, the enclosure resets, and every filesystem call +// that has to reach the disk blocks for about 30 seconds. Node runs those calls +// on libuv's thread pool (four threads by default), so four of them block the +// whole editor: no page, no poll and no job log answers until the disk does. +// "Not mounted" does not describe that (a `stat` there does not fail, it hangs), +// and no in-process call can find it out without paying the hang itself. +// +// SO SOMETHING THAT CANNOT HANG ASKS, AND THIS MODULE REMEMBERS WHAT IT SAID. +// Two detectors write here. The health pass (`controller/storageWatch.ts`, +// every 15 s) reads each location's block device counters in /sys, which never +// touch the drive (`detectLocationHealth` in `storageVolumes.ts`; a child `stat` +// of the root raced against 3 s only where no device can be named). And +// `onDrive` below races every in-process call the gate covers against 3 s and +// marks the location the moment one does not answer. Everything that would +// touch the drive in-process asks this state first and, on a stalled location, +// answers without the call: `inspectChannelMedia` reports `stalled`, the +// free-space column reads "—", the probe reads "Not answering", the recency +// layer skips the tail read. +// +// THE RULES, which are what keep a flaky drive from flapping the UI: +// - ONE `stalled` answer marks the location `stalled` at once. A drive that +// did not answer will not answer the next page either, and every page that +// asks costs a thread. +// - TWO consecutive clean answers clear it. A clean answer is anything else: +// the counters moving or idle, or a child `stat` answering in time whether +// the root was there (`ok`) or not (`absent`: an unmounted drive answers +// ENOENT at once, and that is a different problem, which +// `inspectChannelMedia` already reports). One clean answer in the middle of +// a reset loop is not recovery. +// - A ROOT CHANGE (a re-point) starts the location over: the old root's stall +// says nothing about the new one. +// +// IN MEMORY ONLY. A stall is a fact about this minute, not about the corpus; +// persisting it would outlive the reset loop that caused it. A restarted +// process starts with no stall, and the first pass (at arm time, on an idle +// boot too) registers the locations; the watchdog re-learns a stall the moment +// a page reaches the drive. +// +// ONE MAP PER PROCESS, NOT PER MODULE COPY. The watch that probes is armed from +// `editor/instrumentation.ts`, and the pages that read are another bundle +// layer; Next can load this module once for each. The map lives on +// `globalThis`, the house pattern (`lib/jsonFile-server.ts`, `jobs/registry.ts`), +// so the writer and every reader see one map. +// +// PURE of I/O, and deliberately without execa, so `lib/channelMedia.ts` can ask +// it without pulling a subprocess module into everything that imports that. + +// One probe's answer, and a location's state. +export type LocationHealthState = + // The root answered in time and is a directory. + | "ok" + // The root did not answer within the probe's budget. + | "stalled" + // The root answered in time, and is not a directory (not mounted, or gone). + | "absent"; + +// What decided a location's state. `counters`: the block device's own request +// counters (`/sys/class/block/<dev>/stat`), which never touch the drive; +// `stat`: a child `stat` of the root, when no device could be named (a +// container, no findmnt, no /sys entry). A stall marked by `onDrive`'s watchdog +// keeps the detector the last pass used; its `cause` says what did not answer. +export type HealthDetector = "counters" | "stat"; + +export type LocationHealth = { + id: string; + label: string; + root: string; + state: LocationHealthState; + detector?: HealthDetector; + // When the current state began, ms since epoch. + since: number; + // When the last probe (or observation) was recorded. + checkedAt: number; + // Consecutive clean answers since the location was marked stalled. It clears + // at HEALTH_CLEAN_TO_CLEAR. + cleanStreak: number; + // What did not answer, for a stalled location: the probe's own words. + cause?: string; + // The block device the counters detector reads for this location, when it + // could name one. The watchdog reads its counters too (see `onDrive`). + device?: string; +}; + +// The probe's budget. A `stat` of a directory on a healthy disk answers in +// microseconds; three seconds is a thousand times that, and short enough that +// a page asking during a stall has not waited long. +export const HEALTH_PROBE_TIMEOUT_MS = 3_000; +// How often the watch asks. Short, because the gate is only as current as the +// last answer: a stall that began just after a probe costs every page that +// touches the drive until the next one. +export const HEALTH_PROBE_INTERVAL_MS = 15_000; +// Clean answers in a row that clear a stall. +export const HEALTH_CLEAN_TO_CLEAR = 2; + +// The one wording of the state, for every surface that shows it. +export const NOT_ANSWERING = "drive not answering"; + +// A call waiting for a slot. `resolve` answers whether it took the slot (a +// waiter whose own wait already timed out does not); `rearm` restarts its +// deadline, which a call returning on its key does (see `acquireSlot`). +type Waiter = { + resolve: () => boolean; + reject: (err: Error) => void; + rearm: () => void; +}; + +// A call the watchdog gave up on that has not returned yet: when it began, +// and the device's counters then (null when the counters detector has named +// no device for its location). +type OverdueCall = { startedAt: number; before: BlockStatSample | null }; + +type HealthState = { + byId: Map<string, LocationHealth>; + // `onDrive`'s bookkeeping, by slot key (see `resolveWhere`): calls in + // flight, the ones among them the watchdog has already given up on, and + // the calls waiting for a slot. + inFlight?: Map<string, number>; + overdue?: Map<string, OverdueCall[]>; + waiters?: Map<string, Waiter[]>; + // Test seam: the watchdog's budget. + budgetMs?: number; + // Reads a block device's counters, synchronously and without touching the + // drive (set by `storageVolumes.ts`, which owns /sys). Absent: no check. + readCounters?: (device: string) => BlockStatSample | null; +}; + +declare global { + // eslint-disable-next-line no-var + var __yttStorageHealth__: HealthState | undefined; +} + +type FilledState = HealthState & + Required<Pick<HealthState, "inFlight" | "overdue" | "waiters">>; + +function healthState(): FilledState { + if (!globalThis.__yttStorageHealth__) { + globalThis.__yttStorageHealth__ = { byId: new Map() }; + } + const s = globalThis.__yttStorageHealth__; + // Filled lazily: a dev server's hot reload keeps an object made by an older + // copy of this module. + s.inFlight ??= new Map(); + s.overdue ??= new Map(); + s.waiters ??= new Map(); + return s as FilledState; +} + +// Test seam, and the escape hatch for a process that wants to forget. Calls +// still waiting for a slot are released to run. +export function resetStorageHealth(): void { + const s = healthState(); + s.byId.clear(); + s.inFlight.clear(); + s.overdue.clear(); + for (const q of s.waiters.values()) for (const w of q) w.resolve(); + s.waiters.clear(); +} + +// `storageVolumes.ts` registers its /sys reader here (see `onDrive`, L6 in +// the record: a call that times out while its disk is still completing +// requests is slow, not stalled). A test passes its own, or undefined. +export function setCounterReader( + read: ((device: string) => BlockStatSample | null) | undefined, +): void { + healthState().readCounters = read; +} + +export type HealthTransition = { + id: string; + from: LocationHealthState | null; + to: LocationHealthState; +}; + +// Record one answer about one location. Returns the transition when the state +// changed, null when it did not. +export function recordLocationHealth( + loc: Pick<StorageLocation, "id" | "label" | "root">, + answer: LocationHealthState, + opts: { + now?: number; + cause?: string; + detector?: HealthDetector; + // The counters' device; `null` forgets it (the stat detector answered). + device?: string | null; + } = {}, +): HealthTransition | null { + const t = recordAnswer(loc, answer, opts); + // EVERY TRANSITION TO STALLED refuses the calls waiting for a slot on the + // location, whoever decided it (the pass, the watchdog, a Refresh): they + // would otherwise wait for a slot held by a call the drive is not answering. + if (t?.to === "stalled") { + refuseWaiters(slotKeyOfLocation(loc.id), stalledLocation(loc)); + } + return t; +} + +function recordAnswer( + loc: Pick<StorageLocation, "id" | "label" | "root">, + answer: LocationHealthState, + opts: { + now?: number; + cause?: string; + detector?: HealthDetector; + device?: string | null; + }, +): HealthTransition | null { + const now = opts.now ?? Date.now(); + const map = healthState().byId; + const prev = map.get(loc.id); + const label = loc.label || loc.id; + // First sighting, or the root moved under the same id: start over. + if (!prev || prev.root !== loc.root) { + map.set(loc.id, { + id: loc.id, + label, + root: loc.root, + state: answer, + ...(opts.detector ? { detector: opts.detector } : {}), + ...(opts.device ? { device: opts.device } : {}), + since: now, + checkedAt: now, + cleanStreak: 0, + ...(answer === "stalled" && opts.cause ? { cause: opts.cause } : {}), + }); + return { id: loc.id, from: null, to: answer }; + } + prev.label = label; + prev.checkedAt = now; + if (opts.detector) prev.detector = opts.detector; + if (opts.device === null) delete prev.device; + else if (opts.device) prev.device = opts.device; + if (answer === "stalled") { + prev.cleanStreak = 0; + if (prev.state === "stalled") return null; + const from = prev.state; + prev.state = "stalled"; + prev.since = now; + if (opts.cause) prev.cause = opts.cause; + return { id: loc.id, from, to: "stalled" }; + } + if (prev.state === "stalled") { + prev.cleanStreak += 1; + if (prev.cleanStreak < HEALTH_CLEAN_TO_CLEAR) return null; + prev.state = answer; + prev.since = now; + prev.cleanStreak = 0; + delete prev.cause; + return { id: loc.id, from: "stalled", to: answer }; + } + if (prev.state === answer) return null; + const from = prev.state; + prev.state = answer; + prev.since = now; + return { id: loc.id, from, to: answer }; +} + +// Make sure every configured location has an entry, without recording an +// answer: a new location starts `ok`, and a known one whose root moved starts +// over. The watchdog below can only mark a location it can find, and it finds +// a channel's location among these entries. +export function registerLocationHealth( + locations: ReadonlyArray<Pick<StorageLocation, "id" | "label" | "root">>, + now: number = Date.now(), +): void { + const map = healthState().byId; + for (const loc of locations) { + const prev = map.get(loc.id); + if (prev && prev.root === loc.root) { + prev.label = loc.label || loc.id; + continue; + } + recordLocationHealth(loc, "ok", { now }); + } +} + +// Which detector decided a location's last answer, when it gave none (the +// counters' first sample has nothing to compare with). +export function noteLocationDetector( + id: string, + detector: HealthDetector, + device?: string | null, +): void { + const h = healthState().byId.get(id); + if (!h) return; + h.detector = detector; + if (device === null) delete h.device; + else if (device) h.device = device; +} + +// Drop every location that is no longer configured. +export function pruneLocationHealth(liveIds: Iterable<string>): void { + const keep = new Set(liveIds); + const map = healthState().byId; + for (const id of [...map.keys()]) if (!keep.has(id)) map.delete(id); +} + +export function locationHealth(id: string): LocationHealth | undefined { + const h = healthState().byId.get(id); + return h ? { ...h } : undefined; +} + +// Every location's last answer, by id. A copy: the caller may keep it. +export function allLocationHealth(): Record<string, LocationHealth> { + const out: Record<string, LocationHealth> = {}; + for (const [id, h] of healthState().byId) out[id] = { ...h }; + return out; +} + +// THE GATE. The stalled location `p` is on, or null. `p` is a channel's +// `dataDir` (or anything under a location's root); the match is the one +// `locationOfDataDir` makes, longest root first, so a channel on a nested +// location answers for that location and not its parent. +export function stalledLocationForPath(p: string): LocationHealth | null { + const map = healthState().byId; + if (map.size === 0 || !p) return null; + const entries = [...map.values()]; + const loc = locationOfDataDir( + p, + entries.map((h) => ({ id: h.id, label: h.label, root: h.root, autoRepoint: false })), + ); + if (!loc) return null; + const h = map.get(loc.id); + return h && h.state === "stalled" ? { ...h } : null; +} + +// The location with this id, when it is stalled AND still at this root. The +// gate for a caller that holds a location rather than a channel path +// (`probeLocation`, `volumeFreeBytes`). +export function stalledLocation( + loc: Pick<StorageLocation, "id" | "root">, +): LocationHealth | null { + const h = healthState().byId.get(loc.id); + return h && h.state === "stalled" && h.root === loc.root ? { ...h } : null; +} + +// "since 11:35" in the server's clock, or "since 2026-09-28 11:35" when the +// stall began on another day than `now`. +export function sinceText(since: number, now: number = Date.now()): string { + const d = new Date(since); + const n = new Date(now); + const pad = (x: number) => String(x).padStart(2, "0"); + const hm = `${pad(d.getHours())}:${pad(d.getMinutes())}`; + const sameDay = + d.getFullYear() === n.getFullYear() && + d.getMonth() === n.getMonth() && + d.getDate() === n.getDate(); + return sameDay + ? `since ${hm}` + : `since ${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())} ${hm}`; +} + +// "not answering since 11:35" — the one line /storage, the /channels volume bar +// and the channel's Storage panel show for a stalled location. +export function notAnsweringText(h: Pick<LocationHealth, "since">, now?: number): string { + return `not answering ${sinceText(h.since, now)}`; +} + +// --------------------------------------------------------------------------- +// The watchdog: every call the gate covers, raced against 3 s +// --------------------------------------------------------------------------- +// +// THE DETECTOR THAT CANNOT BE FOOLED BY A CACHE. The 15 s pass reads the +// block device's counters (or, with no device, a child `stat`), and a stall +// that starts between two passes is not seen by it. A page or a poll that +// actually reaches the drive is: `onDrive` runs the call against a 3 s timer, +// and a call that has not answered by then marks its location `stalled` at +// once (since now) and throws `DriveNotAnsweringError`. The call itself is left +// to settle on its own: its thread is held until the drive answers, which is +// the stated limit. +// +// THE BUDGET COVERS A WHOLE UNIT OF WORK. A caller sends a unit through as one +// call (a video directory's few reads, a page's reads of one video), so a slow +// drive that is still answering can be marked by one long unit. Unless its +// disk is visibly still completing requests: on a timeout, when the counters +// detector has named the location's device, its counters are read (from /sys, +// never the drive) and compared with a reading taken when the call began; if +// requests completed meanwhile the drive is slow, not stalled, and only this +// call is refused. +// +// AT MOST DRIVE_CALLS_IN_FLIGHT CALLS PER SLOT KEY ARE IN FLIGHT. A walk of +// `data/` fans out 64 wide, and a stall mid-walk would otherwise put all 64 in +// libuv's queue before the watchdog fired. The rest wait in a queue of our own: +// - a transition to `stalled` (from any detector) refuses them at once; +// - a wait is refused (without marking anything) only when no call on its key +// has returned for the budget plus a small grace: every call that returns +// restarts every waiter's deadline, so a deep queue on a drive that is busy +// but answering waits as long as it takes; +// - when every slot is held by a call the watchdog already gave up on, a new +// call is refused at once, and the location marked stalled again unless +// the disk has been completing requests since the oldest of those calls +// began (slow, not stalled — the same test as a timeout's): none of those +// calls has returned, whatever the last pass said. +// A slot is released when its call really returns, not when the watchdog gave +// up on it. So on one location at most four threads wait on its drive for the +// calls that come through here — every page and poll path, and the snapshot +// walk. A job's own reads that do not come through here are not capped. +// +// THE SLOT KEY: a configured location's id; for a probe of another root under a +// location's id, that root; for a path on no configured location (a root typed +// by hand), the root the path is under (`<root>/<slug>/data` → `<root>`). Only a +// configured location can be marked. +// +// Do not nest `onDrive` for one key: the inner call would wait for a slot the +// outer one holds. + +export const DRIVE_CALL_BUDGET_MS = 3_000; +export const DRIVE_CALLS_IN_FLIGHT = 4; + +// Test seam: shorten (or restore, with no argument) the watchdog's budget. +export function setDriveCallBudget(ms?: number): void { + healthState().budgetMs = ms; +} + +function driveCallBudget(): number { + return healthState().budgetMs ?? DRIVE_CALL_BUDGET_MS; +} + +export class DriveNotAnsweringError extends Error { + // The stalled location, when a known one was marked. + readonly health: LocationHealth | null; + constructor(health: LocationHealth | null, detail?: string) { + super( + detail ?? + (health + ? `${NOT_ANSWERING} (location "${health.label}", ${sinceText(health.since)})` + : `${NOT_ANSWERING} (a read did not answer within ${driveCallBudget() / 1000} s)`), + ); + this.name = "DriveNotAnsweringError"; + this.health = health; + } +} + +export function isDriveNotAnswering(err: unknown): err is DriveNotAnsweringError { + return err instanceof Error && err.name === "DriveNotAnsweringError"; +} + +type Where = string | Pick<StorageLocation, "id" | "label" | "root">; + +type Resolved = { + key: string; + // The location to mark, when marking is right. + loc: Pick<StorageLocation, "id" | "label" | "root"> | null; +}; + +function slotKeyOfLocation(id: string): string { + return `loc:${id}`; +} + +// The root a path on no configured location is under: `<root>/<slug>/data` +// (relocatedDataDir's shape) gives `<root>`; anything else is its own key. +function rootOfUnknownPath(p: string): string { + const clean = p.replace(/\/+$/, ""); + const parts = clean.split("/"); + return parts.length > 2 && parts[parts.length - 1] === "data" + ? parts.slice(0, -2).join("/") || "/" + : clean; +} + +function resolveWhere(where: Where): Resolved { + const map = healthState().byId; + if (typeof where !== "string") { + const entry = map.get(where.id); + // A probe of a candidate root under a location's id is its own key, and + // marks nothing: racing it is right, rewriting that location is not. + if (entry && entry.root !== where.root) { + return { key: `root:${where.root}`, loc: null }; + } + return { key: slotKeyOfLocation(where.id), loc: where }; + } + const found = + map.size > 0 && where + ? locationOfDataDir( + where, + [...map.values()].map((h) => ({ + id: h.id, + label: h.label, + root: h.root, + autoRepoint: false, + })), + ) + : null; + if (found) return { key: slotKeyOfLocation(found.id), loc: found }; + return { key: `root:${rootOfUnknownPath(where)}`, loc: null }; +} + +function refuseWaiters(key: string, health: LocationHealth | null): void { + const s = healthState(); + const q = s.waiters.get(key) ?? []; + s.waiters.delete(key); + for (const w of q) w.reject(new DriveNotAnsweringError(health)); +} + +// Take a slot on `key`, or wait for one (raced against the budget), or be +// refused at once when every slot is held by an overdue call. +async function acquireSlot(r: Resolved): Promise<void> { + const s = healthState(); + const n = s.inFlight.get(r.key) ?? 0; + if (n < DRIVE_CALLS_IN_FLIGHT) { + s.inFlight.set(r.key, n + 1); + return; + } + const overdue = s.overdue.get(r.key) ?? []; + if (overdue.length >= DRIVE_CALLS_IN_FLIGHT) { + // EVERY SLOT IS HELD BY A CALL THE WATCHDOG GAVE UP ON: refused at once. + // Marked stalled only when the disk has not been completing requests + // since the oldest of them began — the watchdog's own slow-or-stalled test + // (see `onDrive`): four slow reads on a busy disk are not a stall. + const oldest = overdue.reduce((a, b) => (b.startedAt < a.startedAt ? b : a)); + const now = readDeviceCounters(r.loc); + if (oldest.before && now && now.completed > oldest.before.completed) { + throw new DriveNotAnsweringError( + null, + `drive slow (${DRIVE_CALLS_IN_FLIGHT} reads on it are past ` + + `${driveCallBudget() / 1000} s, while its disk is still completing others)`, + ); + } + let health: LocationHealth | null = null; + if (r.loc) { + recordLocationHealth(r.loc, "stalled", { + cause: `${DRIVE_CALLS_IN_FLIGHT} reads on it have not answered`, + }); + health = stalledLocation(r.loc); + } + throw new DriveNotAnsweringError(health); + } + const budget = driveCallBudget(); + // THE WAIT'S DEADLINE FOLLOWS PROGRESS, NOT THE QUEUE. A waiter is refused + // only when NO call on its key has returned for a full budget — "nothing on + // this drive answered for 3 s", the condition the watchdog exists for. + // Every call that returns on the key (in time or late) re-arms every waiter + // behind it, so a deep queue on a drive that is busy but answering (a + // 64-wide walk of units that each take a second) is never refused for its + // depth alone. Timed from when the call queued, it was: a waiter at depth d + // waits about d/4 units, whatever the drive is doing. + // + // PLUS A SMALL GRACE. A waiter's timer is created when it queues — before + // the race timers of calls that took their slots in the same tick — so with + // equal deadlines the waiters would give up a moment before the calls they + // wait behind, and be refused without the mark those calls are about to + // make. The grace lets them time out first. + const grace = Math.min(250, Math.round(budget / 4)); + await new Promise<void>((resolve, reject) => { + let settled = false; + let timer: ReturnType<typeof setTimeout> | undefined; + const giveUp = () => { + if (settled) return; + settled = true; + const q = s.waiters.get(r.key); + if (q) { + const i = q.indexOf(waiter); + if (i >= 0) q.splice(i, 1); + } + reject( + new DriveNotAnsweringError( + r.loc ? stalledLocation(r.loc) : null, + `${NOT_ANSWERING} (nothing on it answered for ${budget / 1000} s while a read waited)`, + ), + ); + }; + const arm = () => { + if (timer) clearTimeout(timer); + timer = setTimeout(giveUp, budget + grace); + }; + const waiter: Waiter = { + resolve: () => { + if (settled) return false; + settled = true; + if (timer) clearTimeout(timer); + resolve(); + return true; + }, + reject: (err) => { + if (settled) return; + settled = true; + if (timer) clearTimeout(timer); + reject(err); + }, + rearm: () => { + if (!settled) arm(); + }, + }; + arm(); + const q = s.waiters.get(r.key) ?? []; + q.push(waiter); + s.waiters.set(r.key, q); + }); + // Resolved by a release that handed its slot over: the count is unchanged. +} + +// A slot is released when its call returns (or, on a refusal after a wait, +// without a call). A return is progress: every waiter on the key has its +// deadline restarted, then the slot goes to the first waiter still waiting. +function releaseSlot(key: string): void { + const s = healthState(); + const q = s.waiters.get(key); + if (q) for (const w of q) w.rearm(); + while (q && q.length > 0) { + const next = q.shift() as Waiter; + if (next.resolve()) return; + } + s.inFlight.set(key, Math.max(0, (s.inFlight.get(key) ?? 1) - 1)); +} + +// How many calls are in flight on a location through `onDrive` (for tests and +// for a reader that wants to say so). +export function driveCallsInFlight(id: string): number { + return healthState().inFlight.get(slotKeyOfLocation(id)) ?? 0; +} + +function readDeviceCounters(loc: Resolved["loc"]): BlockStatSample | null { + if (!loc) return null; + const s = healthState(); + const device = s.byId.get(loc.id)?.device; + if (!device || !s.readCounters) return null; + try { + return s.readCounters(device); + } catch { + return null; + } +} + +// Run `call` against the drive `where` is on: refused at once when that +// location is stalled, queued behind DRIVE_CALLS_IN_FLIGHT calls already in +// flight on its key, and raced against the budget. Throws +// DriveNotAnsweringError for a refusal or a timeout; any other error is the +// call's own. +export async function onDrive<T>(where: Where, call: () => Promise<T>): Promise<T> { + const r = resolveWhere(where); + const refused = r.loc ? stalledLocation(r.loc) : null; + if (refused) throw new DriveNotAnsweringError(refused); + await acquireSlot(r); + // Stalled while this call waited: refused, and the slot passed on. + const late = r.loc ? stalledLocation(r.loc) : null; + if (late) { + releaseSlot(r.key); + throw new DriveNotAnsweringError(late); + } + const s = healthState(); + let released = false; + const release = () => { + if (released) return; + released = true; + releaseSlot(r.key); + }; + const startedAt = Date.now(); + const before = readDeviceCounters(r.loc); + let pending: Promise<T>; + try { + pending = call(); + } catch (err) { + release(); + throw err; + } + const TIMED_OUT = Symbol("timed out"); + let timer: ReturnType<typeof setTimeout> | undefined; + // Not unref'd: it is cleared the moment the call answers, and while the call + // is outstanding the timer is what must fire. + const timeout = new Promise<typeof TIMED_OUT>((resolve) => { + timer = setTimeout(() => resolve(TIMED_OUT), driveCallBudget()); + }); + let answer: T | typeof TIMED_OUT; + try { + answer = await Promise.race([pending, timeout]); + } catch (err) { + release(); + throw err; + } finally { + if (timer) clearTimeout(timer); + } + if (answer !== TIMED_OUT) { + release(); + return answer; + } + // The call is still waiting on the drive. Its slot stays held, and counted + // overdue, until it returns; nothing awaits it. + const entry: OverdueCall = { startedAt, before }; + s.overdue.set(r.key, [...(s.overdue.get(r.key) ?? []), entry]); + const done = () => { + const left = (s.overdue.get(r.key) ?? []).filter((e) => e !== entry); + if (left.length > 0) s.overdue.set(r.key, left); + else s.overdue.delete(r.key); + release(); + }; + pending.then(done, done); + const after = readDeviceCounters(r.loc); + if (before && after && after.completed > before.completed) { + // SLOW, NOT STALLED: the disk completed requests while this call waited. + throw new DriveNotAnsweringError( + null, + `drive slow (a read did not answer within ${driveCallBudget() / 1000} s, ` + + `while its disk was still completing others)`, + ); + } + if (r.loc) { + recordLocationHealth(r.loc, "stalled", { + cause: `a read in the editor did not answer within ${driveCallBudget() / 1000} s`, + }); + throw new DriveNotAnsweringError(stalledLocation(r.loc)); + } + refuseWaiters(r.key, null); + throw new DriveNotAnsweringError(null); +} + +// --------------------------------------------------------------------------- +// The block device's counters +// --------------------------------------------------------------------------- +// +// `/sys/class/block/<dev>/stat` is the kernel's own count of a device's +// requests (Documentation/block/stat.rst): field 1 reads completed, 5 writes +// completed, 9 requests in flight now. Reading it never touches the drive. A +// drive that is merely slow, even one grinding through a long write, keeps +// completing requests; one in a reset loop has requests in flight and +// completes none. So, between two samples a pass apart: +// +// stalled ⇔ in flight at both samples AND nothing completed between (reads, +// writes, discards, flushes) +// ok ⇔ anything else (nothing in flight at one of them, or completions +// moved) +// +// and the health rules above turn one `stalled` into a stall and two `ok`s in +// a row into its end. + +export type BlockStatSample = { completed: number; inFlight: number }; + +// Completed: reads (field 1) + writes (5), and, on kernels that count them, +// discards (12) and flushes (16) — in_flight counts those too, so a long flush +// alone (an SMR drive emptying its media cache) must not read as nothing +// completing. +export function parseBlockStat(line: string): BlockStatSample | null { + const f = line.trim().split(/\s+/).map(Number); + if (f.length < 9 || f.slice(0, 9).some((n) => !Number.isFinite(n))) return null; + const opt = (i: number) => (Number.isFinite(f[i]) ? f[i] : 0); + return { completed: f[0] + f[4] + opt(11) + opt(15), inFlight: f[8] }; +} + +export function countersVerdict( + prev: BlockStatSample, + cur: BlockStatSample, +): "ok" | "stalled" { + return prev.inFlight > 0 && cur.inFlight > 0 && cur.completed === prev.completed + ? "stalled" + : "ok"; +} diff --git a/common/lib/storageHealthCounters.test.ts b/common/lib/storageHealthCounters.test.ts @@ -0,0 +1,231 @@ +import { beforeEach, test } from "node:test"; +import assert from "node:assert/strict"; +import path from "node:path"; +import { tmpdir } from "node:os"; +import { chmod, mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; +import { + blockDeviceName, + detectLocationHealth, + MIN_COUNTER_INTERVAL_MS, + resetHealthDetector, +} from "./storageVolumes"; +import { countersVerdict, parseBlockStat } from "./storageHealth"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/storageHealthCounters.test.ts +// +// THE COUNTERS DETECTOR. No test stalls a real drive: findmnt is a fake that +// names a device, and /sys/class/block is a temp directory whose `stat` files +// the test writes — a stalled device is one whose in-flight count stays up +// while its completions stand still. + +beforeEach(() => resetHealthDetector()); + +// A real line from this machine's /sys/class/block/<dev>/stat (17 fields). +const LINE = + "368126412 135952867 36463191157 80435430 29670281 46781260 4001255365 167469727 0 44151394 253218297 877589 0 861936568 4100450 578669 1212688"; + +test("the stat line: reads + writes (+ discards + flushes) completed, and requests in flight", () => { + // Fields 1, 5, 12 and 16 completed; field 9 in flight. + assert.deepEqual(parseBlockStat(LINE), { + completed: 368126412 + 29670281 + 877589 + 578669, + inFlight: 0, + }); + // A long flush alone moves completions: not a stall. + const before = parseBlockStat("10 0 0 0 5 0 0 0 1 0 0 0 0 0 0 7 0"); + const after = parseBlockStat("10 0 0 0 5 0 0 0 1 0 0 0 0 0 0 8 0"); + assert.equal(countersVerdict(before!, after!), "ok"); + // An 11-field line from an older kernel parses the same way. + assert.deepEqual(parseBlockStat("10 0 0 0 5 0 0 0 3 0 0\n"), { completed: 15, inFlight: 3 }); + assert.equal(parseBlockStat(""), null); + assert.equal(parseBlockStat("1 2 3"), null); + assert.equal(parseBlockStat("a b c d e f g h i"), null); +}); + +test("the verdict over a pair of samples", () => { + const s = (completed: number, inFlight: number) => ({ completed, inFlight }); + // In flight at both, nothing completed between: stalled. + assert.equal(countersVerdict(s(100, 2), s(100, 5)), "stalled"); + // Completions moved: slow, perhaps, but answering. + assert.equal(countersVerdict(s(100, 2), s(101, 2)), "ok"); + // Nothing in flight at either end: idle. + assert.equal(countersVerdict(s(100, 0), s(100, 0)), "ok"); + assert.equal(countersVerdict(s(100, 0), s(100, 4)), "ok"); + assert.equal(countersVerdict(s(100, 4), s(100, 0)), "ok"); +}); + +test("a mount source's device name: a partition, a bind or subvolume suffix, nothing that is not /dev", async () => { + assert.equal(await blockDeviceName("/dev/no-such-disk1"), "no-such-disk1"); + assert.equal(await blockDeviceName("/dev/no-such-disk2[/@home]"), "no-such-disk2"); + assert.equal(await blockDeviceName("tmpfs"), null); + assert.equal(await blockDeviceName("server:/export"), null); + assert.equal(await blockDeviceName(""), null); +}); + +// ── the detector end to end, with a fake findmnt and a fake /sys ──────────── + +const FAKE_FINDMNT = `#!/usr/bin/env node +import { readFileSync } from "node:fs"; +import path from "node:path"; +const c = JSON.parse(readFileSync(path.join(import.meta.dirname, "control.json"), "utf8")); +if (c.sleepMs) Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, c.sleepMs); +if (c.exit) process.exit(c.exit); +process.stdout.write(JSON.stringify({ filesystems: [{ source: c.source, uuid: c.uuid ?? null }] }) + "\\n"); +`; + +type H = { + root: string; + bins: { findmntBin: string }; + sys: string; + control: (c: Record<string, unknown>) => Promise<void>; + counters: (device: string, completed: number, inFlight: number) => Promise<void>; +}; + +async function withHarness(fn: (h: H) => Promise<void>): Promise<void> { + const dir = await mkdtemp(path.join(tmpdir(), "ttb-counters-")); + try { + const bin = path.join(dir, "fake-findmnt.mjs"); + await writeFile(bin, FAKE_FINDMNT); + await chmod(bin, 0o755); + const root = path.join(dir, "media"); + await mkdir(root); + const sys = path.join(dir, "sys-block"); + await fn({ + root, + bins: { findmntBin: bin }, + sys, + control: (c) => writeFile(path.join(dir, "control.json"), JSON.stringify(c)), + counters: async (device, completed, inFlight) => { + await mkdir(path.join(sys, device), { recursive: true }); + // Reads `completed`, writes 0, in flight `inFlight`, the rest zero. + await writeFile( + path.join(sys, device, "stat"), + `${completed} 0 0 0 0 0 0 0 ${inFlight} 0 0 0 0 0 0 0 0\n`, + ); + }, + }); + } finally { + await rm(dir, { recursive: true, force: true }); + } +} + +const loc = (root: string, uuid?: string) => ({ + id: "usb", + root, + ...(uuid ? { volume: { uuid, mountpoint: root, relPath: "" } } : {}), +}); + +test("counters: first sample no verdict; stuck → stalled; two clean samples; each pass apart", async () => { + await withHarness(async (h) => { + await h.control({ source: "/dev/fakedisk1", uuid: "u-1" }); + const at = (i: number) => ({ sysBlockDir: h.sys, now: i * 15_000 }); + await h.counters("fakedisk1", 100, 2); + assert.deepEqual(await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(1)), { + answer: null, + detector: "counters", + device: "fakedisk1", + }); + // Still two in flight, nothing completed in 15 s. + await h.counters("fakedisk1", 100, 2); + const stuck = await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(2)); + assert.equal(stuck.answer, "stalled"); + assert.equal(stuck.detector, "counters"); + assert.match(String(stuck.cause), /^its disk \(fakedisk1\) had 2 request\(s\) in flight and completed none in 15 s$/); + // It drains: nothing in flight. + await h.counters("fakedisk1", 140, 0); + assert.equal((await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(3))).answer, "ok"); + // Busy and moving: still ok. + await h.counters("fakedisk1", 190, 3); + assert.equal((await detectLocationHealth(loc(h.root, "u-1"), h.bins, at(4))).answer, "ok"); + }); +}); + +test("counters: a second sample sooner than the minimum interval gives no verdict and keeps the first", async () => { + await withHarness(async (h) => { + await h.control({ source: "/dev/fakedisk1" }); + await h.counters("fakedisk1", 100, 2); + await detectLocationHealth(loc(h.root), h.bins, { sysBlockDir: h.sys, now: 0 }); + const soon = await detectLocationHealth(loc(h.root), h.bins, { + sysBlockDir: h.sys, + now: MIN_COUNTER_INTERVAL_MS - 1, + }); + assert.equal(soon.answer, null); + // Compared with the FIRST sample, not the refused one. + const later = await detectLocationHealth(loc(h.root), h.bins, { + sysBlockDir: h.sys, + now: 15_000, + }); + assert.equal(later.answer, "stalled"); + }); +}); + +test("no device → the child stat, and the verdict says so", async () => { + await withHarness(async (h) => { + const opts = { sysBlockDir: h.sys, now: 0 }; + // A tmpfs or network source names no block device. + await h.control({ source: "tmpfs" }); + assert.deepEqual(await detectLocationHealth(loc(h.root), h.bins, opts), { + answer: "ok", + detector: "stat", + }); + // findmnt cannot say (no such path, a container without it). + await h.control({ exit: 1 }); + assert.equal((await detectLocationHealth(loc(h.root), h.bins, opts)).detector, "stat"); + assert.equal( + (await detectLocationHealth(loc(path.join(h.root, "gone")), h.bins, opts)).answer, + "absent", + ); + // A device with no /sys entry. + await h.control({ source: "/dev/nodisk9" }); + assert.equal((await detectLocationHealth(loc(h.root), h.bins, opts)).detector, "stat"); + // A root whose filesystem is not the location's recorded volume. + await h.counters("fakedisk1", 1, 0); + await h.control({ source: "/dev/fakedisk1", uuid: "someone-else" }); + assert.equal( + (await detectLocationHealth(loc(h.root, "u-1"), h.bins, opts)).detector, + "stat", + ); + // No findmnt binary at all. + assert.equal( + ( + await detectLocationHealth( + loc(h.root), + { findmntBin: path.join(h.root, "no-findmnt") }, + opts, + ) + ).detector, + "stat", + ); + }); +}); + +test("a device already named is read without findmnt; findmnt again only when its /sys entry stops reading", async () => { + await withHarness(async (h) => { + await h.control({ source: "/dev/fakedisk1" }); + await h.counters("fakedisk1", 100, 2); + await detectLocationHealth(loc(h.root), h.bins, { sysBlockDir: h.sys, now: 0 }); + // findmnt now hangs: it is not asked, the known device is read. + await h.control({ sleepMs: 5_000, source: "/dev/fakedisk1" }); + const started = Date.now(); + const v = await detectLocationHealth(loc(h.root), h.bins, { + sysBlockDir: h.sys, + now: 15_000, + timeoutMs: 300, + }); + assert.ok(Date.now() - started < 250, "no findmnt was run"); + assert.equal(v.detector, "counters"); + assert.equal(v.answer, "stalled"); + // The device is gone from /sys (replugged under another name): findmnt is + // asked again — here it does not answer, so no device, and the child stat. + await rm(path.join(h.sys, "fakedisk1"), { recursive: true }); + const again = await detectLocationHealth(loc(h.root), h.bins, { + sysBlockDir: h.sys, + now: 30_000, + timeoutMs: 300, + }); + assert.equal(again.detector, "stat"); + // The detector's samples are on globalThis (the pass and a Refresh run in + // different module copies and compare against one previous sample). + assert.ok(globalThis.__yttHealthDetector__); + }); +}); diff --git a/common/lib/storageHealthProbe.test.ts b/common/lib/storageHealthProbe.test.ts @@ -0,0 +1,112 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import path from "node:path"; +import { tmpdir } from "node:os"; +import { chmod, mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; +import { probeLocationHealth } from "./storageVolumes"; +import { HEALTH_PROBE_TIMEOUT_MS } from "./storageHealth"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/storageHealthProbe.test.ts +// +// THE HEALTH PROBE: `stat` on a location's root as a CHILD PROCESS, raced +// against a timer. No test stalls a real drive: a stalled `stat` is a fake +// binary that never answers (a child blocked in the kernel looks the same from +// here — it does not exit), and the case that matters is that the answer +// arrives on the timer while the child is still running. + +async function withDir(fn: (dir: string) => Promise<void>): Promise<void> { + const dir = await mkdtemp(path.join(tmpdir(), "ttb-health-")); + try { + await fn(dir); + } finally { + await rm(dir, { recursive: true, force: true }); + } +} + +// A fake `stat`: sleeps `sleepMs` (a blocking sleep, so it is simply not +// answering), then prints `out` and exits `code`. +async function fakeStat( + dir: string, + opts: { sleepMs?: number; out?: string; code?: number }, +): Promise<string> { + const bin = path.join(dir, "fake-stat.mjs"); + await writeFile( + bin, + `#!/usr/bin/env node +if (${opts.sleepMs ?? 0} > 0) { + Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ${opts.sleepMs ?? 0}); +} +process.stdout.write(${JSON.stringify(opts.out ?? "")} + "\\n"); +process.exit(${opts.code ?? 0}); +`, + ); + await chmod(bin, 0o755); + return bin; +} + +test("the real stat: a directory is ok, a missing path and a file are absent", async () => { + await withDir(async (dir) => { + const root = path.join(dir, "media"); + await mkdir(root); + await writeFile(path.join(dir, "a-file"), "x"); + assert.equal(await probeLocationHealth({ root }), "ok"); + assert.equal(await probeLocationHealth({ root: path.join(dir, "gone") }), "absent"); + assert.equal(await probeLocationHealth({ root: path.join(dir, "a-file") }), "absent"); + assert.equal(await probeLocationHealth({ root: " " }), "absent"); + }); +}); + +test("a child that never answers is 'stalled' on the timer, without waiting for it", async () => { + await withDir(async (dir) => { + // Twenty seconds asleep: if the probe waited for the child, this test would + // take that long. + const statBin = await fakeStat(dir, { sleepMs: 20_000, out: "directory" }); + const started = Date.now(); + const answer = await probeLocationHealth( + { root: dir }, + { statBin, timeoutMs: 400 }, + ); + const took = Date.now() - started; + assert.equal(answer, "stalled"); + assert.ok(took >= 400, `answered after ${took} ms, before the timer`); + assert.ok(took < 3_000, `answered after ${took} ms`); + }); +}); + +test("the default budget is 3 s", async () => { + assert.equal(HEALTH_PROBE_TIMEOUT_MS, 3_000); + await withDir(async (dir) => { + const statBin = await fakeStat(dir, { sleepMs: 20_000, out: "directory" }); + const started = Date.now(); + assert.equal(await probeLocationHealth({ root: dir }, { statBin }), "stalled"); + const took = Date.now() - started; + assert.ok(took >= 3_000 && took < 6_000, `answered after ${took} ms`); + }); +}); + +test("an answer inside the budget is taken as given", async () => { + await withDir(async (dir) => { + const slowDir = await fakeStat(dir, { sleepMs: 100, out: "directory" }); + assert.equal( + await probeLocationHealth({ root: dir }, { statBin: slowDir, timeoutMs: 2_500 }), + "ok", + ); + const aFile = await fakeStat(dir, { out: "regular file" }); + assert.equal(await probeLocationHealth({ root: dir }, { statBin: aFile }), "absent"); + const missing = await fakeStat(dir, { code: 1 }); + assert.equal(await probeLocationHealth({ root: dir }, { statBin: missing }), "absent"); + }); +}); + +test("a stat that cannot be started fails open: ok, never stalled", async () => { + await withDir(async (dir) => { + assert.equal( + await probeLocationHealth( + { root: dir }, + { statBin: path.join(dir, "no-such-stat") }, + ), + "ok", + ); + }); +}); diff --git a/common/lib/storageVolumes.ts b/common/lib/storageVolumes.ts @@ -1,9 +1,22 @@ import path from "node:path"; -import { lstat, stat } from "node:fs/promises"; +import { readFileSync } from "node:fs"; +import { lstat, readFile, realpath, stat } from "node:fs/promises"; import { execa } from "execa"; import type { Paths } from "./paths"; import { getFreeBytes } from "./diskSpace"; import type { StorageLocation, StorageVolume } from "./storageLocations"; +import { + HEALTH_PROBE_TIMEOUT_MS, + countersVerdict, + isDriveNotAnswering, + onDrive, + parseBlockStat, + setCounterReader, + stalledLocation, + type BlockStatSample, + type HealthDetector, + type LocationHealthState, +} from "./storageHealth"; // STORAGE VOLUME PROBES — is this location's disk here, and if not, where? // @@ -33,7 +46,11 @@ export type StorageLocationStatus = | "absent" // The root is not there and we have no identity to look for — nothing to say // beyond "that path does not exist". - | "missing"; + | "missing" + // The drive did not answer the last health probe (`lib/storageHealth.ts`), so + // this probe did not ask: an in-process `stat` there would hold one of the + // process's I/O threads until the drive answered. No identity, no free space. + | "stalled"; // `known: false` is the fail-open answer and is NOT a problem report: it means // the probe could not ask (no findmnt, a container, a timeout), not that the @@ -226,13 +243,25 @@ export async function probeLocation( const timeoutMs = opts.findmntTimeoutMs ?? FINDMNT_TIMEOUT_MS; const root = loc.root.trim(); + // A DRIVE THAT IS NOT ANSWERING IS NOT ASKED. The `stat` and `statfs` below + // run in-process, and on a stalled disk each holds an I/O thread until the + // drive comes back; the health pass (below) already asked without touching + // it. Both go through `onDrive`'s watchdog, too: one that has not answered + // in 3 s marks the location stalled and this probe answers so. + const stalledProbe: StorageLocationProbe = { + status: "stalled", + identity: { known: false }, + }; + if (stalledLocation(loc)) return stalledProbe; + // AVAILABILITY IS `stat`, AND ONLY `stat`. A root that is a directory is // available even when every identity probe below fails — see the header. let isDir = false; if (root !== "") { try { - isDir = (await stat(root)).isDirectory(); - } catch { + isDir = (await onDrive(loc, () => stat(root))).isDirectory(); + } catch (err) { + if (isDriveNotAnswering(err)) return stalledProbe; isDir = false; } } @@ -245,7 +274,13 @@ export async function probeLocation( // `null` and every byte formatter downstream gets a surprise. An // unmeasurable root reports no free space at all, which is the honest // answer and the one the field is already optional for. - const free = await getFreeBytes(root); + let free: number; + try { + free = await onDrive(loc, () => getFreeBytes(root)); + } catch (err) { + if (isDriveNotAnswering(err)) return stalledProbe; + throw err; + } // Two ways a mount will not be there after a reboot: udisks put it under // /run/media (or /media) because a human plugged it in, or there is no // fstab entry naming its UUID. The fstab call is skipped when the @@ -304,6 +339,323 @@ export async function probeLocation( } } +// --------------------------------------------------------------------------- +// The health probe: is the drive ANSWERING, asked from a child process +// --------------------------------------------------------------------------- +// +// `probeLocation` answers "is the disk here" with an in-process `stat`, which is +// right until the disk is here and not answering: then that `stat` does not +// fail, it waits — for as long as the drive takes, on one of the four threads +// libuv runs every filesystem call on. A child process waiting in the kernel +// holds none of them. So this runs `stat` on the root as a SUBPROCESS and races +// it against a timer, and the timer's answer is `stalled`. +// +// THE TIMER WINS, AND NOTHING WAITS FOR THE CHILD. A process in uninterruptible +// I/O cannot be killed until the I/O returns, so awaiting its exit (which is +// what execa's own `timeout` does) would hand the wait straight back to us. +// The child is sent SIGKILL, which it takes when the drive lets it, and its +// promise settles into a handler nobody awaits. +// +// FAILS OPEN, like everything else here: a `stat` that could not be started +// (no binary) is "could not ask", which is `ok`, never `stalled`. Only a child +// that started and did not answer in time is a stall. + +export type HealthProbeOptions = { + timeoutMs?: number; + // Test seam: the binary to run (it is called as `<bin> -L -c %F -- <root>`). + statBin?: string; +}; + +// One pass's answer about one location. `answer: null` is no verdict (the +// counters' first sample, or a second one taken too soon after the last). +export type HealthVerdict = { + answer: LocationHealthState | null; + detector?: HealthDetector; + // For a `stalled` answer: what did not answer, in words with no path. + cause?: string; + // The block device the counters were read from. + device?: string; +}; + +// What the health pass asks about a location. A bare state is a verdict with +// no detector named (the tests' scripted probes). +export type LocationHealthProbe = ( + loc: Pick<StorageLocation, "id" | "label" | "root" | "volume">, +) => Promise<LocationHealthState | HealthVerdict>; + +export async function probeLocationHealth( + loc: Pick<StorageLocation, "root">, + opts: HealthProbeOptions = {}, +): Promise<LocationHealthState> { + const root = loc.root.trim(); + if (root === "") return "absent"; + const timeoutMs = opts.timeoutMs ?? HEALTH_PROBE_TIMEOUT_MS; + let child: ReturnType<typeof execa>; + try { + child = execa(opts.statBin ?? "stat", ["-L", "-c", "%F", "--", root], { + buffer: true, + reject: false, + stdin: "ignore", + }); + } catch { + return "ok"; + } + const answered: Promise<LocationHealthState> = child.then( + (res) => { + // No exit code: the binary never ran (or was killed after the race was + // already decided). Could not ask. + if (typeof res.exitCode !== "number") return "ok"; + if (res.exitCode !== 0) return "absent"; + const out = typeof res.stdout === "string" ? res.stdout.trim() : ""; + return out === "directory" ? "ok" : "absent"; + }, + () => "ok", + ); + let timer: ReturnType<typeof setTimeout> | undefined; + const timedOut = new Promise<LocationHealthState>((resolve) => { + timer = setTimeout(() => resolve("stalled"), timeoutMs); + timer.unref?.(); + }); + const answer = await Promise.race([answered, timedOut]); + if (timer) clearTimeout(timer); + if (answer === "stalled") { + try { + child.kill("SIGKILL"); + } catch { + /* already gone */ + } + } + return answer; +} + +// --------------------------------------------------------------------------- +// The counters detector: the block device's own request counters +// --------------------------------------------------------------------------- +// +// THE STAT PROBE ABOVE CAN BE ANSWERED FROM THE KERNEL'S CACHE. A root's inode +// is cached whenever anything has used the drive lately, so a child `stat` of +// it answers in microseconds while the reads that actually reach the device +// wait out a reset loop. The block device's counters (lib/storageHealth.ts, +// `countersVerdict`) are the device's own account of what it has done, and +// reading them touches only /sys. So the health pass asks this first: +// +// 1. the root's device: `findmnt -J -T <root> -o SOURCE,UUID` as a child +// raced against 3 s (a findmnt stuck resolving the root holds nothing of +// ours), the `[subvolume]` suffix a bind or btrfs mount adds taken off, +// `/dev/mapper/<x>` resolved to its `dm-N`, then the basename. A partition +// and a mapper device both have `/sys/class/block/<name>/stat`. Asked only +// when the root has no device yet or its device's /sys entry cannot be read +// (a drive replugged under another name): a findmnt per pass would leave +// one child stuck per pass during a long stall. One that times out names +// none this pass; one whose UUID is not the location's recorded one names +// none (the root is then a directory on some other filesystem). +// 2. that device's `stat` line, compared with the previous pass's sample for +// the location (same device, at least MIN_COUNTER_INTERVAL_MS earlier). +// +// NO DEVICE (a container, no findmnt, a network or tmpfs mount, no /sys entry) +// FALLS BACK TO THE CHILD `stat`, and the verdict says which detector answered. + +export type DetectorOptions = HealthProbeOptions & { + // Test seams: where /sys/class/block is, and the clock. + sysBlockDir?: string; + now?: number; +}; + +export const SYS_BLOCK_DIR = "/sys/class/block"; +// Two samples closer than this are not compared: a healthy drive can have a +// request in flight at two instants a moment apart without completing one. +// The pass is 15 s apart; a /storage Refresh just after a pass gives no verdict. +export const MIN_COUNTER_INTERVAL_MS = 10_000; + +type CounterSample = BlockStatSample & { device: string; at: number }; + +type DetectorState = { + samples: Map<string, CounterSample>; + deviceByRoot: Map<string, string>; +}; + +declare global { + // eslint-disable-next-line no-var + var __yttHealthDetector__: DetectorState | undefined; +} + +// ON globalThis, the house pattern: the health pass runs in instrumentation's +// module copy and /storage's Refresh in a page's, and they must compare +// against the same previous sample. +function detectorState(): DetectorState { + globalThis.__yttHealthDetector__ ??= { + samples: new Map(), + deviceByRoot: new Map(), + }; + return globalThis.__yttHealthDetector__; +} + +export function resetHealthDetector(): void { + detectorState().samples.clear(); + detectorState().deviceByRoot.clear(); +} + +// THE WATCHDOG'S READING of a device's counters (lib/storageHealth.ts, +// `onDrive`): synchronous, so it does not wait behind the thread pool it is +// judging, and cheap — a /sys read is answered by the kernel from memory and +// never reaches the drive. +function readCountersNow(device: string): BlockStatSample | null { + try { + return parseBlockStat(readFileSync(path.join(SYS_BLOCK_DIR, device, "stat"), "utf8")); + } catch { + return null; + } +} +setCounterReader(readCountersNow); + +type Raced = { answered: false } | ({ answered: true } & Run); + +// `run`, but raced against a timer that nobody waits past: a child stuck in +// the kernel is sent SIGKILL and left to exit when it can. +async function runRaced(bin: string, args: string[], timeoutMs: number): Promise<Raced> { + let child: ReturnType<typeof execa>; + try { + child = execa(bin, args, { buffer: true, reject: false, stdin: "ignore" }); + } catch { + return { answered: true, ok: false, exitCode: undefined, stdout: "", stderr: "" }; + } + const answered: Promise<Raced> = child.then( + (res) => { + const exitCode = typeof res.exitCode === "number" ? res.exitCode : undefined; + return { + answered: true as const, + ok: exitCode === 0, + exitCode, + stdout: typeof res.stdout === "string" ? res.stdout : "", + stderr: typeof res.stderr === "string" ? res.stderr : "", + }; + }, + () => ({ answered: true as const, ok: false, exitCode: undefined, stdout: "", stderr: "" }), + ); + let timer: ReturnType<typeof setTimeout> | undefined; + const timedOut = new Promise<Raced>((resolve) => { + timer = setTimeout(() => resolve({ answered: false }), timeoutMs); + }); + const out = await Promise.race([answered, timedOut]); + if (timer) clearTimeout(timer); + if (!out.answered) { + try { + child.kill("SIGKILL"); + } catch { + /* already gone */ + } + } + return out; +} + +// The block device name under /sys/class/block for a mount SOURCE, or null. +export async function blockDeviceName(source: string): Promise<string | null> { + const bare = source.replace(/\[.*\]$/, "").trim(); + if (!bare.startsWith("/dev/")) return null; + // /dev/mapper/<x> and /dev/disk/by-*/<x> are links to the kernel's name. + // Resolving them reads /dev, never the drive. + const resolved = await realpath(bare).catch(() => bare); + const name = path.basename(resolved); + return name && name !== "dev" ? name : null; +} + +async function blockDeviceOfRoot( + loc: Pick<StorageLocation, "root" | "volume">, + bins: Pick<VolumeBins, "findmntBin">, + timeoutMs: number, +): Promise<{ device: string | null; timedOut: boolean }> { + const res = await runRaced( + bins.findmntBin, + ["-J", "-T", loc.root, "-o", "SOURCE,UUID"], + timeoutMs, + ); + if (!res.answered) return { device: null, timedOut: true }; + if (!res.ok) return { device: null, timedOut: false }; + let fs0: Record<string, unknown> | undefined; + try { + fs0 = (JSON.parse(res.stdout) as { filesystems?: Record<string, unknown>[] }) + ?.filesystems?.[0]; + } catch { + return { device: null, timedOut: false }; + } + const source = typeof fs0?.source === "string" ? fs0.source : ""; + const uuid = typeof fs0?.uuid === "string" ? fs0.uuid : ""; + const recorded = loc.volume?.uuid?.trim() ?? ""; + if (recorded && uuid && uuid !== recorded) return { device: null, timedOut: false }; + return { device: await blockDeviceName(source), timedOut: false }; +} + +async function readBlockStat( + device: string, + sysBlockDir: string, +): Promise<BlockStatSample | null> { + try { + return parseBlockStat(await readFile(path.join(sysBlockDir, device, "stat"), "utf8")); + } catch { + return null; + } +} + +// One location's verdict for the health pass: the counters when its device can +// be named and read, the child `stat` otherwise. +export async function detectLocationHealth( + loc: Pick<StorageLocation, "id" | "root" | "volume">, + bins: Pick<VolumeBins, "findmntBin">, + opts: DetectorOptions = {}, +): Promise<HealthVerdict> { + const now = opts.now ?? Date.now(); + const timeoutMs = opts.timeoutMs ?? HEALTH_PROBE_TIMEOUT_MS; + const sysBlockDir = opts.sysBlockDir ?? SYS_BLOCK_DIR; + const detector = detectorState(); + // The device this root was last known on, if its /sys entry still reads; + // findmnt only when there is none (see step 1 above). + let device: string | null = detector.deviceByRoot.get(loc.root) ?? null; + let sample = device ? await readBlockStat(device, sysBlockDir) : null; + if (!sample) { + detector.deviceByRoot.delete(loc.root); + const mapped = loc.root.trim() + ? await blockDeviceOfRoot(loc, bins, timeoutMs) + : { device: null, timedOut: false }; + device = mapped.device; + sample = device ? await readBlockStat(device, sysBlockDir) : null; + } + if (device) { + if (sample) { + detector.deviceByRoot.set(loc.root, device); + const prev = detector.samples.get(loc.id); + if (prev && prev.device === device && now - prev.at < MIN_COUNTER_INTERVAL_MS) { + return { answer: null, detector: "counters", device }; + } + detector.samples.set(loc.id, { ...sample, device, at: now }); + if (!prev || prev.device !== device) { + return { answer: null, detector: "counters", device }; + } + const answer = countersVerdict(prev, sample); + return { + answer, + detector: "counters", + device, + ...(answer === "stalled" + ? { + cause: + `its disk (${device}) had ${sample.inFlight} request(s) in flight and ` + + `completed none in ${Math.round((now - prev.at) / 1000)} s`, + } + : {}), + }; + } + } + detector.samples.delete(loc.id); + const answer = await probeLocationHealth(loc, opts); + return { + answer, + detector: "stat", + ...(answer === "stalled" + ? { cause: `a stat of its root did not answer within ${timeoutMs / 1000} s` } + : {}), + }; +} + // udisksctl availability, memoised per binary path. Same shape as a digest // app's probe (`digestApps.ts` claudeCode.probe): `--version`, reject:false, // short timeout. Memoised because /storage asks once per render and the answer @@ -401,6 +753,12 @@ export async function probeLocationMemo( opts: ProbeOptions & { refresh?: boolean; now?: number } = {}, ): Promise<MemoizedProbe> { const now = opts.now ?? Date.now(); + // The stall is asked before the memo, so a remembered "available" from a few + // seconds ago does not outlive the drive's answer. Not remembered either: + // the health probe's state is already the memory. + if (stalledLocation(loc)) { + return { status: "stalled", identity: { known: false }, probedAt: now }; + } const hit = probeMemo.get(loc.id); if ( !opts.refresh && diff --git a/common/styles/tokens.css b/common/styles/tokens.css @@ -255,6 +255,15 @@ html[data-base="light"] { --chart-surface: #ffffff; --chart-grid: rgba(22, 28, 33, 0.09); --chart-axis: #55646e; + /* OTHER: the homepage growth chart's band for the sites it groups + (homepage/app/lib/growthGaps.ts), a near-neutral slate at the foot of + the palette's lightness band (OKLCH L 0.43, C 0.03). 7.36:1 on the + ground, 7.99:1 on the chart surface. Against each slot it can sit on + (the dataviz validator, normal / worst CVD): blue 21.9 / 21.4, green + 17.4 / 15.2, violet 16.2 / 13.4, amber 22.1 / 18.2, magenta 24.5 / + 9.1, rust 14.1 / 10.5 — every CVD pair clears 8; rust is the one + normal pair under 15. */ + --chart-other: #3e545c; --chart-tooltip-bg: #ffffff; } @@ -338,6 +347,14 @@ html[data-base="dark"] { --chart-surface: #16110a; --chart-grid: rgba(233, 220, 197, 0.08); --chart-axis: #8a8170; + /* OTHER, on this base: a near-neutral grey (OKLCH L 0.49, C 0.01), + the dimmest chart mark. 3.22:1 on the ground, 3.06:1 on the chart + surface (the slots: 3.37–6.46:1 on the ground). Against each slot (normal / worst CVD): blue 17.8 / 17.9, + green 19.2 / 15.7, violet 23.4 / 21.5, amber 20.3 / 18.1, magenta + 19.5 / 7.3, rust 13.5 / 9.3 — magenta in the CVD 6–8 floor band + (legal with the legend, the table and Other's place on top), rust the + one normal pair under 15. */ + --chart-other: #62625c; --chart-tooltip-bg: #1b150d; } diff --git a/common/views/pipeline/stageStatus.ts b/common/views/pipeline/stageStatus.ts @@ -601,7 +601,8 @@ export function computeStageStatuses( mediaStatus === "in-transition" ? "running" : mediaStatus === "unreachable" || - mediaStatus === "inconsistent" + mediaStatus === "inconsistent" || + mediaStatus === "stalled" ? "danger" : "neutral", }; diff --git a/common/views/storage.test.ts b/common/views/storage.test.ts @@ -399,3 +399,36 @@ test("a location row carries the clips share of its bytes", () => { // A SUBSET, not a sibling: the clips are already inside `bytes`. assert.equal(row?.bytes, 10_000_000_000); }); + +test("a location whose drive is not answering reads so, and says since when", () => { + const { rows } = buildStorageRows({ + locations: [loc("cold", "/mnt/cold")], + defaultLocationId: "", + probes: { + cold: { status: "stalled", identity: { known: false }, probedAt: NOW }, + }, + notAnswering: { cold: "not answering since 11:35 — a stat of its root did not answer within 3 s." }, + rollups: { cold: rollup({ locationId: "cold", total: 2, unreachable: 2 }) }, + registry: NO_JOBS, + now: NOW, + }); + const row = rows[0]; + assert.equal(row.status, "stalled"); + assert.equal(row.statusLabel, "Not answering"); + assert.match(String(row.notAnswering), /^not answering since 11:35/); + assert.equal(row.freeBytes, undefined); + // Neither re-point nor mount is offered for a drive that is here and stalled. + assert.equal(action(row, "repoint").offered, false); + assert.match(String(action(row, "repoint").withheld), /not answering/); + assert.equal(action(row, "mount").offered, false); + // A row with no entry says nothing. + const quiet = buildStorageRows({ + locations: [loc("cold", "/mnt/cold")], + defaultLocationId: "", + probes: {}, + rollups: {}, + registry: NO_JOBS, + now: NOW, + }).rows[0]; + assert.equal(quiet.notAnswering, undefined); +}); diff --git a/common/views/storage.ts b/common/views/storage.ts @@ -103,6 +103,9 @@ export type StorageRow = { freeBytes?: number; // How long ago the probe behind this row was taken, in ms. lastProbeAgeMs: number; + // "not answering since 11:35 — …", while the health probe finds this + // location's drive not answering (lib/storageHealth.ts). Absent otherwise. + notAnswering?: string; warning?: string; candidateRoot?: string; // Why every action on this row is withheld, or null. One re-point runs at a @@ -201,6 +204,10 @@ export type StorageRowsInputs = { // omitted, because a row that vanishes when a probe fails is worse than one // that says it does not know. probes: Record<string, MemoizedProbe | undefined>; + // By location id: the sentence for a location whose drive is not answering. + // Worded by the shell (lib/storageHealth.ts's `notAnsweringText`), because + // this module takes types only from lib/. + notAnswering?: Record<string, string | undefined>; rollups: Record<string, LocationRollup | undefined>; registry: RegistryReader; udisksctlAvailable?: boolean; @@ -217,6 +224,7 @@ export const STORAGE_STATUS_LABEL: Record<StorageLocationStatus, string> = { unmounted: "Not mounted", absent: "Not attached", missing: "Missing", + stalled: "Not answering", }; export const REPOINT_JOB_KIND = "repoint-storage-location"; @@ -386,6 +394,9 @@ export function buildStorageRows(i: StorageRowsInputs): StorageRowsPayload { clipsText: storageClipsText(clipsBytes), ...(probe?.freeBytes !== undefined ? { freeBytes: probe.freeBytes } : {}), lastProbeAgeMs: probe ? Math.max(0, i.now - probe.probedAt) : 0, + ...(i.notAnswering?.[loc.id] + ? { notAnswering: i.notAnswering[loc.id] } + : {}), ...(probe?.warning ? { warning: probe.warning } : {}), ...(candidateRoot ? { candidateRoot } : {}), busy, diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh @@ -222,6 +222,12 @@ editor) log "idle boot: schedulers, auto-queue runners and sweeps stay STOPPED" fi cd /repo/editor + # Sixteen threads for Node's filesystem pool instead of four. A call on a + # drive that has stalled holds its thread until the drive answers; with four, + # four such calls stop the editor answering at all. More threads buy time for + # calls already in flight — they isolate nothing (the storage health probe + # and its gate are what keep new calls off a stalled drive). + export UV_THREADPOOL_SIZE="${UV_THREADPOOL_SIZE:-16}" exec "$(next_bin /repo/editor)" start --port "${EDITOR_PORT:-3001}" ;; diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -4,6 +4,8 @@ - **Transcripts that arrived after a video was first seen are counted.** The stats behind the homepage, the hub and every site's charts were cached per video and refreshed only when the video's metadata changed, so a transcript that came later — a Whisper run days after the download, or a video downloaded after the last index build — never reached them, and a video with YouTube captions alone had no transcription date. Counts and charts were low; the homepage could show a site with 0 transcripts, 0 channels and 0 hours while it served its videos. A stat is now also redone whenever the index re-reads the video, every transcript has a date, and a captioned video is dated by when its captions arrived rather than by a later Normalize run, so its place on "Transcribed over time" can move. **After updating, rebuild and restart the editor before anything else:** until then, **Build stats dataset** runs the old code and would undo the new stats, while a site, hub or homepage build already runs the new code — and the first stats build of any kind re-reads every video once (about 10–30 minutes on a large archive; it can be stopped and picks up where it stopped). Then build the index, the stats, the homepage, the hub, and the sites. - **A stats build keeps the stats of a channel whose drive is not mounted, and will not undo a newer version's stats.** A channel whose media is on a drive that is not mounted (or is being moved) is left as it was instead of being read as a channel with no videos; a stats rebuild that has to start over refuses until the drive is back. A stats build refuses to clear stats written by a newer version of the editor; set `ARCHILYZER_STATS_ALLOW_DOWNGRADE=1` to roll back on purpose. Its log also says apart how many videos were downloaded since the last index build (they catch up after the next one) and how many the index skipped (no upload date, or it failed on them). - **An index build keeps a channel whose drive is not mounted, instead of dropping it from the sites.** **Build index**, a site build's data phase and `archilyzer index` read a channel whose media is on a drive that is not mounted (or is being moved, or whose link and config disagree) as a channel with no videos: they removed its videos from the index, and the next site build published the channel as gone. Such a channel is now left as the last build had it — its videos stay in the index, its pages stay as they were, and the sites built next still list it — and the log names it, with its storage location: one line per channel, ` Held: N channel(s), K video(s) kept.` at the end of the `Diff:` line, and the channels again on the last line. A data folder that fails to read is held the same way, and a channel with no data folder at all is said in the log instead of passed over. An index rebuild that has to start over (after an update that changes the index's format, or with no index yet) refuses while any channel is held and says which; mount the drive first, or set `ARCHILYZER_INDEX_ALLOW_HELD=1` to rebuild without that channel until its drive is back and the index is built again — on the command for a command-line build (`ARCHILYZER_INDEX_ALLOW_HELD=1 pnpm archilyzer index`), or in the editor's own environment, with a restart, for **Build index** and the site builds started from the editor. +- **umtool's build no longer lists its e2e test data, the e2e server's build folder or `.env.local` among a route's files.** The clip-audio route named its cache files in a way the bundler read as a pattern reaching into umtool's hidden folders, so its list of files took in the e2e fixture (where the tests link the song data), the e2e dev server's build folder and the env file: 1,704 of its 2,167 entries. It now lists what the other routes list (463). Those folders and env files are also excluded from every route's list, and `pnpm test:scripts` reads the last umtool build's lists back and fails on any such entry. A checkout whose umtool build predates its code (this change included) skips that check, saying so, until umtool is rebuilt (`pnpm --filter umtool exec next build`). Nothing changes when umtool runs. +- **A drive that stops answering no longer stops the editor answering.** When a storage location's drive is mounted but not answering (an SMR disk in a USB enclosure resetting under a long write), every page and poll that touched it waited on it, and a few such waits froze the whole editor until the drive came back. Every 15 seconds the editor now reads each location's disk activity counters from the kernel, which never waits on the drive: a disk with requests waiting and none finished since the last look is marked **Not answering**, and the mark comes off after two looks in a row find it working. Where no disk can be named (in a container, say) it asks the drive from a separate process with a 3-second limit instead. Any page or poll that reads the drive also gives up after 3 seconds and marks it the same way, and no more than four such reads wait on one drive at a time. While it is marked, the editor's pages and polls do not read that drive: `/storage` shows the location as **Not answering** with the time it stopped and how it is watched (**Refresh** asks the drive again), the `/channels` volume chip reads "not answering since HH:MM" and its channels' badges "not answering", their videos list, video pages and Cleanup stage say so instead of reading the drive, `/saved-videos` names the channels it did not read, and the index and stats builds keep those channels as they do for an unmounted drive, and a channel that is in the middle of a move still shows as moving. Jobs for those channels are refused until the drive answers, and a channel paused automatically for it says the drive is not answering rather than not there. The drive check keeps running on an editor started with `ARCHILYZER_IDLE_BOOT`. Pages and polls also reuse each channel's media check for 5 seconds. The editor's `start` script and the container now give Node 16 threads for file access instead of 4 (`UV_THREADPOOL_SIZE`); that buys time for reads already waiting on a drive, and a read that was already waiting when the drive stalled still waits until the drive answers. - **Building the homepage now publishes the source: a read-only git mirror, its raw tree and a fresh tarball, behind a gate.** `archilyzer build homepage`, the `/sites` Homepage jobs and `pnpm ops build-homepage` run `archilyzer source publish` between compose and `next build`. It makes a fresh clone of the private `main` (the repository itself is never rewritten), rewrites that copy with git-filter-repo using your scrub rules (file contents and commit messages; your home directory becomes `/home/user` without a rule), and publishes it under `homepage/public` for `git clone https://archilyzer.pages.dev/source/archilyzer.git`, beside `/source/tree/` and the Downloads tarball. Before anything is written, every object of the rewritten history and every file about to be published is searched for every string you have denied; **one hit refuses the build**, and its log names the string only by where you wrote it (`denylist line 3 (len 5)`) and each hit by its object, field and byte offset — never a byte of the object. **A refusal withdraws the source**: the last publish is removed from `homepage/public` and the last build's copy from `homepage/out`, and **Deploy homepage refuses** a build whose source was not audited under today's rules and today's `main` ("run `archilyzer build homepage`, then deploy"). The rules live outside the repo, in `~/.config/archilyzer/source-scrub.txt` and `source-denylist.txt` (`ARCHILYZER_CONFIG_DIR`, `SOURCE_SCRUB_FILE`, `SOURCE_DENYLIST_FILE`); **without them the build refuses**, naming the missing file. **Put everything private in the denylist before any deploy, a preview included**: previews are public, and every deployment stays reachable at its own address until you delete it. Install git-filter-repo once (`pipx install git-filter-repo`; the editor's process needs `~/.local/bin` on its `PATH` to find it) — without it the build fetches it through `pipx run`, which needs the network — and gitleaks if you want its secret scan too. An unchanged `main` with unchanged rules is skipped, so a rebuild costs about 20 seconds only when something moved. A checkout with no git repository (the docker image, a tarball install) builds with the /source page's empty state. `archilyzer source publish --check` audits without writing, `archilyzer source audit <clone>/.git` checks any clone, `archilyzer build homepage --no-source` removes the published source instead, and `archilyzer doctor` reports the tools, the two files (rule counts and permissions, never their contents) and the last publish. `create-archives.sh` is gone. See PUBLISH.md, "The source mirror (homepage)". - **umtool reads the corpus from its checkout (or `TRANSCRIPTS_DIR`), and the song project's data defaults to `~/.local/share/archilyzer/song`.** If yours is elsewhere, link it there before restarting umtool: `mkdir -p ~/.local/share/archilyzer && ln -s <where the data is> ~/.local/share/archilyzer/song` (the data stays where it is). With no `CHANNELS_DIR`, umtool reads the corpus at `$TRANSCRIPTS_DIR/channels`, else the checkout's own `transcripts/channels`; it used to fall back to an absolute path that existed on one machine only. The song project's videos default to `~/reports/quartering-uh-song/videos`; `SONG_DIR` and `VIDEO_ROOT` still win. The song project's tracked manifests record their paths relative to the song folders, and the twenty one-off `umtool/song/*.sh` run logs, which only ever ran on the machine that wrote them, are gone. - **umtool's production build no longer reads the corpus folder.** Since umtool began finding the corpus from its checkout (the bullet above), `next build` treated the checkout's whole `transcripts/channels` as files to bundle. On a real archive it ran out of memory and was killed, so umtool could not be rebuilt. The build now ignores that folder and finishes in about 25 s at under 1 GB, the same as a checkout with no corpus. Nothing changes when umtool runs. diff --git a/editor/app/api/channels/[slug]/videos/[id]/files/[name]/route.ts b/editor/app/api/channels/[slug]/videos/[id]/files/[name]/route.ts @@ -4,6 +4,13 @@ import { stat } from "node:fs/promises"; import { NextResponse } from "next/server"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { makeSafeController } from "yt-dlp-transcript-common/lib/safeStreamController"; +import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; +import { channelMediaStall } from "yt-dlp-transcript-common/lib/channelMedia"; +import { + NOT_ANSWERING, + isDriveNotAnswering, + onDrive, +} from "yt-dlp-transcript-common/lib/storageHealth"; export const dynamic = "force-dynamic"; @@ -123,10 +130,24 @@ export async function GET( return NextResponse.json({ error: "Forbidden" }, { status: 403 }); } + // A drive that is not answering is not asked: the stat and the stream would + // each wait on it. 503, because it is a state that passes. The stat goes + // through the watchdog, so a drive that stops answering now is a 503 too; + // the stream that follows a stat that answered is not raced. + const notAnswering = () => + NextResponse.json( + { error: `Media not read: ${NOT_ANSWERING}.` }, + { status: 503, headers: { "retry-after": "15" } }, + ); + const channelConfig = await readChannelConfig(paths, slug); + if (channelMediaStall(channelConfig)) return notAnswering(); + const drive = channelConfig?.dataDir?.trim(); + let stats; try { - stats = await stat(fullPath); - } catch { + stats = await (drive ? onDrive(drive, () => stat(fullPath)) : stat(fullPath)); + } catch (err) { + if (isDriveNotAnswering(err)) return notAnswering(); return NextResponse.json({ error: "Not found" }, { status: 404 }); } if (!stats.isFile()) { diff --git a/editor/app/api/test/invalidate-cache/route.ts b/editor/app/api/test/invalidate-cache/route.ts @@ -4,6 +4,8 @@ import { resetSnapshotScheduler } from "yt-dlp-transcript-common/jobs/snapshotSc import { resetChannelSnapshotMemo } from "yt-dlp-transcript-common/controller/channels"; import { resetStorageProbeMemo } from "yt-dlp-transcript-common/controller/storageLocations"; import { resetVideoTitleMemo } from "yt-dlp-transcript-common/controller/videoTitles"; +import { forgetChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia"; +import { resetStorageHealth } from "yt-dlp-transcript-common/lib/storageHealth"; import { testRouteDenied } from "../_guard"; export const dynamic = "force-dynamic"; @@ -112,6 +114,12 @@ function invalidate() { // testInfo.outputPath varies by accident rather than on purpose; cleared here // so it is on purpose. resetStorageProbeMemo(); + // And the two memories a stalled drive lives in: inspectChannelMedia's + // five-second answers (keyed by channels dir, slug and the configured target, + // which a reset fixture reproduces exactly) and each location's health, which + // a previous spec's location id would otherwise carry into this one. + forgetChannelMedia(); + resetStorageHealth(); // And the video list's metadata.info.json title memo. It is keyed by the // channel's data/ mtime, which resetData() changes by recreating the dir — but // a spec that rewrites a title in place inside one mtime tick would otherwise diff --git a/editor/app/channels/[slug]/components/MediaNotAnswering.tsx b/editor/app/channels/[slug]/components/MediaNotAnswering.tsx @@ -0,0 +1,67 @@ +import Link from "next/link"; +import { + notAnsweringText, + type LocationHealth, +} from "yt-dlp-transcript-common/lib/storageHealth"; + +// WHAT A PAGE THAT READS A CHANNEL'S `data/` SHOWS WHILE ITS DRIVE IS NOT +// ANSWERING, instead of reading it. +// +// The videos list and the video page read the channel's media directory on +// every render — a readdir, a stat per file, a metadata head read per title. On +// a drive that is mounted and not answering (lib/storageHealth.ts) each of those +// calls waits for the drive, on one of the few threads every page and poll in +// this process shares. So the page asks the health state first, and on a +// stalled drive it says so and reads nothing. The rest of the channel (its +// overview, its report, its settings) comes off the corpus disk and still +// renders. +// +// Server-only: it takes the health entry from the page, which read it from +// memory — or from the DriveNotAnsweringError a page's read got from `onDrive`'s +// watchdog. +export function MediaNotAnswering({ + slug, + stall, + what, +}: { + slug: string; + // The stalled location, or null when the drive is on no location the health + // state knows and a read of it did not answer within 3 s. + stall: LocationHealth | null; + // What would have been shown: "The video list", "This video". + what: string; +}) { + return ( + <div className="flex flex-col gap-3"> + <p + role="status" + aria-label="media not answering" + className="rounded border border-destructive/50 bg-destructive/5 px-3 py-2 text-sm text-destructive" + > + {stall ? ( + <> + {what} reads this channel&apos;s media, which is on &ldquo; + {stall.label}&rdquo; — a drive that is {notAnsweringText(stall)}. + Nothing is read from it until it answers again (it is checked every + 15 s). + </> + ) : ( + <> + {what} reads this channel&apos;s media, and a read of its drive did + not answer within 3 s. Nothing more is read from it on this page. + </> + )} + </p> + <p className="text-sm text-muted-foreground"> + <Link href={`/channels/${slug}`} className="underline hover:text-foreground"> + The channel + </Link>{" "} + still shows its report, and{" "} + <Link href="/storage" className="underline hover:text-foreground"> + Storage + </Link>{" "} + shows the drive. + </p> + </div> + ); +} diff --git a/editor/app/channels/[slug]/components/stages/StorageStage.tsx b/editor/app/channels/[slug]/components/stages/StorageStage.tsx @@ -93,7 +93,8 @@ type Props = { clipsBytes: number | null; // Free space on the volume the media is on RIGHT NOW — the platter for a // relocated channel, the corpus disk otherwise. - freeBytes: number; + // Null when the drive did not answer the statfs ("—"). + freeBytes: number | null; volumeDir: string; // Why both buttons are off, or null when they are live. Running/queued jobs // for this channel, or a relocation marker left by an interrupted move. @@ -151,7 +152,7 @@ export function StorageStage({ </dd> <dt className="text-muted-foreground">Free on that volume</dt> <dd aria-label="free on media volume"> - {formatBytes(freeBytes)}{" "} + {freeBytes === null ? "—" : formatBytes(freeBytes)}{" "} <span className="text-xs text-muted-foreground font-mono"> ({volumeDir}) </span> @@ -214,7 +215,8 @@ export function StorageStage({ // a reason, one click earlier. unvouched={ location.status === "inconsistent" || - location.status === "unreachable" + location.status === "unreachable" || + location.status === "stalled" ? `This channel's media location is ${location.status}: ${ location.detail ?? "disk and config do not agree" } Moving back would delete the relocated copy, so it is refused until the location reads "relocated · reachable".` diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -40,7 +40,15 @@ import { type ShardOp, } from "yt-dlp-transcript-common/controller/shard"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { inspectChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia"; +import { + channelMediaStall, + inspectChannelMedia, +} from "yt-dlp-transcript-common/lib/channelMedia"; +import { MediaNotAnswering } from "./components/MediaNotAnswering"; +import { + isDriveNotAnswering, + onDrive, +} from "yt-dlp-transcript-common/lib/storageHealth"; import { getFreeBytes } from "yt-dlp-transcript-common/lib/diskSpace"; import { platformQueueKey, @@ -515,12 +523,34 @@ export default async function ChannelDetailPage({ /> ); case "cleanup": { + // The saved-video summary below reads a pointer in every video dir, and + // this stage's actions all act on the media: on a drive that is not + // answering the stage says so instead (see MediaNotAnswering). + const stall = channelMediaStall(config); + if (stall) { + return ( + <MediaNotAnswering slug={slug} stall={stall} what="The Cleanup stage" /> + ); + } // Saved-video store summary for this channel + whether backups are - // configured, for the Retention & persistence section. + // configured, for the Retention & persistence section. Its reads go + // through the watchdog (`notAnswering`): a drive that stops answering + // on the way is named there, and the stage says so. + const notAnswering: string[] = []; const savedTotals = await savedVideoTotals({ paths, channelSlug: slug, + notAnswering, }); + if (notAnswering.length > 0) { + return ( + <MediaNotAnswering + slug={slug} + stall={channelMediaStall(config)} + what="The Cleanup stage" + /> + ); + } return ( <CleanupStage slug={slug} @@ -586,7 +616,19 @@ export default async function ChannelDetailPage({ // marker — reading the file again here was a second read of the same // bytes that could disagree with the status rendered beside it. const marker = media.marker ?? null; - const freeBytes = await getFreeBytes(volumeDir); + // A statfs of the channel's drive goes through the watchdog + // (lib/storageHealth.ts): refused on a stalled location, given up on + // after 3 s, and read "—" either way. + let freeBytes: number | null; + try { + freeBytes = + volumeDir === media.target && media.target + ? await onDrive(media.target, () => getFreeBytes(volumeDir)) + : await getFreeBytes(volumeDir); + } catch (err) { + if (!isDriveNotAnswering(err)) throw err; + freeBytes = null; + } // The SAME two conditions storageActions.ts refuses on, stated here as // prose so the button is off with a reason rather than off and silent — // and stated in the action too, because a disabled button is a courtesy diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -42,6 +42,12 @@ import { TagsPanel } from "./components/TagsPanel"; import { MetadataHistoryDetails } from "./components/MetadataHistoryDetails"; import { loadVideoTags } from "./lib/videoTags"; import { loadVideoOperationPanels } from "./lib/videoOperationPanels"; +import { channelMediaStall } from "yt-dlp-transcript-common/lib/channelMedia"; +import { + isDriveNotAnswering, + onDrive, +} from "yt-dlp-transcript-common/lib/storageHealth"; +import { MediaNotAnswering } from "../../components/MediaNotAnswering"; export const dynamic = "force-dynamic"; @@ -80,8 +86,18 @@ export async function generateMetadata({ params: Promise<{ slug: string; id: string }>; }): Promise<Metadata> { const { slug, id } = await params; - const meta = await loadMeta(slug, id); - const subject = meta.title ?? id; + // The title is read off the drive; a drive that is not answering is not + // asked, and one that does not answer in 3 s is given up on. + const drive = (await readChannelConfig(getPaths(), slug))?.dataDir?.trim(); + let subject = id; + try { + const meta = await (drive + ? onDrive(drive, () => loadMeta(slug, id)) + : loadMeta(slug, id)); + subject = meta.title ?? id; + } catch (err) { + if (!isDriveNotAnswering(err)) throw err; + } return { title: `${subject} — Video — ${slug}` }; } @@ -93,63 +109,116 @@ export default async function VideoDetailPage({ const { slug, id } = await params; const config = await readChannelConfig(getPaths(), slug); if (!config) notFound(); - const dirData = await loadVideoDir(slug, id); - const meta = await loadMeta(slug, id); - const videoDir = path.join(getPaths().channelsDir, slug, "data", id); - const downloadOutcome = await loadDownloadOutcome(videoDir); - const availabilityRecord = await loadAvailability(videoDir); - const availabilityHistory = availabilityRecord?.history ?? []; - // Every rewrite of metadata.info.json that changed its bytes (release 10 - // slice N). One small bounded file, read once; null → nothing drawn. - const metadataHistory = metadataHistoryView( - await loadMetadataHistory(videoDir), - Date.now(), - ); - const doNotClean = await isDoNotClean(videoDir); - const excludedFromTruncatedCheck = - await isExcludedFromTruncatedCheck(videoDir); - const savedVideo = await loadSavedVideo(videoDir); - // The windows another tool asked this editor to fetch. One readdir of - // data/<id>/clips/ plus a stat per file — and no per-CHANNEL count anywhere, - // because that would be a walk of every video dir to draw one number. - const clipWindows = await listClipWindows(videoDir); - // Where each subtitle track came from, so the panel can say "YouTube - // auto-captions" vs "manual captions" — and offer to replace the former with a - // transcript of our own. One 4 KB head read per VTT (see subtitleProvenance). - const vttProvenance: Record<string, SubtitleProvenance> = {}; - for (const f of dirData.files) { - if (!isTranscriptVtt(f.name)) continue; - vttProvenance[f.name] = await resolveVttProvenance(videoDir, f.name); + // Everything below reads the video's directory on the channel's drive; a + // drive that is not answering is not read. See MediaNotAnswering. + const stall = channelMediaStall(config); + if (stall) { + return <MediaNotAnswering slug={slug} stall={stall} what="This video's page" />; } - const cov = await readTranscriptCoverage(videoDir); - const coverage = cov - ? { - lastCueEnd: cov.cov.lastCueEnd, - duration: cov.cov.duration, - coverage: cov.cov.coverage, - incomplete: - !excludedFromTruncatedCheck && - isIncompleteTranscript(cov.cov, { - isLivestream: cov.isLivestream, - }), - } - : null; + // EVERY READ BELOW IS OF THIS VIDEO'S DIRECTORY, on the channel's drive when + // it is relocated, so they go through the watchdog as one unit: a drive that + // has not answered them in 3 s is marked stalled and the page says so. + const loadAll = async () => { + const dirData = await loadVideoDir(slug, id); + const meta = await loadMeta(slug, id); + const videoDir = path.join(getPaths().channelsDir, slug, "data", id); + const downloadOutcome = await loadDownloadOutcome(videoDir); + const availabilityRecord = await loadAvailability(videoDir); + const availabilityHistory = availabilityRecord?.history ?? []; + // Every rewrite of metadata.info.json that changed its bytes (release 10 + // slice N). One small bounded file, read once; null → nothing drawn. + const metadataHistory = metadataHistoryView( + await loadMetadataHistory(videoDir), + Date.now(), + ); + const doNotClean = await isDoNotClean(videoDir); + const excludedFromTruncatedCheck = + await isExcludedFromTruncatedCheck(videoDir); + const savedVideo = await loadSavedVideo(videoDir); + // The windows another tool asked this editor to fetch. One readdir of + // data/<id>/clips/ plus a stat per file — and no per-CHANNEL count anywhere, + // because that would be a walk of every video dir to draw one number. + const clipWindows = await listClipWindows(videoDir); + // Where each subtitle track came from, so the panel can say "YouTube + // auto-captions" vs "manual captions" — and offer to replace the former with a + // transcript of our own. One 4 KB head read per VTT (see subtitleProvenance). + const vttProvenance: Record<string, SubtitleProvenance> = {}; + for (const f of dirData.files) { + if (!isTranscriptVtt(f.name)) continue; + vttProvenance[f.name] = await resolveVttProvenance(videoDir, f.name); + } + const cov = await readTranscriptCoverage(videoDir); + const coverage = cov + ? { + lastCueEnd: cov.cov.lastCueEnd, + duration: cov.cov.duration, + coverage: cov.cov.coverage, + incomplete: + !excludedFromTruncatedCheck && + isIncompleteTranscript(cov.cov, { + isLivestream: cov.isLivestream, + }), + } + : null; - // ONE PANEL PER REGISTRY OPERATION, each carrying the state that entry's own - // state() reports. The page used to re-derive the digest's freshness here — - // its own target resolution and its own per-section fold beside the - // registry's — which is how a page and a work list end up describing the same - // disk differently. - const panels = await loadVideoOperationPanels({ - paths: getPaths(), - channelSlug: slug, - videoId: id, - settings: getSettings(), - }); + // ONE PANEL PER REGISTRY OPERATION, each carrying the state that entry's own + // state() reports. The page used to re-derive the digest's freshness here — + // its own target resolution and its own per-section fold beside the + // registry's — which is how a page and a work list end up describing the same + // disk differently. + const panels = await loadVideoOperationPanels({ + paths: getPaths(), + channelSlug: slug, + videoId: id, + settings: getSettings(), + }); - // The curated tags on this video, with each one's provenance. Reads tags.json - // plus (for rule hits) ONE key out of the transcript index — not a scan. - const tagsView = await loadVideoTags(slug, id); + // The curated tags on this video, with each one's provenance. Reads tags.json + // plus (for rule hits) ONE key out of the transcript index — not a scan. + const tagsView = await loadVideoTags(slug, id); + return { + dirData, + meta, + videoDir, + downloadOutcome, + availabilityHistory, + metadataHistory, + doNotClean, + excludedFromTruncatedCheck, + savedVideo, + clipWindows, + vttProvenance, + coverage, + panels, + tagsView, + }; + }; + const drive = config.dataDir?.trim(); + let loaded: Awaited<ReturnType<typeof loadAll>>; + try { + loaded = await (drive ? onDrive(drive, loadAll) : loadAll()); + } catch (err) { + if (!isDriveNotAnswering(err)) throw err; + return ( + <MediaNotAnswering slug={slug} stall={err.health} what="This video's page" /> + ); + } + const { + dirData, + meta, + videoDir, + downloadOutcome, + availabilityHistory, + metadataHistory, + doNotClean, + excludedFromTruncatedCheck, + savedVideo, + clipWindows, + vttProvenance, + coverage, + panels, + tagsView, + } = loaded; const registry = getRegistry(); const existingQueues = registry.activeQueueNames(); diff --git a/editor/app/channels/[slug]/videos/page.tsx b/editor/app/channels/[slug]/videos/page.tsx @@ -36,6 +36,12 @@ import { computeVideoRows, readDataDirVideoIds } from "../lib/videoRowsServer"; import { normalizeBuckets } from "yt-dlp-transcript-common/views/pipeline/stageStatus"; import { attachCuratedTags } from "../lib/videoTagRows"; import { readChannelVideoTitles } from "yt-dlp-transcript-common/controller/videoTitles"; +import { channelMediaStall } from "yt-dlp-transcript-common/lib/channelMedia"; +import { + isDriveNotAnswering, + onDrive, +} from "yt-dlp-transcript-common/lib/storageHealth"; +import { MediaNotAnswering } from "../components/MediaNotAnswering"; export const dynamic = "force-dynamic"; @@ -100,6 +106,13 @@ export default async function ChannelVideosPage({ // A social channel has posts, not videos — there is no data directory to list // and nothing here would render. 404 rather than an empty workspace. if (isSocialChannel(config)) notFound(); + // THE LIST IS READ OFF THE DRIVE (a readdir of data/, a head read per title, + // the selected video's files), so a drive that is not answering is not read: + // the page says so instead. See MediaNotAnswering. + const stall = channelMediaStall(config); + if (stall) { + return <MediaNotAnswering slug={slug} stall={stall} what="The video list" />; + } const registry = getRegistry(); const existingQueues = registry.activeQueueNames(); @@ -131,14 +144,30 @@ export default async function ChannelVideosPage({ ); const channelDataDir = path.join(paths.channelsDir, slug, "data"); - const channelDataDirIds = await readDataDirVideoIds(channelDataDir); - // What each video is called: one key range over the transcript index, one - // read of the channel's metadata-scan.json, and a head read of - // metadata.info.json only for what those two did not name. Measured at - // ~80 ms for a synthetic 5,000-id channel (plans/release-8.md, slice V). - const titles = await readChannelVideoTitles(paths, slug, [ - ...new Set([...channelDataDirIds, ...(snapshot.undownloadedIds ?? [])]), - ]); + // THE READS OF THE DRIVE go through the watchdog when the channel is + // relocated: a drive that has not answered them in 3 s is marked stalled and + // the page says so instead (see MediaNotAnswering). + const drive = config.dataDir?.trim(); + const onMedia = <T,>(call: () => Promise<T>): Promise<T> => + drive ? onDrive(drive, call) : call(); + let channelDataDirIds: string[]; + let titles: Awaited<ReturnType<typeof readChannelVideoTitles>>; + try { + [channelDataDirIds, titles] = await onMedia(async () => { + const ids = await readDataDirVideoIds(channelDataDir); + // What each video is called: one key range over the transcript index, + // one read of the channel's metadata-scan.json, and a head read of + // metadata.info.json only for what those two did not name. Measured at + // ~80 ms for a synthetic 5,000-id channel (plans/release-8.md, slice V). + const named = await readChannelVideoTitles(paths, slug, [ + ...new Set([...ids, ...(snapshot.undownloadedIds ?? [])]), + ]); + return [ids, named] as const; + }); + } catch (err) { + if (!isDriveNotAnswering(err)) throw err; + return <MediaNotAnswering slug={slug} stall={err.health} what="The video list" />; + } const rows = computeVideoRows({ channelDataDirIds, snapshot, @@ -188,8 +217,9 @@ export default async function ChannelVideosPage({ ); if (selectedVideoId) { const videoDir = path.join(channelDataDir, selectedVideoId); - const [dirData, title, outcome, availabilityRecord, cov, excludedTrunc] = - await Promise.all([ + let loadedVideo; + try { + loadedVideo = await onMedia(() => Promise.all([ loadVideoDir(channelDataDir, selectedVideoId), // Already read for the list — the title map covers every row. Promise.resolve(titles.get(selectedVideoId)?.title ?? null), @@ -197,7 +227,13 @@ export default async function ChannelVideosPage({ loadAvailability(videoDir), readTranscriptCoverage(videoDir), isExcludedFromTruncatedCheck(videoDir), - ]); + ])); + } catch (err) { + if (!isDriveNotAnswering(err)) throw err; + return <MediaNotAnswering slug={slug} stall={err.health} what="The video list" />; + } + const [dirData, title, outcome, availabilityRecord, cov, excludedTrunc] = + loadedVideo; const coverage = cov ? { lastCueEnd: cov.cov.lastCueEnd, diff --git a/editor/app/channels/components/ChannelVolumeBar.tsx b/editor/app/channels/components/ChannelVolumeBar.tsx @@ -42,6 +42,9 @@ export type ChannelVolume = { // statfs of the root, or undefined when the root is not there (an unmounted // drive). Never inferred from a parent — see volumeFreeBytes. freeBytes?: number; + // "not answering since 11:35", while the health probe finds the drive not + // answering (its free space is then not asked either). + notAnswering?: string; }; export function ChannelVolumeBar({ @@ -96,10 +99,14 @@ export function ChannelVolumeBar({ detail={ `${v.channels} ch · ${formatBytes(v.bytes)}` + (v.unmeasured > 0 ? ` +${v.unmeasured}?` : "") + - ` · ${v.freeBytes === undefined ? "free —" : `${formatBytes(v.freeBytes)} free`}` + (v.notAnswering + ? ` · ${v.notAnswering}` + : ` · ${v.freeBytes === undefined ? "free —" : `${formatBytes(v.freeBytes)} free`}`) } title={ - v.freeBytes === undefined + v.notAnswering + ? `${v.label}: the drive is ${v.notAnswering}. Pages and polls do not touch it until it answers twice in a row; its channels read "not answering".` + : v.freeBytes === undefined ? `${v.label}: the root is not there — an unmounted drive reports no free space rather than its parent's.` : `${v.label}: ${v.channels} channel(s) hold ${formatBytes(v.bytes)}${ v.unmeasured > 0 diff --git a/editor/app/channels/page.tsx b/editor/app/channels/page.tsx @@ -8,6 +8,10 @@ import { import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { inspectChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia"; import { + notAnsweringText, + stalledLocation, +} from "yt-dlp-transcript-common/lib/storageHealth"; +import { getSite, listSiteIds, listSites, @@ -221,6 +225,14 @@ export default async function ChannelsPage({ // TWO SYSCALLS PER VOLUME, NOT A PROBE. See volumeFreeBytes: this table draws // 71 rows on every auto-refresh and the rule is that tables never shell out. const freeByVolume = await volumeFreeBytes({ paths, locations }); + // A DRIVE THAT IS NOT ANSWERING, said on its chip. From memory — the health + // pass (the block device's counters every 15 s) or the watchdog on a read is + // what found it (lib/storageHealth.ts); the table asks nothing. + const notAnsweringByVolume: Record<string, string | undefined> = {}; + for (const loc of locations) { + const stall = stalledLocation(loc); + if (stall) notAnsweringByVolume[loc.id] = notAnsweringText(stall); + } // ONE ROW PER CHANNEL, off the shared builder (common/views/channelRow.ts): // the dashboard and the operation pages build theirs the same way. The view // carries no `config` — this table is a client component, and a channel @@ -293,6 +305,9 @@ export default async function ChannelsPage({ bytes: measured.reduce((sum, c) => sum + (c.mediaBytes ?? 0), 0), unmeasured: rows.length - measured.length, freeBytes: freeByVolume[id], + ...(notAnsweringByVolume[id] + ? { notAnswering: notAnsweringByVolume[id] } + : {}), }; }) // The unnamed-root chip only exists when something is actually on one. diff --git a/editor/app/components/MediaLocationBadge.tsx b/editor/app/components/MediaLocationBadge.tsx @@ -10,18 +10,19 @@ import type { ChannelRowMedia } from "yt-dlp-transcript-common/views/channelRow" // the filesystem; the inspect() call that produces the location happens on the // server, once per row, and only its result travels. // -// WHAT THE FIVE STATUSES LOOK LIKE, and why there are only three appearances: +// WHAT THE SIX STATUSES LOOK LIKE, and why there are only three appearances: // // in-place → NOTHING. The overwhelming majority of channels are in place, // and a badge on every row saying "normal" is noise that makes // the two that matter harder to see, not easier. // ok → neutral. Relocated and reachable is a fact worth stating (the // bytes are not on the corpus disk) but it is not a problem. -// everything → red. unreachable, in-transition and inconsistent are all -// else "do not trust what this channel's dirs say right now": the +// everything → red. unreachable, in-transition, inconsistent and stalled are +// else all "do not trust what this channel's dirs say right now": the // first because the drive is not mounted, the second because a // move is half-done, the third because disk and config disagree -// and nothing here is willing to guess which one is right. +// and nothing here is willing to guess which one is right, the +// fourth because the drive is mounted and not answering. // // The `detail` string is the operator's prose from inspect() — the drive path, // the phase, the disagreement — and it goes on `title` so a row badge carries @@ -52,6 +53,7 @@ const LABELS: Record<ChannelMediaStatus, string | null> = { unreachable: "Media unreachable", "in-transition": "Media moving", inconsistent: "Media inconsistent", + stalled: "Media not answering", }; // The one-word state, for the compact rendering of a NAMED location: "on @@ -63,6 +65,7 @@ const SHORT_STATUS: Record<ChannelMediaStatus, string | null> = { unreachable: "unreachable", "in-transition": "moving", inconsistent: "inconsistent", + stalled: "not answering", }; // Null means "draw nothing" — an in-place channel, or no location at all (a diff --git a/editor/app/saved-videos/page.tsx b/editor/app/saved-videos/page.tsx @@ -22,7 +22,11 @@ export default async function SavedVideosPage() { const paths = getPaths(); const settings = getSettings(); const backup = settings.savedVideoBackup; - const entries = await listSavedVideos({ paths }); + // Channels whose drive is not answering are not read (one pointer read per + // video dir, each of which would wait on it); the page names them. + const notAnswering: string[] = []; + const entries = await listSavedVideos({ paths, notAnswering }); + notAnswering.sort(); const scheduler = await readSchedulerState(paths); const byChannel = new Map<string, ChannelSummary>(); @@ -96,6 +100,21 @@ export default async function SavedVideosPage() { <section aria-label="per-channel saved videos" className="flex flex-col gap-2"> <h2 className="text-base font-semibold">By channel</h2> + {notAnswering.length > 0 && ( + <p + role="status" + aria-label="saved videos not read" + className="rounded border border-destructive/50 bg-destructive/5 px-3 py-2 text-sm text-destructive" + > + Not read, because the drive their media is on is not answering:{" "} + {notAnswering.join(", ")}. Their saved videos are not in the counts + above until it answers again (see{" "} + <Link href="/storage" className="underline"> + Storage + </Link> + ). + </p> + )} {channels.length === 0 ? ( <p className="text-sm text-muted-foreground"> No saved videos yet. Set a channel&apos;s keep-latest window, then run diff --git a/editor/app/storage/actions.ts b/editor/app/storage/actions.ts @@ -33,6 +33,12 @@ import { enqueueRepointJob } from "./lib/repointJob"; import { enqueueSavedVideosRelocation } from "./lib/savedVideosJob"; import { enqueueEvictClipWindows } from "./lib/evictClipsJob"; import { savedVideosStoreBusyReason } from "./lib/storeBusy"; +import { refreshLocationHealth } from "yt-dlp-transcript-common/controller/storageWatch"; +import { forgetChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia"; +import { + locationHealth, + notAnsweringText, +} from "yt-dlp-transcript-common/lib/storageHealth"; // THE SIX THINGS AN OPERATOR MAY DO TO A STORAGE LOCATION. // @@ -244,6 +250,24 @@ export async function refreshStorageLocationAction( const settings = getSettings(); const location = settings.storage.locations.find((l) => l.id === id); if (!location) return { ok: false, error: `There is no storage location "${id}".` }; + // IS IT ANSWERING, asked first and without touching the drive (the block + // device's counters in /sys, or a child `stat` raced against 3 s where no + // device can be named): the probe below runs in-process, and on a stalled + // drive it is not asked at all. The counters give no answer within 10 s of + // the pass's last sample. The channels' remembered answers go too — the + // operator has just done something about the drive. + await refreshLocationHealth(location); + forgetChannelMedia(); + const health = locationHealth(location.id); + if (health?.state === "stalled") { + revalidateStorage(); + return { + ok: true, + note: + `${location.label}: ${notAnsweringText(health)} — ${health.cause ?? "its root did not answer"}. ` + + `Pages skip this drive until it answers twice in a row (checked every 15 s).`, + }; + } const probe = await probeLocationMemo(location, paths, { refresh: true }); const wrote = await recordProbedIdentity({ locationId: id, probe }); diff --git a/editor/app/storage/buildStorage.ts b/editor/app/storage/buildStorage.ts @@ -3,6 +3,12 @@ import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { getFreeBytes } from "yt-dlp-transcript-common/lib/diskSpace"; import { udisksctlAvailable } from "yt-dlp-transcript-common/lib/storageVolumes"; +import { + isDriveNotAnswering, + notAnsweringText, + onDrive, + stalledLocation, +} from "yt-dlp-transcript-common/lib/storageHealth"; import { listChannelBriefs } from "yt-dlp-transcript-common/controller/channels"; import { channelsOnLocation, @@ -81,11 +87,40 @@ export async function buildStorage(): Promise<StorageRowsPayload> { // situation the operator opened the page to understand. The store's SIZE is // the only thing cached; its location, status and marker are read fresh every // render, because those are the safety facts. + // THE DRIVES THAT ARE NOT ANSWERING, in words. From memory: the health pass + // (the block device's counters every 15 s) or the watchdog on a page's read + // is what found it (lib/storageHealth.ts). + const now = Date.now(); + const notAnswering: Record<string, string> = {}; + for (const loc of locations) { + const stall = stalledLocation(loc); + if (stall) { + const watched = + stall.detector === "counters" + ? " Watched through its disk's request counters." + : stall.detector === "stat" + ? " Watched with a stat of its root (no disk could be named here)." + : ""; + notAnswering[loc.id] = + `${notAnsweringText(stall, now)} — ${stall.cause ?? "its root did not answer"}. ` + + `Pages and polls skip this drive until it answers twice in a row.${watched}`; + } + } const store = await inspectSavedVideosStore(paths, settings); - const storeMeasured = - store.status === "unreachable" || store.status === "in-transition" - ? { bytes: 0, files: 0 } - : await measureTreeCached(paths.savedVideosDir); + // A store on a location walks that location's drive: through the watchdog, + // so a drive that stops answering mid-walk leaves the size at 0 (the store's + // status line says why on the next render) instead of holding the page. + const storeLocation = locations.find((l) => l.id === store.locationId); + let storeMeasured = { bytes: 0, files: 0 }; + if (store.status !== "unreachable" && store.status !== "in-transition") { + try { + storeMeasured = await (storeLocation + ? onDrive(storeLocation, () => measureTreeCached(paths.savedVideosDir)) + : measureTreeCached(paths.savedVideosDir)); + } catch (err) { + if (!isDriveNotAnswering(err)) throw err; + } + } return buildStorageRows({ locations, savedVideos: { @@ -106,9 +141,10 @@ export async function buildStorage(): Promise<StorageRowsPayload> { }, defaultLocationId: settings.storage.defaultLocationId, probes, + notAnswering, rollups, registry: getRegistry(), udisksctlAvailable: udisksctl, - now: Date.now(), + now, }); } diff --git a/editor/app/storage/components/StorageLocationsTable.tsx b/editor/app/storage/components/StorageLocationsTable.tsx @@ -254,6 +254,15 @@ function LocationCard({ <dd aria-label="location last probe">{formatAge(row.lastProbeAgeMs)}</dd> </dl> + {row.notAnswering && ( + <p + role="status" + aria-label="location not answering" + className="text-xs rounded border border-destructive/50 bg-destructive/5 px-3 py-2 text-destructive" + > + {row.notAnswering} + </p> + )} {row.warning && ( <p role="status" diff --git a/editor/instrumentation.ts b/editor/instrumentation.ts @@ -4,8 +4,10 @@ // // See editor/app/scheduler/heartbeat.ts and SCHEDULED_SYNC.md. // -// Everything armed here except the shutdown reaper is skipped when the process -// boots idle (ARCHILYZER_IDLE_BOOT) — see isIdleBoot below. +// Everything armed here that starts or writes work is skipped when the process +// boots idle (ARCHILYZER_IDLE_BOOT) — see isIdleBoot below. What stays armed +// only stops work or only reads: the shutdown reaper, the persisted-pause +// restore, the storage boot probe and the drive health pass. // // The one STATIC import in this file, and safe as one because idleBoot.ts // imports nothing and touches no Node API: the Edge bundle's static Node-API @@ -83,8 +85,25 @@ export async function register() { /* a failed re-pause must not block server readiness */ } - // ONE PASS OVER THE STORAGE LOCATIONS, and it runs on an IDLE BOOT TOO — - // the only thing below the reaper that does. + // THE DRIVE HEALTH PASS, every 15 s — and ON AN IDLE BOOT TOO, like the + // probe below. It reads each location's block device counters (never the + // drive) and keeps, in memory only, which drives are not answering; every + // page and poll asks it before touching a drive, and its watchdog marks a + // drive a page reached and got no answer from. It writes nothing and starts + // no work, and without it nothing would ever clear such a mark. The + // five-minute pass that may auto-pause channels is armed below the idle gate. + // See common/controller/storageWatch.ts and common/lib/storageHealth.ts. + try { + const { startStorageHealthWatch } = await import( + "yt-dlp-transcript-common/controller/storageWatch" + ); + startStorageHealthWatch({ log: (line) => console.log(line) }); + } catch { + /* a health pass that fails to arm must not block server readiness */ + } + + // ONE PASS OVER THE STORAGE LOCATIONS, and it runs on an IDLE BOOT TOO: it + // only reads, like the health pass above. // // The pass probes each location (is the disk here, and if not, where?) and, // for a location the operator armed with `autoRepoint`, re-points it to @@ -165,10 +184,11 @@ export async function register() { // is the thing that looks, on a five-minute cadence, and auto-pauses (and // later restores) the channels on a location that is not there. // - // BELOW THE IDLE GATE, deliberately, and unlike the boot probe above: this - // one WRITES settings.channelPriority, and a container pointed at somebody - // else's corpus for the first time has no business rewriting that corpus's - // priority document. See common/controller/storageWatch.ts. + // BELOW THE IDLE GATE, deliberately, and unlike the boot probe and the + // health pass above: this one WRITES settings.channelPriority, and a + // container pointed at somebody else's corpus for the first time has no + // business rewriting that corpus's priority document. See + // common/controller/storageWatch.ts. try { const { startStorageWatch } = await import( "yt-dlp-transcript-common/controller/storageWatch" diff --git a/editor/package.json b/editor/package.json @@ -8,7 +8,7 @@ "dev:test": "WORKER_TOKEN=test-worker-token TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json EDITOR_CHANGELOG_FILE=$(pwd)/test-changelog.md EXPORT_CHANGELOG_FILE=$(pwd)/test-export-changelog.md YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs GALLERY_DL_BIN=$(pwd)/e2e/fixtures/bin/fake-gallery-dl.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null DIARIZE_BIN=$(pwd)/e2e/fixtures/bin/fake-diarize.mjs FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs FFPROBE_BIN=$(pwd)/e2e/fixtures/bin/fake-ffprobe.mjs OLLAMA_URL=http://127.0.0.1:${OLLAMA_STUB_PORT:-11435} CLAUDE_BIN=$(pwd)/e2e/fixtures/bin/fake-claude.mjs FINDMNT_BIN=$(pwd)/e2e/fixtures/bin/fake-findmnt.mjs UDISKSCTL_BIN=$(pwd)/e2e/fixtures/bin/fake-udisksctl.mjs next dev --port ${PORT:-3011}", "start:test": "WORKER_TOKEN=test-worker-token TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json EDITOR_CHANGELOG_FILE=$(pwd)/test-changelog.md EXPORT_CHANGELOG_FILE=$(pwd)/test-export-changelog.md YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs GALLERY_DL_BIN=$(pwd)/e2e/fixtures/bin/fake-gallery-dl.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null DIARIZE_BIN=$(pwd)/e2e/fixtures/bin/fake-diarize.mjs FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs FFPROBE_BIN=$(pwd)/e2e/fixtures/bin/fake-ffprobe.mjs OLLAMA_URL=http://127.0.0.1:${OLLAMA_STUB_PORT:-11435} CLAUDE_BIN=$(pwd)/e2e/fixtures/bin/fake-claude.mjs FINDMNT_BIN=$(pwd)/e2e/fixtures/bin/fake-findmnt.mjs UDISKSCTL_BIN=$(pwd)/e2e/fixtures/bin/fake-udisksctl.mjs next start --port ${PORT:-3011}", "build": "next build", - "start": "next start --port ${EDITOR_PORT:-3001}", + "start": "UV_THREADPOOL_SIZE=${UV_THREADPOOL_SIZE:-16} next start --port ${EDITOR_PORT:-3001}", "lint": "eslint", "e2e": "node ../scripts/queue-lock.mjs --ports PORT:3011,EXPORT_PORT:3010,OLLAMA_STUB_PORT:11435 -- playwright test", "e2e:ui": "playwright test --ui" diff --git a/homepage/CHANGELOG.md b/homepage/CHANGELOG.md @@ -1,6 +1,7 @@ # Homepage Changelog ## [Unreleased] +- **The growth chart draws its smallest instances together as Other.** Two or more instances that each hold under 5% of the chart's total are one band, **Other**, on top of the stack, in a near-neutral grey of its own (`--chart-other`: 7.36:1 on the Light ground, 3.22:1 on the Dark one, and apart from every instance colour for colour-blind readers); an instance at exactly 5% keeps its band, and a single one under 5% is not grouped. The other instances keep their bands and their colours. The legend lists them and Other; the caption says what Other is and the chart's description names the instances in it; every month's hover title and the Numbers by year table still name every instance. At this release's numbers Hasanalyzer, Rekietalyzer and Jasolyzer are Other. The instance cards and `/stats` are unchanged. The e2e fixture's fifth site transcribes 4 a day rather than 5, so two of its six sites are grouped. - **An unlisted site is not on the homepage.** A site whose settings turn off **List on the Archilyzer homepage and hub** (`listed: false`) has no Official Instances card, chart series, `/stats` entry or recent item, is not in `channel-sites.json` or `stats/`, and the channels only it carries count in none of the numbers, the headline totals included. The summary's version is 6. The e2e fixture has a seventh, unlisted site that no page names. - **`/#instances` goes straight to Official Instances.** The section carries `id="instances"`, clear of the sticky header, and every archive's header now links there (`INSTANCES_URL` in `common/lib/project.ts`). With no sites the link lands on the top of the page. diff --git a/homepage/app/components/ArchiveGrowthChart.tsx b/homepage/app/components/ArchiveGrowthChart.tsx @@ -3,8 +3,16 @@ import type { HomepageSummarySite, } from "yt-dlp-transcript-common/lib/homepageSummary"; import { monthLabel } from "yt-dlp-transcript-common/lib/homepageChart"; -import { siteChartColors } from "yt-dlp-transcript-common/lib/siteColor"; -import { PLOT_SIZES, gapSegments, growthStack, runsOf } from "../lib/growthGaps"; +import { + FOLD_PERCENT, + OTHER_LABEL, + PLOT_SIZES, + gapSegments, + growthLayers, + growthStack, + layerColors, + runsOf, +} from "../lib/growthGaps"; // The front page's showpiece: every official instance's back catalogue as // stacked strata, one month per step, from the oldest upload to the last @@ -36,6 +44,28 @@ import { PLOT_SIZES, gapSegments, growthStack, runsOf } from "../lib/growthGaps" // The legend above the plot wears the same colours. The layers stack in the // summary's order (STACK_ORDER, unchanged). // +// OTHER (release 14, slice CF; lib/growthGaps.ts). Two or more sites each +// under 5 % of the chart's total fold into ONE band, "Other", on top of the +// stack. The kept sites keep their slots — siteChartColors runs over EVERY +// site, so a fold never repaints one — and Other wears its own near-neutral +// grey, `--chart-other` (tokens.css: Light #3e545c, Dark #62625c; 7.36 / +// 3.22:1 on the ground, 7.99 / 3.06:1 on the chart surface), not the axis +// labels' colour. The legend shows the kept sites and Other; the hover title +// and the table name every site, and the image's label and the caption say +// what Other holds. A grey is below the validator's chroma floor by +// definition (it is the de-emphasis role, not a categorical slot); against +// each slot it can sit on, Other is (normal ΔE / worst of protan and deutan, +// light · dark): +// blue 21.9 / 21.4 · 17.8 / 17.9 — the only neighbour in today's data +// (Bonnellyzer, the top kept site in every month Other has data); +// green 17.4 / 15.2 · 19.2 / 15.7; violet 16.2 / 13.4 · 23.4 / 21.5; +// amber 22.1 / 18.2 · 20.3 / 18.1; magenta 24.5 / 9.1 · 19.5 / 7.3; +// rust 14.1 / 10.5 · 13.5 / 9.3. +// Every CVD pair clears the floor (6), and on Light the target (8); Dark's +// magenta is in the 6–8 band and rust is under the normal-vision 15 on both +// bases: there the gap between the bands, Other's place on top, the legend +// and the table carry it. +// // VALIDATED (the dataviz skill's validator, each base's values on its // --chart-surface and on the page ground), with the operator's accents // (Jeralyzer Brass, Anilyzer Sakura, Bonnellyzer Blue, Hasanalyzer Violet, @@ -69,23 +99,32 @@ export function ArchiveGrowthChart({ const n = months.length; if (n === 0 || sites.length === 0) return null; - const { totals, peak, yMax, ticks, order, stack: stacked } = growthStack(months, sites); + // The bands, bottom-up: the kept sites, then Other when two or more fold. + const { + totals, + peak, + yMax, + ticks, + layers: bands, + stack: stacked, + } = growthStack(months, sites, growthLayers(months, sites)); if (peak === 0) return null; const x = (i: number) => (n === 1 ? W / 2 : (i / (n - 1)) * W); const y = (v: number) => H - (v / yMax) * H; const r = (v: number) => Math.round(v * 10) / 10; - // One colour per site, in `sites` order (the legend's too). - const colors = siteChartColors(sites); + // One colour per band, in stack order (the legend's too). + const colors = layerColors(bands, sites); + const folded = bands.find((b) => b.other)?.sites.map((si) => sites[si].siteTitle) ?? []; const layers = stacked.map(({ lo, hi }, k) => { - const si = order[k]; const top = hi.map((v, i) => `${r(x(i))},${r(y(v))}`); const bottom = lo.map((v, i) => `${r(x(i))},${r(y(v))}`).reverse(); return { - site: sites[si], - color: colors[si], + key: bands[k].key, + label: bands[k].other ? OTHER_LABEL : sites[bands[k].sites[0]].siteTitle, + color: colors[k], top, area: `M${top.join("L")}L${bottom.join("L")}Z`, }; @@ -95,7 +134,7 @@ export function ArchiveGrowthChart({ const gapSets = PLOT_SIZES.map((size) => { const segments = gapSegments(stacked, yMax, size); const paths = layers.map((l, k) => ({ - key: l.site.siteId, + key: l.key, d: runsOf(segments[k]) .map((run) => `M${[...run, run.at(-1)! + 1].map((i) => l.top[i]).join("L")}`) .join(""), @@ -113,11 +152,18 @@ export function ArchiveGrowthChart({ const last = monthLabel(months[n - 1].month); const peakIdx = totals.indexOf(peak); const all = totals.reduce((a, b) => a + b, 0); + // The legend is hidden from assistive tech, so the image's label says what + // Other holds; the caption says what it is. + const otherNote = folded.length + ? `${folded.slice(0, -1).join(", ")} and ${folded.at(-1)}, each under ` + + `${FOLD_PERCENT}% of the total, are drawn together as ${OTHER_LABEL}.` + : ""; const label = `Stacked area chart of ${all.toLocaleString()} transcripts by the month each ` + `video was published, ${first} to ${last}, across ${sites.length} official ` + `instances. The busiest month is ${monthLabel(months[peakIdx].month)}, ` + - `with ${peak.toLocaleString()}.`; + `with ${peak.toLocaleString()}.` + + (otherNote ? ` ${otherNote}` : ""); // Per-year table for anyone who wants the numbers rather than the shape. const byYear = new Map<string, Record<string, number>>(); @@ -130,10 +176,10 @@ export function ArchiveGrowthChart({ return ( <figure className="flex flex-col gap-4"> <ul className="flex flex-wrap gap-x-5 gap-y-2 list-none" aria-hidden="true"> - {sites.map((s, i) => ( - <li key={s.siteId} className="flex items-center gap-2 text-sm text-[var(--muted-foreground)]"> - <span className="h-2.5 w-2.5 shrink-0 rounded-[2px]" style={{ backgroundColor: colors[i] }} /> - {s.siteTitle} + {layers.map((l) => ( + <li key={l.key} className="flex items-center gap-2 text-sm text-[var(--muted-foreground)]"> + <span className="h-2.5 w-2.5 shrink-0 rounded-[2px]" style={{ backgroundColor: l.color }} /> + {l.label} </li> ))} </ul> @@ -163,7 +209,7 @@ export function ArchiveGrowthChart({ /> ))} {layers.map((l) => ( - <path key={l.site.siteId} d={l.area} fill={l.color} /> + <path key={l.key} d={l.area} fill={l.color} /> ))} {/* THE SURFACE GAP (the marks spec): touching bands are parted by a 2 px gap in the colour behind the plot — the page ground, as @@ -194,6 +240,8 @@ export function ArchiveGrowthChart({ ))} {months.map((m, i) => { const w = W / n; + // Every site with data that month, by name — a folded one too: + // the title and the table are where Other's sites are found. const parts = sites .map((s) => [s.siteTitle, m.bySite[s.siteId] ?? 0] as const) .filter(([, v]) => v > 0) @@ -245,6 +293,8 @@ export function ArchiveGrowthChart({ <figcaption className="text-sm text-[var(--muted-foreground)]"> Transcripts by the month each video was published, all official instances. + {folded.length > 0 && + ` Instances under ${FOLD_PERCENT}% of the total are drawn together as ${OTHER_LABEL}.`} </figcaption> <details className="text-sm"> <summary className="cursor-pointer text-[var(--muted-foreground)] underline decoration-[var(--border-strong)] underline-offset-4 hover:text-[var(--foreground)]"> diff --git a/homepage/app/lib/growthGaps.test.ts b/homepage/app/lib/growthGaps.test.ts @@ -1,13 +1,20 @@ import { test } from "node:test"; import assert from "node:assert/strict"; +import { siteChartColors } from "yt-dlp-transcript-common/lib/siteColor"; import { buildFixtureSummary } from "../../e2e/fixture-summary"; import { GAP_PX, MIN_KEEP_PX, + OTHER_COLOR, + OTHER_KEY, PLOT_SIZES, bandAbove, + foldedSites, gapSegments, + growthLayers, growthStack, + layerColors, + ownLayers, type Band, type PlotSize, } from "./growthGaps"; @@ -83,15 +90,30 @@ function slivers(): { stack: Band[]; yMax: number } { return { stack, yMax }; } -test("the fixture summary: gaps are drawn, and no band is ever covered, at every height", () => { +test("the fixture summary, as the chart draws it (Other on top): gaps are drawn, and no band is ever covered, at every height", () => { const summary = buildFixtureSummary(); - const { stack, yMax } = growthStack(summary.monthly ?? [], summary.sites); + const months = summary.monthly ?? []; + const layers = growthLayers(months, summary.sites); + // The fixture's last two sites are under 5 %: they fold. + assert.deepEqual(foldedSites(months, summary.sites), [4, 5]); + assert.equal(layers.at(-1)!.key, OTHER_KEY); + const { stack, yMax } = growthStack(months, summary.sites, layers); let drawn = 0; + let underOther = 0; for (const size of PLOT_SIZES) { - drawn += gapSegments(stack, yMax, size).flat().length; + const segs = gapSegments(stack, yMax, size); + drawn += segs.flat().length; + // The gap under Other parts it from the top kept site; nothing sits on + // Other, so its own upper edge has none. + underOther += segs[stack.length - 2].filter((i) => bandAbove(stack, stack.length - 2, i) === stack.length - 1).length; + assert.deepEqual(segs[stack.length - 1], [], `${size.px}px: a gap along Other's top`); assert.deepEqual(covered(stack, yMax, size), [], `${size.px}px`); } assert.ok(drawn > 0, "no gap drawn at all"); + assert.ok(underOther > 0, "no gap between the top kept site and Other"); + // Unfolded, the same summary keeps its promise too. + const own = growthStack(months, summary.sites); + for (const size of PLOT_SIZES) assert.deepEqual(covered(own.stack, own.yMax, size), [], `unfolded ${size.px}px`); }); test("slivers beside large bands, with one-month spikes: no band is ever covered", () => { @@ -126,3 +148,142 @@ test("no gap along the stack's top, and none under the run length", () => { assert.deepEqual(gapSegments(stack, yMax, size), [[], []], `${size.px}px`); } }); + +// ── The fold ────────────────────────────────────────────────────────────────── +// +// A site under 5 % of the placed total (every site's transcripts over the whole +// plotted range) folds into ONE Other band on top, when two or more do. The +// range's sums are what count: `monthsOf` spreads each site's sum over three +// months, so no single month decides. + +type Named = { siteId: string; siteTitle: string; accentId?: string }; + +function sitesOf(ids: readonly string[], accents: readonly (string | undefined)[] = []): Named[] { + return ids.map((siteId, i) => ({ siteId, siteTitle: siteId.toUpperCase(), accentId: accents[i] })); +} + +function monthsOf(sites: readonly Named[], sums: readonly number[]) { + // Thirds, the remainder in the last month. + return [0, 1, 2].map((k) => ({ + bySite: Object.fromEntries( + sites.map((s, i) => [s.siteId, k < 2 ? Math.floor(sums[i] / 3) : sums[i] - 2 * Math.floor(sums[i] / 3)]), + ), + })); +} + +test("the fold with today's proportions: the three largest keep their bands, the other three are one Other band on top", () => { + // The family's shares in the 2026-09-28 summary, in its order and with its + // accents: 42.03, 38.40, 9.18, 4.33, 3.79, 2.26 %. + const sites = sitesOf(["a", "b", "c", "d", "e", "f"], ["brass", "sakura", "blue", "violet", "green", "vermilion"]); + const sums = [4203, 3840, 918, 433, 379, 226]; + const months = monthsOf(sites, sums); + assert.deepEqual(foldedSites(months, sites), [3, 4, 5]); + const layers = growthLayers(months, sites); + assert.deepEqual( + layers.map((l) => [l.key, l.sites, l.other]), + [ + ["a", [0], false], + ["b", [1], false], + ["c", [2], false], + [OTHER_KEY, [3, 4, 5], true], + ], + ); + // Other is the sum of its sites, on top: the stack's top is every month's + // total, as before the fold. + const { stack, totals } = growthStack(months, sites, layers); + assert.deepEqual(stack[3].hi.map((v, i) => v - stack[3].lo[i]), months.map((m) => m.bySite.d + m.bySite.e + m.bySite.f)); + assert.deepEqual(stack[3].hi, totals); + // Colours: the kept sites their families' slots, Other the neutral grey. + assert.deepEqual(layerColors(layers, sites), ["var(--chart-4)", "var(--chart-5)", "var(--chart-1)", OTHER_COLOR]); + assert.equal(OTHER_COLOR, "var(--chart-other)"); +}); + +test("the threshold's edge: exactly 5 % keeps its band, just under folds", () => { + const sites = sitesOf(["a", "b", "c", "d"]); + // b and c are 999 of 20,000 (4.995 %), d exactly 1,000 (5 %). + assert.deepEqual(foldedSites(monthsOf(sites, [17_002, 999, 999, 1_000]), sites), [1, 2]); + // Two sites at exactly 5 %: neither folds. + const three = sitesOf(["a", "b", "c"]); + assert.deepEqual(foldedSites(monthsOf(three, [18_000, 1_000, 1_000]), three), []); + assert.deepEqual(growthLayers(monthsOf(three, [18_000, 1_000, 1_000]), three), ownLayers(three)); +}); + +test("a fold of one is no fold: a single site under 5 % keeps its band", () => { + const sites = sitesOf(["a", "b", "c"]); + const months = monthsOf(sites, [60, 36, 4]); + assert.deepEqual(foldedSites(months, sites), []); + assert.deepEqual(growthLayers(months, sites), ownLayers(sites)); + assert.ok(growthLayers(months, sites).every((l) => !l.other)); +}); + +test("no site under 5 %: every site keeps its band, as before the fold", () => { + const sites = sitesOf(["a", "b", "c"]); + const months = monthsOf(sites, [50, 30, 20]); + assert.deepEqual(foldedSites(months, sites), []); + const layers = growthLayers(months, sites); + assert.deepEqual(layers, ownLayers(sites)); + // The same stack as growthStack's default. + assert.deepEqual(growthStack(months, sites, layers), growthStack(months, sites)); +}); + +test("all but one under 5 %: one kept band and one Other band", () => { + const sites = sitesOf(["a", "b", "c", "d", "e", "f"]); + const months = monthsOf(sites, [80, 4, 4, 4, 4, 4]); + const layers = growthLayers(months, sites); + assert.deepEqual( + layers.map((l) => [l.key, l.sites]), + [ + ["a", [0]], + [OTHER_KEY, [1, 2, 3, 4, 5]], + ], + ); + const { stack, totals } = growthStack(months, sites, layers); + assert.equal(stack.length, 2); + assert.deepEqual(stack[1].hi, totals); +}); + +test("a site with nothing in the plotted range is under 5 %, and folds with another", () => { + const sites = sitesOf(["a", "b", "c"]); + assert.deepEqual(foldedSites(monthsOf(sites, [96, 4, 0]), sites), [1, 2]); + // Nothing plotted at all: nothing folds. + assert.deepEqual(foldedSites(monthsOf(sites, [0, 0, 0]), sites), []); +}); + +test("every site under 5 % (over twenty sites): every site folds, one Other band", () => { + const ids = Array.from({ length: 25 }, (_, i) => `s${i}`); + const sites = sitesOf(ids); + const layers = growthLayers(monthsOf(sites, ids.map(() => 40)), sites); + assert.deepEqual( + layers.map((l) => [l.key, l.sites.length]), + [[OTHER_KEY, 25]], + ); +}); + +test("a kept site's colour is the one it wears unfolded, whoever folds", () => { + // No accents: each site's slot is its index's (siteChartColors), so a list + // of the kept sites alone would repaint d — the chart runs it over them all. + const sites = sitesOf(["a", "b", "c", "d"]); + const months = monthsOf(sites, [500, 20, 20, 460]); + const layers = growthLayers(months, sites); + assert.deepEqual( + layers.map((l) => l.key), + ["a", "d", OTHER_KEY], + ); + const unfolded = layerColors(ownLayers(sites), sites); + assert.deepEqual(unfolded, siteChartColors(sites)); + const colours = layerColors(layers, sites); + assert.deepEqual(colours, [unfolded[0], unfolded[3], OTHER_COLOR]); + assert.notEqual(colours[1], siteChartColors([sites[0], sites[3]])[1], "the kept list alone would repaint d"); + // With accents, the same: each kept site keeps its family's slot. + const named = sitesOf(["a", "b", "c", "d", "e", "f"], ["brass", "sakura", "blue", "violet", "green", "vermilion"]); + const all = siteChartColors(named); + const nm = monthsOf(named, [4203, 3840, 918, 433, 379, 226]); + const nl = growthLayers(nm, named); + assert.deepEqual( + layerColors(nl, named).slice(0, -1), + nl.slice(0, -1).map((l) => all[l.sites[0]]), + ); + // The fold never hands a kept site the grey, nor Other a site's slot. + assert.ok(!layerColors(nl, named).slice(0, -1).includes(OTHER_COLOR)); + assert.ok(!all.includes(OTHER_COLOR)); +}); diff --git a/homepage/app/lib/growthGaps.ts b/homepage/app/lib/growthGaps.ts @@ -1,16 +1,34 @@ -// THE GROWTH CHART'S SURFACE GAPS, worked out (ArchiveGrowthChart.tsx draws -// them). Pure: no React, no DOM, so a unit test can check every segment. +import { siteChartColors } from "yt-dlp-transcript-common/lib/siteColor"; + +// THE GROWTH CHART'S LAYERS AND SURFACE GAPS, worked out (ArchiveGrowthChart.tsx +// draws them). Pure: no React, no DOM, so a unit test can check every layer and +// every segment. +// +// THE FOLD. A site under FOLD_PERCENT % of the placed total — every site's +// transcripts over the whole plotted range, the sum the bands are placed from +// — is drawn in ONE "Other" band on top of the stack, in the chart's neutral +// grey (OTHER_COLOR). Exactly FOLD_PERCENT % keeps its band; the comparison is +// on integers (100 × a site's sum < FOLD_PERCENT × the total), so the edge is +// exact. A fold of one is no fold: with a single site under the line, every +// site keeps its band. The kept sites keep their colours: each is its +// siteChartColors slot over EVERY site, folded or not, so a site's colour never +// changes because another one folded. The legend shows the kept sites and +// Other; the hover title and the "Numbers by year" table still name every site. // -// The marks spec parts touching bands by a 2 px gap in the colour behind the -// plot. The chart draws it centred on each band's upper edge, so each of the -// two bands gives 1 px. A band too thin to give its pixel and keep one of its -// own colour must touch its neighbour instead, or the gap erases it. "Thin" is -// measured AT RIGHT ANGLES to the edge, where the stroke's 2 px are measured: -// on a steep edge a band's perpendicular thickness is its vertical height × -// cos θ, and on a one-month dip at a phone's width it is nearly nothing. And it -// is measured at the NARROWEST plot each of the chart's three heights is drawn -// at, where the edges are steepest. A gap is drawn along a segment only where -// both bands keep at least MIN_KEEP_PX there after every gap that touches them. +// THE GAPS. The marks spec parts touching bands by a 2 px gap in the colour +// behind the plot. The chart draws it centred on each band's upper edge, so +// each of the two bands gives 1 px. A band too thin to give its pixel and keep +// one of its own colour must touch its neighbour instead, or the gap erases +// it. "Thin" is measured AT RIGHT ANGLES to the edge, where the stroke's 2 px +// are measured: on a steep edge a band's perpendicular thickness is its +// vertical height × cos θ, and on a one-month dip at a phone's width it is +// nearly nothing. And it is measured at the NARROWEST plot each of the chart's +// three heights is drawn at, where the edges are steepest. A gap is drawn along +// a segment only where both bands keep at least MIN_KEEP_PX there after every +// gap that touches them. The gaps work on bands, not sites: Other is one band, +// the top one, so the gap under it parts it from the kept site beneath (the +// highest with a height that month), and its own upper edge, where nothing +// sits, has none. export const GAP_PX = 2; export const MIN_KEEP_PX = 1; @@ -34,6 +52,8 @@ export type PlotSize = (typeof PLOT_SIZES)[number]; // The stack, bottom-up: each band's lower and upper value per month. export type Band = { lo: readonly number[]; hi: readonly number[] }; +type Month = { bySite: Record<string, number | undefined> }; + // The layers stack in the summary's order (the first five as they come, any // more after them). const STACK_ORDER = [0, 1, 2, 3, 4]; @@ -43,6 +63,68 @@ function stackOrder(n: number): number[] { return [...head, ...tail]; } +// ── The fold ────────────────────────────────────────────────────────────────── + +export const FOLD_PERCENT = 5; +export const OTHER_LABEL = "Other"; +// Other's own chart token (tokens.css `--chart-other`), a near-neutral grey: +// 7.36:1 on the Light ground and 3.22:1 on the Dark one (7.99 / 3.06 on the +// chart surface). tokens.css has its separation from each chart slot. +export const OTHER_COLOR = "var(--chart-other)"; + +// One band of the stack: a site's own (`sites` is its index in the summary's +// list), or Other (the indexes of every folded site, in the list's order). +// `key` is the site's id, or OTHER_KEY, which no site id can be (an id is +// `[a-z0-9][a-z0-9-]*`). +export type GrowthLayer = { key: string; sites: readonly number[]; other: boolean }; +export const OTHER_KEY = "(other)"; + +// Each site's transcripts over the whole plotted range. +function siteSums(months: readonly Month[], sites: readonly { siteId: string }[]): number[] { + return sites.map((s) => months.reduce((a, m) => a + (m.bySite[s.siteId] ?? 0), 0)); +} + +// The sites folded into Other, as indexes into `sites` in its order: every +// site under FOLD_PERCENT % of the placed total, when there are two or more of +// them; none otherwise. +export function foldedSites(months: readonly Month[], sites: readonly { siteId: string }[]): number[] { + const sums = siteSums(months, sites); + const all = sums.reduce((a, b) => a + b, 0); + if (all <= 0) return []; + const small = sums.flatMap((v, i) => (100 * v < FOLD_PERCENT * all ? [i] : [])); + return small.length >= 2 ? small : []; +} + +// Every site its own band, in stack order: the chart before the fold. +export function ownLayers(sites: readonly { siteId: string }[]): GrowthLayer[] { + return stackOrder(sites.length).map((i) => ({ key: sites[i].siteId, sites: [i], other: false })); +} + +// The chart's bands, bottom-up: the kept sites in stack order, then Other on +// top when anything folds. +export function growthLayers( + months: readonly Month[], + sites: readonly { siteId: string }[], +): GrowthLayer[] { + const folded = foldedSites(months, sites); + if (folded.length === 0) return ownLayers(sites); + const out = new Set(folded); + return [ + ...ownLayers(sites).filter((l) => !out.has(l.sites[0])), + { key: OTHER_KEY, sites: folded, other: true }, + ]; +} + +// Each layer's colour: a kept site its siteChartColors slot over EVERY site +// (so folding never repaints it), Other the neutral grey. +export function layerColors( + layers: readonly GrowthLayer[], + sites: readonly { accentId?: string }[], +): string[] { + const chart = siteChartColors(sites); + return layers.map((l) => (l.other ? OTHER_COLOR : chart[l.sites[0]])); +} + // A clean tick step giving three or four gridlines under `max`. function niceStep(max: number): number { const raw = max / 3.5; @@ -53,12 +135,16 @@ function niceStep(max: number): number { return 10 * pow; } -// The chart's numbers: each month's total, the peak, the value scale (yMax -// and its gridlines), and the bands bottom-up (`order[k]` is band k's index in -// `sites`). +// ── The stack ───────────────────────────────────────────────────────────────── + +// The chart's numbers: each month's total (every site, folded or not), the +// peak, the value scale (yMax and its gridlines), and one band per layer, +// bottom-up (`stack[k]` is `layers[k]`'s; a layer's value in a month is the sum +// of its sites'). Without `layers`, every site is its own band. export function growthStack( - months: readonly { bySite: Record<string, number | undefined> }[], + months: readonly Month[], sites: readonly { siteId: string }[], + layers: readonly GrowthLayer[] = ownLayers(sites), ) { const n = months.length; const totals = months.map((m) => sites.reduce((a, s) => a + (m.bySite[s.siteId] ?? 0), 0)); @@ -67,16 +153,19 @@ export function growthStack( const yMax = Math.ceil(peak / step) * step; const ticks: number[] = []; if (peak > 0) for (let t = step; t <= yMax; t += step) ticks.push(t); - const order = stackOrder(sites.length); const base = new Array<number>(n).fill(0); - const stack: Band[] = order.map((si) => { + const stack: Band[] = layers.map((layer) => { const lo = base.slice(); - const hi = months.map((m, i) => (base[i] += m.bySite[sites[si].siteId] ?? 0)); + const hi = months.map( + (m, i) => (base[i] += layer.sites.reduce((a, si) => a + (m.bySite[sites[si].siteId] ?? 0), 0)), + ); return { lo, hi }; }); - return { totals, peak, yMax, ticks, order, stack }; + return { totals, peak, yMax, ticks, layers, stack }; } +// ── The gaps ────────────────────────────────────────────────────────────────── + const height = (b: Band, i: number) => b.hi[i] - b.lo[i]; // The band a gap along band k's upper edge would share at month i: the next diff --git a/homepage/e2e/fixture-summary.ts b/homepage/e2e/fixture-summary.ts @@ -27,6 +27,9 @@ import type { VideoStat } from "../../common/lib/stats"; // • every site transcribes every day of the last 200, so every line is full // and every series has a non-zero start for Indexed; // • uploads spread over 2019–2026, for the growth chart's months; +// • the last two sites each under 5 % of the growth chart's total (4.35 and +// 3.26 %), so the chart folds them into its Other band (release 14 slice +// CF; fixture-five transcribes 4 a day, not 5, for it); // • a seventh site, UNLISTED (site.json `listed: false`, release 14 slice // HS): its own two channels transcribe every day like the rest, and it // also exposes the first site's first channel. The summary names it nowhere @@ -65,7 +68,7 @@ export const FIXTURE_SITES: readonly FixtureSite[] = [ { siteId: "fixture-two", siteTitle: "Fixture Two", accent: FIXTURE_PALE_HEX, wordmarkLead: "Fixture", channels: 3, daily: 9 }, { siteId: "fixture-three", siteTitle: "Fixture Three", wordmarkLead: "Fix", channels: 3, daily: 7 }, { siteId: "fixture-four", siteTitle: "Fixture Four", channels: 2, daily: 8 }, - { siteId: "fixture-five", siteTitle: "Fixture Five", channels: 2, daily: 5 }, + { siteId: "fixture-five", siteTitle: "Fixture Five", channels: 2, daily: 4 }, { siteId: "fixture-six", siteTitle: "Fixture Six", accent: "vermilion", channels: 2, daily: 3 }, ]; diff --git a/homepage/e2e/growth-chart.spec.ts b/homepage/e2e/growth-chart.spec.ts @@ -1,13 +1,17 @@ import { test, expect, type Page } from "@playwright/test"; import { buildFixtureSummary } from "./fixture-summary"; import { painted, rgbOf } from "../../common/testing/chartPixels"; +import { siteChartColors } from "../../common/lib/siteColor"; +import { OTHER_COLOR, growthLayers } from "../app/lib/growthGaps"; // The growth chart's bands are parted by the marks spec's SURFACE GAP: 2 px // along each band's upper edge, in the colour behind the plot — the page // ground, which the chart sits on — never a line in the text colour // (ArchiveGrowthChart.tsx, globals.css `.growth-gap`), and the reader's Canvas // in forced colours. The fixture summary (fixture-summary.ts) has six sites -// with monthly data, so the chart and its gaps render. +// with monthly data, so the chart and its gaps render; its last two are each +// under 5 % of the chart's total, so they are drawn as one Other band on top +// (lib/growthGaps.ts, release 14 slice CF). const BASE_KEY = "ytdlp-tb:base"; @@ -87,8 +91,9 @@ for (const [width, shown] of [ // THE DATA IS WHAT IS PAINTED. Read back from a screenshot of the plot: at the // busiest month the stack's topmost painted row is within 1 px of where the -// month's true total sits on the value scale, and every site with data shows -// pixels of its own colour — the gaps take no band away. +// month's true total sits on the value scale, and every band with data — each +// kept site's and Other's — shows pixels of its own colour: the gaps take no +// band away. for (const width of [390, 1280]) { test(`${width} px: the chart paints its peak at its true height and every band`, async ({ page }) => { await page.setViewportSize({ width, height: 900 }); @@ -96,6 +101,7 @@ for (const width of [390, 1280]) { const summary = buildFixtureSummary(); const months = summary.monthly ?? []; const sites = summary.sites; + const layers = growthLayers(months, sites); const totals = months.map((m) => sites.reduce((a, x) => a + (m.bySite[x.siteId] ?? 0), 0)); const peak = Math.max(...totals); const peakAt = totals.indexOf(peak); @@ -111,10 +117,14 @@ for (const width of [390, 1280]) { const expected = (1 - peak / yMax) * box.height; const x = (peakAt / (months.length - 1)) * box.width; const ground = rgbOf(await page.evaluate(() => getComputedStyle(document.body).backgroundColor)); + // The legend's swatches, one per band, bottom-up (kept sites, then Other). const swatches = await page .locator("figure ul li span") .evaluateAll((els) => els.map((e) => getComputedStyle(e).backgroundColor)); - const withData = sites.map((st) => months.some((m) => (m.bySite[st.siteId] ?? 0) > 0)); + expect(swatches).toHaveLength(layers.length); + const withData = layers.map((l) => + months.some((m) => l.sites.some((si) => (m.bySite[sites[si].siteId] ?? 0) > 0)), + ); const shot = await painted(page, plot, { columns: [x - 1, x, x + 1], ground, @@ -124,8 +134,116 @@ for (const width of [390, 1280]) { }); const top = Math.min(...shot.tops.filter((t): t is number => t !== null)); expect(Math.abs(top - expected), `top ${top} vs ${expected.toFixed(1)}`).toBeLessThanOrEqual(1); - sites.forEach((st, i) => { - if (withData[i]) expect(shot.counts[i].columns, `${st.siteTitle}'s colour`).toBeGreaterThan(0); + layers.forEach((l, k) => { + if (withData[k]) expect(shot.counts[k].columns, `${l.key}'s colour`).toBeGreaterThan(0); }); }); } + +const resolveFill = (page: Page, css: string) => + page.evaluate((c) => { + const el = document.createElement("span"); + el.style.backgroundColor = c; + document.body.append(el); + const out = getComputedStyle(el).backgroundColor; + el.remove(); + return out; + }, css); + +// WCAG contrast of two "rgb(…)" colours. +function contrast(a: string, b: string): number { + const lum = (css: string) => { + const [r, g, bl] = rgbOf(css).map((v) => { + const c = v / 255; + return c <= 0.04045 ? c / 12.92 : ((c + 0.055) / 1.055) ** 2.4; + }); + return 0.2126 * r + 0.7152 * g + 0.0722 * bl; + }; + const [hi, lo] = [lum(a), lum(b)].sort((p, q) => q - p); + return (hi + 0.05) / (lo + 0.05); +} + +// THE FOLD (lib/growthGaps.ts). The fixture's last two sites are each under +// 5 % of the chart's total: ONE Other band, drawn last (on top), in the chart's +// neutral grey. The legend shows the kept sites and Other; the table and +// every month's title name every site. +test("two sites under 5 % are one Other band on top: the legend shows the kept sites and Other, the table and the titles name all six", async ({ + page, +}) => { + const summary = buildFixtureSummary(); + const months = summary.monthly ?? []; + const sites = summary.sites; + const layers = growthLayers(months, sites); + expect(layers.map((l) => l.key)).toEqual([ + "fixture-one", + "fixture-two", + "fixture-three", + "fixture-four", + "(other)", + ]); + expect(layers.at(-1)!.sites).toEqual([4, 5]); + const titles = sites.map((s) => s.siteTitle); + await page.goto("/"); + const figure = page.locator("figure").filter({ has: page.locator('[role="img"]') }); + + // The legend: the four kept sites, then Other. + await expect(figure.locator("ul[aria-hidden='true'] li")).toHaveText([...titles.slice(0, 4), "Other"]); + + // The bands, in paint order: Other's is the last, in the neutral grey. + const fills = await figure + .locator("svg > path:not(.growth-gap)") + .evaluateAll((els) => els.map((e) => e.getAttribute("fill"))); + expect(fills).toEqual([...siteChartColors(sites).slice(0, 4), OTHER_COLOR]); + + // The image's label names the folded sites; the caption says what Other is. + await expect(figure.locator('[role="img"]')).toHaveAttribute( + "aria-label", + /Fixture Five and Fixture Six, each under 5% of the total, are drawn together as Other\.$/, + ); + await expect(figure.locator("figcaption")).toContainText( + "Instances under 5% of the total are drawn together as Other.", + ); + + // Every month's title names every site with data that month, folded or not. + const hits = await figure + .locator("svg rect.growth-hit title") + .evaluateAll((els) => els.map((e) => e.textContent ?? "")); + expect(hits).toHaveLength(months.length); + let foldedNamed = 0; + months.forEach((m, i) => { + for (const s of sites) { + const v = m.bySite[s.siteId] ?? 0; + if (v > 0) expect(hits[i], `${m.month}`).toContain(`\n${s.siteTitle} ${v.toLocaleString("en-US")}`); + } + if (hits[i].includes("\nFixture Five ") || hits[i].includes("\nFixture Six ")) foldedNamed++; + }); + expect(foldedNamed).toBeGreaterThan(0); + expect(hits.join("\n")).not.toContain("\nOther"); + + // The table: a column per site, all six, then Total. + await figure.locator("summary", { hasText: "Numbers by year" }).click(); + await expect(figure.locator("table thead th")).toHaveText(["Year", ...titles, "Total"]); +}); + +test("Other is its own grey (--chart-other), at least 3:1 on both grounds, and no kept site's colour", async ({ + page, +}) => { + await page.goto("/"); + for (const base of ["light", "dark"] as const) { + await page.evaluate(([k, v]) => localStorage.setItem(k, v), [BASE_KEY, base]); + await page.reload(); + await expect(page.locator("html")).toHaveAttribute("data-base", base); + const swatches = await page + .locator("figure ul[aria-hidden='true'] li > span") + .evaluateAll((els) => els.map((e) => getComputedStyle(e).backgroundColor)); + const other = swatches.at(-1)!; + expect(other, base).toBe(await resolveFill(page, OTHER_COLOR)); + // Its own token, set on each base (a Dark block without it would fall + // back to the Light value from :root). + expect(other, base).toBe(base === "light" ? "rgb(62, 84, 92)" : "rgb(98, 98, 92)"); + expect(other, base).not.toBe(await resolveFill(page, "var(--chart-axis)")); + const ground = await page.evaluate(() => getComputedStyle(document.body).backgroundColor); + expect(contrast(other, ground), `${base}: Other on the ground`).toBeGreaterThanOrEqual(3); + expect(swatches.slice(0, -1), base).not.toContain(other); + } +}); diff --git a/homepage/e2e/instance-colours.spec.ts b/homepage/e2e/instance-colours.spec.ts @@ -4,6 +4,7 @@ import { test, expect, type Page } from "@playwright/test"; import { resolveAccent } from "../../common/lib/accent"; import { ACCENTS } from "../../common/lib/brand"; import { ACCENT_CHART_SLOT, siteChartColors } from "../../common/lib/siteColor"; +import { OTHER_COLOR } from "../app/lib/growthGaps"; import { FIXTURE_PALE_HEX, FIXTURE_SITES, FIXTURE_SUMMARY_NAME } from "./fixture-summary"; // The Official Instances cards wear their site's accent (release 10): a named @@ -16,7 +17,9 @@ import { FIXTURE_PALE_HEX, FIXTURE_SITES, FIXTURE_SUMMARY_NAME } from "./fixture // The dev server reads the synthetic summary (fixture-summary.ts): site 0 // Brass, site 1 a pale custom hex, 2–4 none, 5 Vermilion — so site 3's own // slot (chart-4) is Brass's and it takes the lowest free one, and six sites -// wear all six validated slots. +// wear all six validated slots. Sites 4 and 5 are each under 5 % of the +// chart's total, so the chart draws them as one Other band (release 14 slice +// CF): the legend has the four kept sites, in their own slots, and Other. const BASE_KEY = "ytdlp-tb:base"; @@ -89,6 +92,8 @@ test("each instance card wears its site's accent on every base, in its chart lay expect(chart[0]).toBe(`var(--chart-${ACCENT_CHART_SLOT.brass! + 1})`); expect(chart[5]).toBe("var(--chart-6)"); expect([...chart].sort()).toEqual([1, 2, 3, 4, 5, 6].map((k) => `var(--chart-${k})`)); + // The chart's bands: the four kept sites, then Other (sites 4 and 5). + const kept = 4; await page.goto("/"); const on = { dark: "onDark", light: "onLight" } as const; @@ -99,13 +104,16 @@ test("each instance card wears its site's accent on every base, in its chart lay expect(stripes).toHaveLength(n); // The summary's monthly series draws the chart, legend and all: the - // legend (and so each layer) wears each site's chart colour, six apart. + // legend (and so each layer) wears each kept site's chart colour — its + // slot among all six, unchanged by the fold — then Other's grey, five + // apart. await expect(page.getByRole("img", { name: /transcripts by the month/i })).toHaveCount(1); - expect(legend, "legend swatches").toHaveLength(n); - for (let i = 0; i < n; i++) { + expect(legend, "legend swatches").toHaveLength(kept + 1); + for (let i = 0; i < kept; i++) { expect(legend[i], `legend ${i}`).toBe(await resolve(page, chart[i])); } - expect(new Set(legend).size, "six layers, six colours").toBe(n); + expect(legend[kept], "Other").toBe(await resolve(page, OTHER_COLOR)); + expect(new Set(legend).size, "five layers, five colours").toBe(kept + 1); // Site 0: the NAMED accent at this base's value — not the published // on-dark hex on every base — and its legend swatch is the amber slot, @@ -124,16 +132,42 @@ test("each instance card wears its site's accent on every base, in its chart lay } // Sites 2–4 have no accent: their chart colour, the same as their legend - // swatch. + // swatch where they keep a band (2 and 3; 4 is in Other). for (let i = 2; i <= 4; i++) { expect(stripes[i], `site ${i}`).toBe(await resolve(page, chart[i])); - expect(stripes[i], `site ${i} vs legend`).toBe(legend[i]); + if (i < kept) expect(stripes[i], `site ${i} vs legend`).toBe(legend[i]); } - // Site 5: Vermilion, the sixth slot's family — the rust layer, not the - // golden angle's hue 328 beside the magenta slot. + // Site 5: Vermilion, the sixth slot's family — its chart colour is the + // rust (chart[5], asserted above), not the golden angle's hue 328 beside + // the magenta slot. It is in Other on the growth chart; the test below + // reads the rust off its line on /stats/. expect(stripes[5]).toBe(await resolve(page, ACCENTS.vermilion[on[base]])); - expect(legend[5]).toBe(await resolve(page, "var(--chart-6)")); - expect(hueGap(stripes[5], legend[5]), "vermilion vs its layer").toBeLessThan(25); + expect(hueGap(stripes[5], await resolve(page, chart[5])), "vermilion vs its chart colour").toBeLessThan(25); + } +}); + +// The growth chart folds site 5 into Other, so the rust is checked where it is +// still drawn: /stats/' site lines (homepageChartData, one line per site in the +// summary's order, each in its siteChartColors colour). +test("on /stats/ each site's line wears its chart colour: the Vermilion site's is the rust", async ({ + page, +}) => { + const sites = fixtureSites(); + const chart = siteChartColors(sites); + await page.goto("/stats/"); + for (const base of ["dark", "light"] as const) { + await useBase(page, base); + const lines = page.locator(".recharts-line-curve"); + await expect(lines).toHaveCount(sites.length); + const strokes = await lines.evaluateAll((els) => els.map((el) => getComputedStyle(el).stroke)); + const want = []; + for (const c of chart) want.push(await resolve(page, c)); + expect(strokes, base).toEqual(want); + expect(strokes[5], `${base}: the Vermilion site's line`).toBe(await resolve(page, "var(--chart-6)")); + expect( + hueGap(strokes[5], await resolve(page, ACCENTS.vermilion[base === "dark" ? "onDark" : "onLight"])), + "vermilion vs its line", + ).toBeLessThan(25); } }); diff --git a/homepage/e2e/marketing.spec.ts b/homepage/e2e/marketing.spec.ts @@ -113,8 +113,10 @@ test("the growth chart renders from the summary, or not at all", async ({ return; } await expect(chart).toBeVisible(); + // The caption; with sites folded into Other (lib/growthGaps.ts) a second + // sentence says what Other is. await expect( - page.getByText(/all official instances\.$/i), + page.getByText(/all official instances\.( |$)/i), ).toBeVisible(); }); diff --git a/plans/FACTS.md b/plans/FACTS.md @@ -6844,8 +6844,12 @@ S4, as shipped"; `release-10.md` "Slice L2 / L1, as shipped". Every anchor below release 11), else `seriesColor(i)` when free (the first SIX are `var(--chart-1..6)`), else the lowest free slot — never two sites in one colour, and any six wear the six validated slots. The growth chart, its legend and `/stats` (`SiteGrid`, `homepageChartData`, the By-site leaderboard) wear - it; the accents themselves fail the dataviz validator as a chart palette (no five with Brass - and Blue pass). `useHubSites` lists the official + it (release 14 slice CF: the growth chart draws two or more sites each under 5 % of its total as + ONE Other band on top, in its own `--chart-other` (tokens.css, Light `#3e545c`, Dark `#62625c`; + not in `REQUIRED_TOKENS`); `homepage/app/lib/growthGaps.ts` `growthLayers` / + `layerColors`, which run `siteChartColors` over EVERY site, so a kept site keeps its slot); the + accents themselves fail the dataviz validator as a chart palette (no five with Brass and Blue + pass). `useHubSites` lists the official instances only once `/hub-summary.json` has SETTLED (found, missing or unreadable; `useHubSummary.ts:45`, `networkMode: "always"`), so they never reorder a moment later. The order is applied in the browser: `hub-sites.json` on disk is unchanged. @@ -7383,7 +7387,8 @@ Slices Q (`4855f70b`) and R (`ffdeb2cd`): [`release-12.md`](release-12.md), the `path.resolve` / fs call on the result becomes an asset reference: to a file, or, when the joined path is a directory, to every file under it (`DirAssetReference`). - **A worktree build will not show it.** A worktree has no `transcripts/`, so the reference is - empty. **Test with the corpus visible.** + empty. **Test with the corpus visible.** Nor does it have umtool's e2e fixture (`.e2e-song`, + `.next-e2e`), which is what slice UT's pattern reached (below). - **What happened:** slice Q wrote `path.join(REPO_ROOT, "transcripts", "channels")` with `REPO_ROOT = findRepoRoot(process.cwd())` in `umtool/lib/paths.mjs`. The walk's fallback, `path.resolve(start, "..")`, evaluates to the project root. @@ -7394,18 +7399,51 @@ Slices Q (`4855f70b`) and R (`ffdeb2cd`): [`release-12.md`](release-12.md), the ModuleReference>::resolve_reference failed … Symlink [project]/transcripts/channels/<slug>/archive is invalid, it points out of the filesystem root`. - A worktree built it in 30 s at 0.8 GB, which is how the gate passed. -- **The opt-out is per expression and documented:** `path.join(/* turbopackIgnore: true */ - process.cwd(), bar)`. That exact text is Turbopack's own advice in its "whole project was traced" - message; the table is in the Next docs, `03-api-reference/08-turbopack.md`, "Magic comments". It - goes before the FIRST argument of each path or fs call on such a value, and it changes nothing - at run time. Per call, not per value: +- **The opt-out is per expression, and it is not documented for path or fs calls** (corrected by + release 15, slice UT). `path.join(/* turbopackIgnore: true */ process.cwd(), bar)` is Turbopack's + own advice, in the text of its "Encountered unexpected file in NFT list" issue (the "whole project + was traced" warning; the 16.2.3 native binary carries it), and Next's own server uses it, on + the join and on the fs call around it: `next/dist/server/next-server.js:620` (16.2.3) reads + `existsSync(/* turbopackIgnore: true */ (0, _path.join)(/* turbopackIgnore: true */ this.dir, + 'static'))`. The Next docs (`08-turbopack.md` and + `02-guides/lazy-loading.md`, "Magic Comments") list the comment only for `import()`, `require()`, + `require.resolve()` and `new Worker()`. It goes before the FIRST argument of each call, and it + changes nothing at run time. Measured in slice UT, it works on a `path.join` and on an fs call. + Per call, not per value: - a nested call needs its own marker (`path.dirname(/* turbopackIgnore: true */ fileURLToPath(import.meta.url))`); - - an outer fs call on an opted-out `path.join` is covered. -- **Not followed by the tracer:** `os.homedir()` and `process.env.*`. A build with `HOME` pointed at - a synthetic home inside the project, full of out-of-root symlinks under `reports/`, - `.local/share/archilyzer/song` and `.cache/`, succeeded. So `~/reports` and the XDG song path are - safe as `path.join(os.homedir(), …)`. + - **an fs call on an opted-out `path.join` is NOT covered**, nested or through a variable: it + traces the join's value. Slice UT, on umtool's clip-audio route. With 4 probe files in its dot + directories: the join opted out and the fs calls on its value kept, 4 traced; those calls + stubbed, 0. With the primary's fixture: one `existsSync(path.join(/* opt-out */ CACHE_DIR, …))`, + 1,704; the same with the `existsSync` opted out as well, 0; the join held in a variable and + only `existsSync(/* opt-out */ cached)` reading it, 0. On a cwd-derived value the join is a + known path, so the outer call traces the one file it names; the guard lets that through (its + comment says so; the release 15 review ruled its four sites safe). Next's own + `next-server.js:620` (above) opts out both calls. +- **A value the tracer cannot know is a dynamic part, not ignored** (corrected by release 15, slice + UT). `process.env.*`, `os.homedir()`, a parameter and an imported binding all make patterns over + the app's own directory (`umtool/`): + - 66 of umtool's 68 routes trace its whole tree outside dot-directories (361 files, + `next.config.ts` among them). Opting out every path op in `song/paths.mjs`, `lib/paths.mjs` + and `lib/paths.ts` (`path.resolve(process.env.X ?? path.join(os.homedir(), …))` and the like) + left 31 of the 68 routes clean; every path op in all 53 modules that have one (319 calls), + 49 (`_global-error` and `_not-found` were clean before). The rest come through fs calls + (`lib/report/snapshots.mjs` is the first the warning names). + - The clip-audio route's `path.join(CACHE_DIR, `${stamp}.${asMp3 ? "mp3" : "wav"}`)` (CACHE_DIR + imported, built on the env or home directory) also took in the dot-directories: 1,704 of its + 2,167 entries were the e2e fixture (`.e2e-song`), the e2e server's build directory + (`.next-e2e`) and `.env.local`. Hoisting the ternary into a `const` changed nothing. Without + the ternary (`${stamp}.wav`), as a ternary of two joins, or with the name handed to a function + from another module (the fix, `cacheFile` in `umtool/lib/paths.mjs`), it did not. The video + route's `.mp4` join did not reach the `.mp4` inside `.e2e-song`. + - **A pattern walk does not enter a symlinked directory**, in the root or out of it: a link at + `umtool/.e2e-song/data/planted` to `<primary>/transcripts/channels` gave 0 entries before and + after the fix, while the old route's walk listed the real files beside it (`data/mkvocals`); + one to the worktree's `common/` (675 files) gave 0 on the old route. That is what the + synthetic-HOME build showed, not that the env and home directory are unfollowed. A KNOWN + directory (slice Q's `<root>/transcripts/channels`) is different: it is walked through its + symlinks. - **The guard is `scripts/next-build-trace.test.mjs`** (in `test:scripts`; it was `umtool-build-trace.test.mjs` until release 14 widened it). It scans umtool's app, components, lib, `report-to-video/*.mjs` and `song/paths.mjs`, and, since release 14 (F8), `homepage/app`, @@ -7414,14 +7452,37 @@ Slices Q (`4855f70b`) and R (`ffdeb2cd`): [`release-12.md`](release-12.md), the `process.cwd()`, `import.meta.url|dirname|filename`, `__dirname`, or a call to a function declared in the file whose body carries one, must open with the opt-out. - It is static and per module, as Turbopack's value analysis is: an imported binding is opaque to - it. + it (to Turbopack it is an unknown, which is a dynamic part: see above). - With the slice Q `paths.mjs` it fails on the defect's line. + - **Since release 15 (slice UT):** + - The scan set also follows every relative import out of those folders, which adds + `common/bin/_publicFile.ts`, `homepage/content/docs.ts` and seven `umtool/song` modules + (umtool 211 → 218 modules, the rest 895 → 897); a test pins them. + - The checked calls include `open`, `writeFile`, `appendFile`, `createWriteStream` (and their + Sync forms) and the `fs.promises.` / `fsPromises.` prefixes. No new finding. + - **It reads umtool's last build back.** Every `.nft.json` under `umtool/.next` but the + build's own `cache/` and `dev/` fails on an entry outside the repo, under `transcripts/`, or + through any name starting with a dot other than the build's own directory and + `node_modules/.pnpm`. + - **It skips, saying so,** when there is no build (`umtool/.next/server` or `BUILD_ID` + missing), and when `BUILD_ID` is older than `umtool/next.config.ts` or any umtool module it + scans (review M1). A merge or checkout gives the changed files new mtimes, and umtool runs + under `next dev`, which does not refresh `.next`, so a stale build skips until umtool is + rebuilt instead of failing on a call already fixed. + - **What it can see** (review L2). Without the excludes, the old clip-audio route failed it with + 1,704 entries (the primary's pre-fix build: 1,705, `test-results/.last-run.json` too). With + the excludes umtool's config now carries, such a pattern shows only through a name they + miss: `test-results/.last-run.json` after an e2e run, `.next-shots`, the corpus, a path + outside the repo. A checkout with no e2e run behind it is blind to it; the fix at the call + is what keeps the route clean. - **The build gate** is run with the corpus visible and under a memory cap (the command is in `plans/tools/implementer-rules.md`). Linking `<primary>/transcripts` into a worktree is for a BUILD only. Remove the link afterwards: never run an app, an index or a fixture builder through it. - **The other apps were safe by accident; since release 14 (F8) they are by rule:** - `common/lib/paths.ts` builds every path on the repo root through one opted-out `under()`, and - `findMonorepoRoot()`'s walk and its `process.cwd()` fallback are opted out. + `findMonorepoRoot()`'s walk and its `process.cwd()` fallback are opted out. Since release 15 + `under(first, ...rest)` puts the marker before a named first argument, not a spread (the + release 14 review's L3); `getPaths()` is unchanged. - `homepage/app/lib/source.ts`' directory join on `public` (the source mirror) and the homepage's and export's other cwd joins carry the opt-out. - The guard covers them. A homepage build with the published source measured the same with and @@ -7605,3 +7666,88 @@ this section is stale, by +16 near the top and +203 at the end; they are not rew no digest, which is removed (`digests.remove`, `:962`/`:965`). `mtimes` is then written with the video's current mtimes, so the loss lasts until any of its tracked mtimes (metadata, transcript, subs, availability, digest) moves. + +## The storage health gate (verified 2026-09-29, branch `r15/drive-stall`) + +The record is [`release-15.md`](release-15.md), "Slice DS, as shipped", with the parent's rulings +(Q1–Q5) and the review's fixes (M1–M3, L1–L10). Anchors are at the branch tip after the merge of +`main` `bab894db`. + +- **A drive can be mounted and not answering.** Every in-process fs call on it waits on one of + libuv's threads (4 by default, 16 in the editor's `start`) until it answers (~30 s for the observed + USB reset loop); only a child process isolates a call. A child `stat` of a location's ROOT does not + detect it reliably: the root's inode is in the kernel's cache whenever the drive was used lately. +- **The state is `common/lib/storageHealth.ts`**, one map on `globalThis.__yttStorageHealth__` (the + pass writes it from instrumentation's module copy; pages read it from theirs), with each entry's + `detector` and the counters' `device`. `recordLocationHealth` (`:182`): one `stalled` answer stalls + at once, and every transition to stalled refuses `onDrive`'s waiting calls; `HEALTH_CLEAN_TO_CLEAR` + (2) clean answers in a row clear it; `absent` is clean; a new root starts over. + `registerLocationHealth` (`:267`) creates entries with no answer. `stalledLocationForPath` + (`:319`) matches like `locationOfDataDir`; `stalledLocation` (`:335`) is by id AND root. +- **Detector 1, every 15 s: the block device's counters** (`detectLocationHealth`, + `lib/storageVolumes.ts:601`). The root's device from the last pass (findmnt `-J -T <root> -o + SOURCE,UUID`, raced against 3 s, only when there is none or its `/sys` entry stops reading; + another volume's UUID names none; `[…]` stripped, `/dev/mapper` resolved, basename); then + `/sys/class/block/<dev>/stat` (`parseBlockStat`, `storageHealth.ts:735`): completed = fields 1 + 5 + + 12 + 16 (reads, writes, discards, flushes), in flight = field 9. Stalled ⇔ in flight at both + samples AND nothing completed between; samples at least `MIN_COUNTER_INTERVAL_MS` (10 s) apart; the + first gives no verdict. in_flight counts only requests dispatched to the driver: one requeued + during a host reset is not counted, so a sample in that window can read clean (the watchdog covers + it). No device → the child `stat -L -c %F` probe (`probeLocationHealth`, `:386`). The samples are + on `globalThis.__yttHealthDetector__` (the pass and /storage's Refresh share them). +- **Detector 2, on every gated call: `onDrive(where, call)`** (`storageHealth.ts:633`). Refused with + no call on a stalled location; otherwise raced against `DRIVE_CALL_BUDGET_MS` (3 s; test seam + `setDriveCallBudget`); the budget covers the whole unit passed in. A timeout marks the location + stalled (since now) and throws `DriveNotAnsweringError`, leaving the call to settle — unless the + location's device counters (read synchronously from `/sys` through the reader `storageVolumes.ts` + registers with `setCounterReader`, `:509`) moved since the call began: then the call is refused + as slow and nothing is marked. At most `DRIVE_CALLS_IN_FLIGHT` (4) calls per slot key in flight + (`acquireSlot`, `storageHealth.ts:501`): the rest queue in JS. A waiting call's deadline follows progress: every call that + returns on the key (in time or late) restarts it (`releaseSlot`, `:599`, re-arms every waiter), and a + waiting call is refused unmarked only when nothing on the key has returned for the budget plus a + quarter of it (at most 250 ms) — never for the queue's depth alone. Waiting calls are refused at + once by any transition to stalled; and when every slot is held by a call already past its budget + (`overdue`, each kept with the counters reading from when it began), a new call is refused at once, + and the location is marked stalled again only if the disk has completed nothing since the oldest of + them began (otherwise "drive slow", unmarked). A slot is freed when its call really returns. The + slot key: a configured location's id; a probe of another root under its id, that root (marks + nothing); a path on no configured location, the root it is under (`rootOfUnknownPath`, `:457`; + marks nothing). Do not nest it for one key. The timer is not unref'd. +- **The cadence** is `runStorageHealthPass` (`controller/storageWatch.ts:433`): prune, register, then + every location concurrently; every 15 s from `startStorageHealthWatch` (`:546`), plus one at arm + time, armed by `editor/instrumentation.ts` ABOVE the idle gate (it writes nothing). The five-minute + pass (`startStorageWatch`, `:521`) stays below it. A CLI process has no pass (its inspects are still + raced). `refreshLocationHealth` (`:485`) is /storage's Refresh. +- **The gate order in `inspectChannelMedia`** (`lib/channelMedia.ts:316`): config → memo (a + remembered `in-transition` is returned as is, anything else is gated first) → the relocation + marker (`:362`, corpus disk) → the gate (`:380`) → the link (corpus disk) → the target's `stat` + through `onDrive` (`:445`). A stall is never memoised. Other gated calls: `probeLocation` + (`storageVolumes.ts:255`, its stat and statfs through `onDrive`) and its memo; `volumeFreeBytes` + (`controller/storageLocations.ts:288`, `:325`, `:334`); `readChannelStat` (`controller/channels.ts:215`, + the walk through `onDrive`, `null` on a stall); the snapshot walk (`controller/channelSnapshot.ts:751`, + its listing, keep-latest keys and per-video unit); the recency tail reads; the move-root check; the + saved-video store; `listSavedVideos` with `notAnswering`; the videos list, the video page, the + Cleanup stage, the Storage stage's statfs and the media file route. `channelMediaStall(config)` + (`channelMedia.ts:231`) is the no-I/O question for a holder of a config. +- **`stalled` is a sixth `ChannelMediaStatus` and a sixth `StorageLocationStatus`** ("Not + answering"). `HELD_REASON`, `MediaLocationBadge`'s two tables and `STORAGE_STATUS_LABEL` are the + `Record`s that make tsc name every table a seventh would need. `isMediaHeld` holds it, so both + pool-wide builds hold a stalled channel; the storage watch counts it as down (two passes pause), + on a location or not, and the pause record carries `cause: "not-answering"` (`ChannelAutoPause`, + `lib/channelPriority.ts`; absent = not there). +- **`inspectChannelMedia` is memoised for 5 s** (`CHANNEL_MEDIA_MEMO_MS`, `:265`), keyed by channels + dir, slug and configured `dataDir`, on `globalThis.__yttChannelMediaMemo__`. `{ fresh: true }` + skips it and does not store; the deciders that pass it are listed in the record (the guard and its + six callers, both movers, both builds, the watch, eviction, the re-point preflight, doctor). The + runners' tick shares the status poll's `buildChannelWork` and so reads the memo. + `forgetChannelMedia` (`:291`) is called by the channel mover's marker writes and clear, + `clearRelocationMarker`, a re-point, /storage's Refresh and the e2e `invalidate-cache` route. +- **`UV_THREADPOOL_SIZE`** defaults to 16 in `editor/package.json`'s `start` and in + `docker/entrypoint.sh`. `ports.test.ts` reads every `${NAME:-N}` in a script as a port and names it + as the one exception (`NUMERIC_NOT_PORTS`, `common/lib/ports.test.ts:29`). +- **Not covered:** a call already in flight when the drive stalls (at most four per drive for the + calls through `onDrive` — every page and poll path and the snapshot walk; a job's own reads that do + not go through it, `measureTree`, the index build's processing phase, the snapshot's sequential + reconcile pass, are not capped); a hand-typed root is capped but never marked; a drive already + stalled at boot before the second counter sample, unless a page reaches it; per-click server + actions; the file route's stream; the corpus disk itself. diff --git a/plans/release-14.md b/plans/release-14.md @@ -22,9 +22,11 @@ slice HP added to it on the operator's ruling of the same day. Rules: | Lows, chart gap, T1, H1 (H3 folded in), H2 | `r14/two-grounds-headers` | The final review's Lows; the charts' surface gap; two grounds and each site in its own accent; the export and hub headers carry the social row and the toggle as one group, with an Archilyzer link to the homepage's `#instances` in place of the sites dropdown and the hub link; Changelog to the footer; after its review, the narrow header keeps the name and shows only the marked links (every header) | `common/components/{ThemeProvider,ThemeScript,ThemeToggle,SocialScroll}.tsx` + `themeConfig.ts` (and the deleted `ThemeMenu`, `ThemeRadios`), `common/components/charts/{ChartView,CrossSiteChart,surfaceGap}`, `common/styles/tokens.css`, `common/lib/{brand,accent,siteColor,paths,project,socialSvg,siteSchema,settingsSchema}.ts` + tests, `scripts/next-build-trace.test.mjs`, `export/app/components/{Header,MobileMenu,Footer}.tsx` (and the deleted `SiblingSwitcher`), `export/app/{layout.tsx,globals.css,changelog/page.tsx,lib/brand.ts}`, `export/e2e{,-hub}/**` (the theme, header and branding specs), `export/playwright.config.ts`, `editor/app/{layout.tsx,globals.css,sites/components/SiteForm.tsx}`, `editor/e2e/theme.spec.ts`, `homepage/app/{page.tsx,layout.tsx,globals.css,lib/*,changelog/page.tsx,components/{Header,ArchiveGrowthChart,ArchiveCards}.tsx}`, `homepage/e2e/**`, `homepage/content/docs/operate.md`, `SETTINGS.md`, `SITE.md` | | HS | `r14/hidden-sites` | Hidden sites: `site.json` `listed` (absent = listed). An unlisted site builds and deploys as before, and is left off the homepage (cards, chart, `/stats`), the hub (members, federated search, `corpus.json`, `llms.txt`), every other site's footer and the published id lists (`channel-sites.json`, the pooled `stats/`); the channels only it exposes count in no public total. One checkbox in the site form | `common/lib/{siteSchema,site,homepageSummary}.ts` + tests, `common/controller/{poolSummary,buildStats}.ts` + tests, `common/bin/{compose-homepage,compose-hub}.ts` + `compose-hub.test.ts`, `editor/app/sites/{actions.ts,components/SiteForm.tsx}`, `editor/e2e/{helpers.ts,sites-crud.spec.ts}`, `homepage/e2e/{fixture-summary.ts,unlisted-site.spec.ts}`, `SITE.md`, `plans/FACTS.md` (Naming hazards) | | S1 | `r14/first-search` | A clear screen until the first Search, on every site's search page and the hub's: no results area until the visitor asks (a Search, a profile load, or a link that carries a query or a filter), with the bar's "Press Enter or click Search to apply" line meanwhile; two page-life flags, the hold of release 8 unchanged and the gate new | `common/components/{SearchSessionContext,SearchResults,SearchBar}.tsx`, `export/e2e/first-search.spec.ts` (new), `export/e2e/{browse-all,workspace-shell,charts,restore-no-refire,responsive,tag-chips}.spec.ts`, `export/e2e/helpers.ts` (`showAll`), `export/e2e-hub/federated-search.spec.ts`, `export/CHANGELOG.md`, `plans/export-header-first-search.md` | +| CF | `r14/chart-fold` | The homepage's growth chart folds two or more sites each under 5 % of its total into one Other band on top, in its own near-neutral grey (`--chart-other`); the kept sites keep their colours; the legend shows them and Other, the hover titles and the table every site | `homepage/app/lib/growthGaps.ts` + test, `homepage/app/components/ArchiveGrowthChart.tsx`, `common/styles/tokens.css` (`--chart-other`), `homepage/e2e/{fixture-summary.ts,growth-chart.spec.ts,instance-colours.spec.ts,marketing.spec.ts}`, `homepage/CHANGELOG.md`, `plans/FACTS.md` | **Order:** HP → `r14/two-grounds-headers` → HS (`r14/hidden-sites`) and S1 (`r14/first-search`), -siblings. The shared files are the three changelogs' `[Unreleased]` sections and this record. +siblings → CF (`r14/chart-fold`), off `main` after HS, with `main` merged in after S1. The shared +files are the three changelogs' `[Unreleased]` sections and this record. ## Record @@ -1410,11 +1412,251 @@ The fix changes only which flag each place sets, so the full suites were not re- load average was 33. On the second attempt the dev server did not start within 120 s. The third run passed 13/13. +### Slice CF, as shipped — the growth chart folds the small sites into Other (2026-09-29) + +Branch `r14/chart-fold` off `main` `69e058d6` (slice HS merged), worktree +`~/Projects/homepage-social-visible` (block #3: homepage e2e 3340, homepage static 3331), one Opus +implementer. Scratch files `cf-*` in the job's `tmp`. The rulings (2026-09-29, not re-opened): +1. A site under **5 %** of the placed total — the chart's total over the whole plotted range, the + sum the bands are placed from — folds into ONE "Other" band, drawn on top of the stack, in a + neutral grey from the chart tokens that keeps ≥ 3:1 on the Light and the Dark ground. +2. A fold of one is no fold: with a single site under the threshold, nothing folds. +3. The legend shows the kept sites plus "Other"; the hover title and the "Numbers by year" table + still name every site. +4. Colour follows the site, never its rank: a kept site's colour does not change because another + folded. +5. The instance cards and `/stats` are unchanged; the surface gap draws correctly with Other on + top. +6. (The review, ruled by the parent.) Other wears a dedicated `--chart-other` token in both chart + blocks of `tokens.css`: Light `#3e545c`, Dark `#62625c` (the near-neutral). + +| sha | what | +|---|---| +| `8351d24f` | `homepage:` the fold in `lib/growthGaps.ts` (`foldedSites`, `growthLayers`, `ownLayers`, `layerColors`; `growthStack` stacks layers); the chart draws the layers, its legend, label and caption; the unit tests; the e2e fixture's fifth site at 4 a day | +| `6e0b652d` | `homepage:` e2e: `growth-chart.spec.ts` (the fold, the grey), `instance-colours.spec.ts`, `marketing.spec.ts` adjusted | +| `e95a8058` | `plans:` this record, the slices table, Order and Rollout; FACTS; the homepage changelog | +| `17832633` | `common, homepage:` `--chart-other` in both chart blocks with its numbers; `OTHER_COLOR` points at it; the chart's header comment; `growth-chart.spec.ts` resolves `OTHER_COLOR` and pins each base's value (review M1) | +| `c42d2bbc` | `homepage:` `instance-colours.spec.ts` reads each site line's stroke on `/stats/`, the Vermilion site's the rust (review L2) | +| `afddd66e` | `plans:` the review's fixes in this record (the grey, the Review table, the screenshots, Found and left, Decisions); the CF row's files; FACTS; the homepage changelog | +| `9267ec08` | merge of `main` `721ed0eb` (slice S1); `plans/release-14.md` resolved by hand (below), the changelogs clean | +| _this_ | `plans:` the post-merge gates; the commit table | + +**What shipped.** +- **The rule** (`homepage/app/lib/growthGaps.ts`, pure, beside `growthStack`): + - `foldedSites(months, sites)`: each site's sum over the plotted months against the sum of all of + them. A site is under when `100 × its sum < FOLD_PERCENT × the total` (`FOLD_PERCENT = 5`), on + integers, so exactly 5 % keeps its band. The folded sites are returned only when two or more + are under; none when the total is 0. + - `growthLayers(months, sites)`: the kept sites in stack order (the summary's, `STACK_ORDER` + unchanged), then `{ key: "(other)", sites: [the folded indexes], other: true }` — a key no site + id can be (`[a-z0-9][a-z0-9-]*`). `ownLayers(sites)` is every site its own band. + - `growthStack(months, sites, layers = ownLayers(sites))`: one band per layer, a layer's value + in a month the sum of its sites'. It returns `layers` in place of `order` (the chart was + `order`'s only reader). Totals, peak, value scale and gridlines are over every site, so the + stack's top and the scale are the same folded or not. + - `layerColors(layers, sites)`: `siteChartColors` over EVERY site, and a kept site takes its own + entry, so a fold never repaints one (with no accents, the kept list alone would move a site + after a folded one to a lower slot; the unit test proves it); Other is `OTHER_COLOR`. +- **The grey is its own token, `--chart-other`** (`common/styles/tokens.css`, beside `--chart-axis` + in both chart blocks, each with its numbers in a comment as `--chart-6` has; review M1): Light + `#3e545c` (OKLCH L 0.43, C 0.03), Dark `#62625c` (L 0.49, C 0.01). Contrast: 7.36:1 on the Light + ground and 7.99:1 on its chart surface; 3.22:1 on the Dark ground and 3.06:1 on its chart surface + (the Dark slots are 3.37–6.46:1 on the ground: Other is the dimmest mark there). Against each + chart slot it can sit on (the dataviz skill's validator's measures, OKLab ΔE × 100, normal / the + worse of protan and deutan): + + | Slot | Light | Dark | + |---|---|---| + | blue (`--chart-1`) | 21.9 / 21.4 | 17.8 / 17.9 | + | green (`--chart-2`) | 17.4 / 15.2 | 19.2 / 15.7 | + | violet (`--chart-3`) | 16.2 / 13.4 | 23.4 / 21.5 | + | amber (`--chart-4`) | 22.1 / 18.2 | 20.3 / 18.1 | + | magenta (`--chart-5`) | 24.5 / 9.1 | 19.5 / 7.3 | + | rust (`--chart-6`) | 14.1 / 10.5 | 13.5 / 9.3 | + + - Every CVD pair clears the floor (6) on both bases, and the target (8) on Light. Dark's magenta + (7.3) is in the 6–8 floor band, legal with the legend, the table and Other's place on top. + Rust is the one pair under the normal-vision 15, on both bases. + - A grey is under the validator's chroma floor by definition (it is the de-emphasis role, not a + categorical slot). + - Other touches only the kept site beneath it (the highest with a height that month). In + today's data that is Bonnellyzer, blue, in every month Other has data. + - The first build used `--chart-axis` (Light `#55646e`, Dark `#8a8170`: magenta at CVD 4.5 / + 4.2, and on Dark green at 4.1, under the floor). The review's sweep found in-band greys that + clear CVD 8 against every slot; the parent ruled the token above. +- **The chart** (`ArchiveGrowthChart.tsx`): + - the legend is the layers: the kept sites' titles, then "Other"; + - the areas and the gaps are keyed by layer; + - the hover title is unchanged: every site with data that month, by name, folded or not; + - the table is unchanged: a column per site; + - the image's `aria-label` gains, when something folds, "Hasanalyzer, Rekietalyzer and + Jasolyzer, each under 5% of the total, are drawn together as Other." (the legend is hidden + from assistive tech); + - the caption gains, when something folds, "Instances under 5% of the total are drawn together as + Other."; + - the header comment carries the grey's numbers. +- **The gaps are unchanged code.** They work on bands, so Other is one band, the top one: + `bandAbove` of the top kept site is Other wherever Other has a height, and Other's own upper edge + has no gap (nothing sits on it). On today's summary the gap segments are the same folded and + unfolded, and none is drawn under Other: Bonnellyzer is under 3 px wherever Other sits on it, so + they touch, as the three small bands did before. + + | Plot height | Segments | Runs | Under Other | + |---|---|---|---| + | 200 px | 73 | 7 | 0 | + | 260 px | 112 | 6 | 0 | + | 300 px | 119 | 7 | 0 | + + On the fixture, gaps are drawn under Other, and no band is covered at any height (unit test). +- **Unchanged:** the instance cards, `/stats`, `siteChartColors`, the hub; in `tokens.css` only + `--chart-other` is added. +- **With today's data** (the primary's `homepage-summary.json` of 2026-09-28, 75,821 transcripts + on the chart): Jeralyzer 42.03 %, Anilyzer 38.40 %, Bonnellyzer 9.18 % keep their bands; + Hasanalyzer 4.33 %, Rekietalyzer 3.79 % and Jasolyzer 2.26 % are Other. + +**The e2e fixture** (`homepage/e2e/fixture-summary.ts`). Its six sites were 60.21, 12.90, 9.68, +8.60, 5.38 and 3.23 % of the chart: one under 5 %, no fold. The smallest change that gives two: +`fixture-five` transcribes 4 a day, not 5 (1,600 recordings, not 2,000). The shares are now 60.87, +13.04, 9.78, 8.70, **4.35** and **3.26 %**, and `fixture-five` and `fixture-six` fold. The +unlisted site stays out of the chart (it is out of the summary). Every changed expectation: +- The fixture's listed totals: 36,799 transcripts, not 37,199; with the unlisted site listed, + 39,199, not 39,599 (HS's record has the old pair). No spec reads either: `unlisted-site.spec.ts` + computes both sides. +- `growth-chart.spec.ts`, the pixel test: one colour per band (four kept sites and Other), not per + site. +- `instance-colours.spec.ts`: the legend has five swatches, not six (the four kept sites in their + slots among all six, then Other's grey); `fixture-five`'s card is no longer compared with a + legend swatch (it has none); `fixture-six`'s hue check reads its chart colour, the rust, not a + legend swatch. +- `marketing.spec.ts`: the caption's check is `/all official instances\.( |$)/i`, not `/…\.$/i`, + since the Other sentence can follow. + +**Tests.** +- `growthGaps.test.ts`: the fixture test now runs the chart's layers (the fold is `[4, 5]`; a gap is + drawn between the top kept site and Other; none along Other's top; no band covered at any + height, folded or not). New: + - today's proportions (42.03 / 38.40 / 9.18 / 4.33 / 3.79 / 2.26 %, the family's accents): the + three largest keep their bands, Other holds the other three, its top is every month's total, + the colours are amber, magenta, blue and the grey; + - the edge: 999 of 20,000 (4.995 %) folds, 1,000 (exactly 5 %) does not; two sites at exactly + 5 % fold nothing; + - a single site under 5 %: no fold; + - no site under: `growthLayers` is `ownLayers`, and the stack is `growthStack`'s default; + - all but one under: one kept band and Other; + - a site with nothing in the range folds with another; nothing plotted folds nothing; + - 25 equal sites: every site folds (see "Decisions"); + - colour stability: the kept sites' colours equal the unfolded run's, and differ from what the + kept list alone would give. +- `growth-chart.spec.ts`, new: + - two sites under 5 % are one Other band on top: the layers are the four kept sites and + `(other)`; the legend reads the four titles and "Other"; the areas' fills, in paint order, are + the four slots and `OTHER_COLOR` (`var(--chart-other)`) last; the label names "Fixture Five and Fixture Six"; + the caption says what Other is; every month's title names every site with data that month and + no title says "Other"; the table's header is Year, the six titles, Total; + - Other is `OTHER_COLOR` on both grounds, rendered `rgb(62, 84, 92)` on Light and `rgb(98, 98, 92)` + on Dark (so each base declares its own), not the axis colour, at least 3:1 on the ground, and no + kept site's colour (review M1). +- `instance-colours.spec.ts`, new (review L2): on `/stats/`, on both bases, the six site lines' + rendered strokes are the six sites' chart colours in the summary's order; the Vermilion site's is + the rust (`--chart-6`) and within 25° of Vermilion's hue. The card test's own site-5 check is the + card's hue against its chart colour; the tautology it replaced (`chart[5]` against + `var(--chart-6)` twice) is gone. + +**They bite** (each change made by hand, the unit tests run, the change reverted): +- `<=` for `<` in the rule: the edge test fails. +- A fold of one allowed: the single-site test fails. +- Colours from `siteChartColors` over the kept sites alone: the colour-stability test fails (the + today's-proportions test does not: its sites' slots come from their accents). +- Other at the bottom of the stack: four tests fail (the fixture's, today's, all-but-one, colour). + +#### Gates (at `6e0b652d`; logs `$T/cf-*.log`) + +- **tsc** clean in every package on the tree of `6e0b652d`, run before the first commit (79 s); + `8351d24f`'s tree has `main`'s versions of the three spec files, which import nothing it changed. +- **Unit:** common **2,229/2,229**; homepage **20/20** (12 before, 8 new). +- **Build**, capped at 5 GB with no swap, from a clean `.next`, with the primary's summary copied + into the worktree's `homepage/public` for the screenshots (the worktree's own put back after): + `pnpm --filter homepage exec next build` **ok**, 39 s, max RSS 779,504 KB. `main`'s chart, + built the same way for the "before" shots: 80 s, 754,160 KB. After the review (`17832633`, for + the retaken shots): 20 s, 809,752 KB. +- **Homepage e2e, full** (97 at `main` after HS; 2 new): + + | Run | Passed | Failed | Time | + |---|---|---|---| + | first (`$T/cf-e2e-homepage.log`, after 9.5 min in the queue) | 98 | 1 | 4.2 min | + | again (`$T/cf-e2e-homepage2.log`, after 5.5 min in the queue) | **99** | **0** | 3.9 min | + + The first run's failure was `marketing.spec.ts`' "Changelog is reachable from the footer on + every page": the 30 s test timeout, reached on the seventh of its seven pages (the dev server + compiling each on first visit, the machine's load average 11–17 with other suites running). It + passed in 5.3 s in the second run and in the three-spec run below; nothing in it reads the + chart. +- **Along the way:** `growth-chart`, `instance-colours` and `marketing` specs, 21 passed, 0 failed + (1.8 min). +- **Numbers tool:** none. +- Not run: the export, hub and editor suites and their builds (no file of theirs changed). + +**Screenshots** (`~/reports/release-14/shots/chart3/`, 2×, the static server on 3331, today's +summary): +- `after-{390,1280}-{light,dark}.png`: the chart with `--chart-other` (retaken after the review; + the operator judges the Dark one); `…-2019-2026.png`: the plot from 2019 on, where Other lies; +- `after-axis-…`: the same from the first build, Other in `--chart-axis`, for comparison; +- `before-…`: the same from `main`'s chart and the same summary (six bands); +- `after-1280-{light,dark}-table.png`: "Numbers by year" open, a column per site. + +**Found and left:** +- **The pairs short of the validator's targets** are Dark's magenta (CVD 7.3, the 6–8 floor band) + and rust on both bases (normal 14.1 / 13.5). They matter only when that site is the top kept one under Other; today it + is blue. +- **`/stats`' channel breakdown has an "Other (N)" of its own**, in `--muted-foreground` + (`common/lib/homepageChart.ts` `OTHER_COLOR`), not `--chart-other`. `/stats` is unchanged, as + ruled. +- **`--chart-other` is not in `REQUIRED_TOKENS`** (`common/components/themeConfig.ts`, the list + `themeTokens.test.ts` checks every base declares): that file is outside this slice. + `growth-chart.spec.ts` pins each base's rendered value instead (a Dark block without it would + paint the Light value from `:root`). +- **The fold is the homepage chart's only.** `/stats`, the hub and the sites' charts draw every + site, as ruled. + +#### Decisions the operator could overturn + +| What I assumed | The alternative | +|---|---| +| The caption gains one sentence saying what Other is, only when something folds | the caption unchanged | +| The image's label names the folded sites | name only how many | +| The legend reads "Other", with no count or names | "Other (3)" | +| The hover title lists every site with data in the summary's order, with no Other subtotal | an "Other N" line, its sites under it | +| When every site is under 5 % (21 or more sites), every site folds: one Other band | keep the largest, or fold nothing | +| A site with nothing in the plotted range is under 5 % and folds with another | leave it out of the chart | + +#### Review (verdict SHIP AFTER FIXES; `$T/cf-review.md`) + +| Finding | Fix | +|---|---| +| M1: the axis grey was under the CVD floor against magenta on both bases (4.5 / 4.2) and against green on Dark (4.1); the record's sweep sentence overstated the case against a better grey | `17832633`: `--chart-other` in both chart blocks (Light `#3e545c`, Dark `#62625c`, as ruled), `OTHER_COLOR` points at it, the chart's comment and the spec follow; the sweep sentence is withdrawn and the token's measured numbers stated ("What shipped"); the labels' shared colour is gone from Found and left; the four `after-*` shots retaken, the first build's kept as `after-axis-*` | +| L1: the CF row omitted `plans/FACTS.md` | `afddd66e`: the row lists it, and `common/styles/tokens.css` | +| L2: `instance-colours.spec.ts` checked the sixth site's colour against itself | `c42d2bbc`: a rendered check of every site line's stroke on `/stats/`, the sixth the rust | +| L3: "Found and left" left rust out of the weak pairs | moot with M1; the line names Dark's magenta and rust | +| L4: the merge of `main` (`721ed0eb`, S1 merged) conflicts in this record only | `9267ec08`: S1's row, then CF's; S1's section, then CF's, before "## Rollout"; the Order line and the Rollout's intro name S1 and CF; both sets of live checks. The changelogs merged clean, every bullet under `[Unreleased]` (checked by eye) | +| L5: the changelog's "at this release's numbers" will drift | left, as a release note | + +The caption and the image-label sentences are kept (the review: acceptable additions). + +**Gates after the review and the merge of `main`** (at `9267ec08`; logs `$T/cf-tsc2.log`, +`$T/cf-tsc3.log`, `$T/cf-gates2.log`, `$T/cf-e2e3.log`): +- **tsc** clean in every package before the fix commits (44 s) and on the merged tree before the + merge commit (164 s). +- **Unit:** common **2,229/2,229**; homepage **20/20**. +- **e2e**, detached and queued: `growth-chart`, `instance-colours` and `marketing`, **22 passed, 0 + failed** (1.1 min; the new `/stats/` line test among them). The full suite was not re-run, as + directed. +- **Build** (for the retaken shots, at `17832633`): 20 s, 809,752 KB, capped. + ## Rollout -Release 14 is slice HP (merged, `bfa1ff3c`), `r14/two-grounds-headers` and slice HS -(`r14/hidden-sites`), each after the parent's merge. Every command below is typed **from the -primary checkout's root**. There is no `archilyzer` on PATH, so it is `pnpm archilyzer …`. The +Release 14 is slice HP (merged, `bfa1ff3c`), `r14/two-grounds-headers`, slice HS +(`r14/hidden-sites`), slice S1 (`r14/first-search`) and slice CF (`r14/chart-fold`), each after the +parent's merge. Every command below is typed **from the primary checkout's root**. There is no `archilyzer` on PATH, so it is `pnpm archilyzer …`. The command forms are the ones verified in `plans/stats-cache-key.md`'s rollout. **Preconditions.** @@ -1464,6 +1706,10 @@ hub goes before the sites, because in basic mode they share `export/out`. The si and `localStorage.getItem("ytdlp-tb:base")` now reads `"light"`. - The homepage's growth chart has no slash in the page colour through any band. `/stats` in Area mode has coloured top lines. +- The growth chart's legend lists the sites with 5 % or more of the chart's total, then + **Other** (a grey band on top); the caption ends "Instances under 5% of the total are drawn + together as Other."; **Numbers by year** has a column for every site. `/stats` still has a line + per site. - Every site's settings in the editor show **List on the Archilyzer homepage and hub**, ticked. With none unticked, the homepage, the hub and every footer list the same sites as before, and `https://archilyzer.pages.dev/homepage-summary.json` reads `"version":6`. diff --git a/plans/release-15.md b/plans/release-15.md @@ -17,7 +17,7 @@ prompt carries its ruling, and this record carries what was built. Rules: | Slice | Branch | What | Owns | |---|---|---|---| | IG | `r15/index-hold` | The index build holds an unreachable channel instead of emptying it | `common/controller/buildIndex.ts` + new `buildIndex.test.ts`, `common/controller/buildStats.ts` (the hold's words move to a shared module), new `common/lib/channelMediaHold.ts`, `common/lib/envVars.ts`, `ENVIRONMENT.md`; records: `plans/{STATE,FACTS,stats-cache-key}.md` | -| DS | `r15/drive-stall` | A stalled drive does not stop the editor answering | per its prompt | +| DS | `r15/drive-stall` | A stalled drive does not stop the editor answering | new `common/lib/storageHealth.ts`; `lib/{storageVolumes,channelMedia,channelMediaHold}.ts`, `controller/storageWatch.ts` and the gated callers; `/storage`, `/channels`, the videos pages; `UV_THREADPOOL_SIZE` (`editor/package.json`, `docker/entrypoint.sh`, `envVars.ts`) | | UT | `r15/umtool-trace` | umtool's build stops tracing the whole `umtool/` folder | per its prompt | | SS | `r15/site-scope` | The editor's site picker paints the stored site at once: the selection is a cookie | `editor/app/lib/activeSite{,Server,Actions}.ts` + `activeSite.test.ts`, `editor/app/components/SiteScope{Provider,Select}.tsx`, `editor/app/layout.tsx`, the scope lines of `editor/app/page.tsx` and `editor/app/channels/page.tsx`, a comment in `editor/next.config.ts`, `editor/e2e/site-scope.spec.ts`; records: `plans/FACTS.md` | @@ -217,6 +217,604 @@ checkout's code, so they hold from the moment `main` has this branch. The editor **Build index** button runs its built bundle, so it holds only after the editor is rebuilt and restarted. +### Slice UT, as shipped — umtool's build stops tracing its dot-directories (2026-09-29) + +Branch `r15/umtool-trace` off `main` `ccf90892`, worktree `~/Projects/r12-source-mirror` (block +#13: editor 4301, test 4311, export 4310), one Opus implementer. Scratch files `ut-*` in the job's +`tmp`. The ruling: find the one expression that widens the clip-audio route's trace and fix it +there; exclude the fixture, the e2e build and env files as a second line; narrow or drop the +`ignoreIssue`; make the trace guard read a build back and close the release 14 review's L1 and L3. + +**What was wrong.** With the primary's e2e fixture in place, umtool's +`app/api/clip/[key]/audio/route.js.nft.json` listed 2,167 files: the 463 its sibling routes list, +178 under `.e2e-song/` (the fixture, where `make-fixture.mjs` links the song data), 1,525 under +`.next-e2e/` (the e2e dev server's build directory, 1.1 GB) and `.env.local`. Turbopack's warning +for it ("Encountered unexpected file in NFT list", the "whole project was traced" text) was silenced +by the config's `ignoreIssue`. The traces are not consumed while `output: "standalone"` stays off, +so nothing broke; a fixture with more in it, or a standalone build, would have carried it. + +**The bisect.** One change per build, in this worktree with four probe files planted in +`.e2e-song/probe/` and `.next-e2e/probe/`; the audio route's trace, total / under dot-directories. +The route as on `main`: **467 / 4**. + +| Change (line on `main`) | Entries | +|---|---| +| `existsSync(file)` :51 stubbed | 467 / 4 | +| **the join `path.join(CACHE_DIR, `${stamp}.${asMp3 ? "mp3" : "wav"}`)` :58 written as a string concatenation** | **463 / 0** | +| `existsSync(cached)` :61, `readFile(cached)` :62, `mkdir(CACHE_DIR)` :78, `readFile(tmpMp3)` :89, `rename(tmpMp3, cached)` :90, `writeFile(tmp)` :93 or `rename(tmp, cached)` :94 stubbed, each alone | 467 / 4 each | +| the `tmpWav` / `tmpMp3` joins :82-83 as concatenations; `writeFile(tmpWav)` :84 stubbed; both `writeFile`s stubbed; either `writeFile` opted out | 467 / 4 each | +| the opt-out on the :58 join | 467 / 4 | +| the :58 ternary hoisted into a `const ext` | 467 / 4 | +| **:58 without the ternary** (`${stamp}.wav`) | **463 / 0** | +| :58 as `path.join(CACHE_DIR, asMp3 ? `${stamp}.mp3` : `${stamp}.wav`)` | 463 / 0 | +| :58 as a ternary of two joins | 463 / 0 | +| opt-outs on `existsSync(cached)` and `readFile(cached)`, or either alone; both stubbed | 467 / 4 each | +| all five readers and writers of `cached` stubbed | 467 / 4 | +| **all five stubbed, and the opt-out on the :58 join** | **463 / 0** | +| all five stubbed, and the join as a concatenation | 463 / 0 | + +With the primary's fixture (the table's 4 are 1,704 there), on top of the last-but-one row: + +| Change | Entries | +|---|---| +| one `existsSync(path.join(/* opt-out */ CACHE_DIR, …))` | 2,167 / 1,704 | +| the same `existsSync` opted out as well | 463 / 0 | +| `existsSync(/* opt-out */ cached)` as the only reader of the opted-out join | 463 / 0 | + +**The expression** is the :58 join, and in it the ternary inside the template literal. The join and +the fs calls on its value each trace the pattern (the join alone with every reader stubbed; the +readers alone with the join opted out; `existsSync` is one such reader), so no single opt-out or +stub cleared it. Without the ternary, the pattern stays out of the dot-directories. + +**The fix** (`a305b956`). `umtool/lib/paths.mjs` gains `cacheFile(name)`, `path.join(/* opt-out */ +CACHE_DIR, name)`, re-exported by `lib/paths.ts`. The audio route names all four of its cache files +through it (`cached`, `tmpWav`, `tmpMp3`, and `tmp`, now `cacheFile(`${stamp}.wav.tmp`)`, the same +path as `${cached}.tmp` in that branch). A value returned by a function from another module is +opaque to the tracer, so the call site traces nothing: the route lists 463, as its siblings do. The +video and face-frame routes join `CACHE_DIR` with a fixed extension; they were measured clean (the +video route's `.mp4` join did not reach the `.mp4` in `.e2e-song`) and name their cache files the +same way, so no route joins `CACHE_DIR` itself. The run-time paths are unchanged. + +**The second line** (`f18fa034`). `outputFileTracingExcludes: { "/*": ["./.e2e-song/**/*", +"./.next-e2e/**/*", "./.env*"] }` (Next 16.2.3's `05-config/01-next-config-js/output.md`: route +globs to globs from the project root; Turbopack reads the key natively, `collect-build-traces.js` +is the webpack path). Measured alone, with `main`'s route: 2,167 → 463. + +**The warning stays silenced, narrow as it was (path + title), with its measured reason.** Dropping +it shows one warning on every build, and after the fix it is still true: +- 66 of the 68 routes trace umtool's whole tree outside dot-directories: its 361 files, + `next.config.ts` (the file the warning names) among them. Only `_global-error` and `_not-found` + do not. +- Path and fs calls on env, home-directory and parameter values do it. Opting out every path op in + `song/paths.mjs`, `lib/paths.mjs` and `lib/paths.ts` left 31 of the 68 routes clean (the audio + route 463 → 102). Opting out all 319 path ops in the 53 modules that have one left 49 clean. The + two clean before are among them. The rest come through fs calls; the next import trace the + warning names is `lib/report/snapshots.mjs`. +- That walk skips dot-directories and does not enter symlinks. The warning names the same file for + it as for the audio route's, so it cannot tell the two apart; the second line and the guard below + cover the dot-directories instead. The config's comment says all of this. + +**A symlinked directory is not entered by these patterns.** The planted link +`umtool/.e2e-song/data/planted` → `<primary>/transcripts/channels` gave 0 entries before and after +the fix, and an in-root link to `common/` (675 files) gave 0 on `main`'s route. What the pattern +reached was the fixture's real files. + +**The guard** (`scripts/next-build-trace.test.mjs`, `9a375ecd`, and after the review `a4d100b4`, +`c3e8a2c7`), 6 → 10 tests: +- **(a) It reads umtool's last build back.** Every `.nft.json` under `umtool/.next` but the + build's own `cache/` and `dev/` fails on an entry outside the repo, under `transcripts/`, or + through any name starting with a dot but the build's own directory and `node_modules/.pnpm`. + Unit tests pin `forbiddenTrace` and which directories are read. + - **It skips, saying so,** with no build, and (review M1) when `umtool/.next/BUILD_ID` is older + than `umtool/next.config.ts` or any umtool module the guard scans. The message gives the + build's time, the first newer file and how to rebuild. A merge or checkout gives the changed + files new mtimes, and umtool runs under `next dev`, which does not refresh `.next`, so a stale + build skips instead of failing on a call already fixed. + - **What it can see** (review L2). Without the excludes, `main`'s route failed it with 1,704 + entries (the first 20 listed); the primary's pre-fix build fails it with 1,705 (the review: + `test-results/.last-run.json` too). With the excludes in place, a pattern like that one shows + only through a name they miss: `test-results/.last-run.json` after an e2e run, `.next-shots`, + the corpus, a path outside the repo. A checkout with no e2e run behind it is blind to it; the + fix at the call is what keeps the route clean. +- **(b) The scan set follows relative imports** out of the listed folders, to any depth. It adds + the review's L1 modules and no others: `common/bin/_publicFile.ts`, `homepage/content/docs.ts` + (895 → 897) and seven `umtool/song` modules, `reasons`, `archive-url`, `pitch`, `flatness`, + `clipwindow`, `deplosive`, `orderfeat` (211 → 218; `song/paths.mjs` was listed by hand before and + is now reached). A test pins them, and that a song CLI and a common CLI stay out. The comments + that called them CLI-only are gone. +- **(c) The checked calls** add `open`, `writeFile`, `appendFile`, `createWriteStream` and their + Sync forms, and the `fs.promises.` / `fsPromises.` prefixes. No new finding. +- **(d)** `common/lib/paths.ts` `under(first, ...rest)` (`8b3409c2`): the opt-out sits before a + named first argument. `getPaths()` hashed identical, old module against new, from the repo root, + `editor/` and a directory outside the repo (56 keys). The guard passes; every caller type-checks. +- The header no longer says the env and home directory are unfollowed or that the opt-out is + documented. The nested-call exemption's comment says it is a simplification (below). + +**FACTS**, "A path joined from `process.cwd()` …", corrected: +- "documented": the Next docs list `turbopackIgnore` only for `import()`, `require()`, + `require.resolve()` and `new Worker()`. The path form is Turbopack's own advice, in the warning's + text in the 16.2.3 binary, and Next's own server uses it on the join and on the fs call around + it (`next/dist/server/next-server.js:620`; review L4). +- "an outer fs call on an opted-out `path.join` is covered": it is not (the second bisect table). +- "Not followed by the tracer: `os.homedir()` and `process.env.*`": such values are dynamic parts, + and make patterns over the app directory. The synthetic-HOME build showed that a pattern walk + does not enter symlinks. +- The guard's entry, the worktree caveat (no fixture either) and `under()`'s shape. + +**Commits** + +| Commit | What | +|---|---| +| `a305b956` | `umtool:` `cacheFile`; the audio, video and face-frame routes name their cache files through it | +| `f18fa034` | `umtool:` `outputFileTracingExcludes`; the `ignoreIssue` kept, its comment the measured reason | +| `8b3409c2` | `common:` `under(first, ...rest)` | +| `9a375ecd` | `scripts:` the post-build check, the relative-import scan set, the opens and writes, the header | +| `750fc850` | `plans:` this section; FACTS; the editor changelog | +| `a4d100b4` | `scripts:` review M1: the post-build check skips a build older than the code it judges | +| `c3e8a2c7` | `scripts:` review L1 (only the build's own `cache/` and `dev/` unread, pinned), L2 (the check's comment says what it can see), L3 | +| `b4d6607d` | `umtool:` review L3 in the `ignoreIssue` comment | +| `31bb0f8c` | `plans:` the review's findings to their commits; FACTS (M1, L2, L3, L4); the changelog (M1); the rollout note | +| `7c6d4969` | merge of `main` (`721ed0eb`, release 14 HS and S1); clean, the changelog bullet still under `[Unreleased]` | +| this commit | `plans:` the gates after the review and the merge | + +#### Gates (logs `$T/ut-*.log`) + +- **tsc** clean over the combined tree before the first commit, 241 s (load average ~26). The + commits are independent pieces of that tree; since then only a comment changed in a + type-checked file (`cacheFile`'s, in `umtool/lib/paths.mjs`). +- **`test:scripts`:** 194 passed, 1 skipped (195): `main`'s 191 + 1 and the guard's three new + tests. Before the final build it failed exactly the post-build test, on a build of `main`'s route. +- **common:** 2,220/2,220, 107 s. +- **umtool's build, capped at 5 GB with no swap, with the primary's `transcripts/` linked in, the + primary's `.e2e-song` and `.next-e2e` hard-linked in, an empty `.env.local`, and the planted + link** (all removed afterwards; none committed): + + | Tree | Wall | User | Max RSS | Audio route | `.e2e-song` | `.next-e2e` | `.env*` | `transcripts` / planted | + |---|---|---|---|---|---|---|---|---| + | `main`'s route and config | 23 s | 57 s | 809 MB | 2,167 | 178 | 1,525 | 1 | 0 / 0 | + | the branch | 34 s | 64 s | 765 MB | 463 | 0 | 0 | 0 | 0 / 0 | + + The wall times swing with the machine's load (another slice's e2e and builds ran alongside); + compile was 7.5 s and 10.7 s, TypeScript 12 s and 19 s. +- **The editor's build**, capped, with the corpus linked in (`paths.ts` changed): 97 s wall, 159 s + user, max RSS 1,576 MB (IG's: 64 s / 1,642 MB, under less load); 0 of its 81 traces' entries + under `transcripts/`, and none that `forbiddenTrace` refuses. +- **e2e** (umtool's own filter, `SONG_DIR=~/reports/quartering-uh-song/data`, queued): + `find.spec.ts` and `triage.spec.ts` fetch the audio route, `faces.spec.ts` the face-frame route, + and `find.spec.ts` names the video route: **6 passed, 29 skipped, 0 failed, 27 s** (6.3 min with + the queue). The skips are the fixture's: this machine has no `wav48/`, `asr/` or `media/`, so + every spec that fetches one of the three routes skipped, and they are not exercised at run time + here. What stands for that: calling `cacheFile` gives the same path as the old expression for + all six names (the four audio names, and the video and frame temporaries). +- **After the e2e run** (which built this worktree its own fixture and `.next-e2e`), a last capped + build with the corpus linked: the audio route 463, none under a dot-directory; `test:scripts` + 194 passed, 1 skipped. The first run of that `test:scripts` failed `queue-lock.test.mjs`'s FIFO + case once (`S1E1S3E3S2E2`) under a load average of about 26; it passed on the rerun, and this + slice does not touch the queue lock. +- **Numbers tool:** none. +- **After the review and the merge of `main`** (at `7c6d4969`): + - tsc clean, 98 s; + - common **2,229/2,229** (`main`'s 2,229), 108 s; + - `test:scripts` **195 passed, 1 skipped (196)**: `main`'s 191 + 1 and the guard's four new + tests. The two runs before it each failed `queue-lock.test.mjs`'s "prints a banner naming the + holder while waiting" under a load average of about 26, the known flake (alone, 11/11 twice); + this slice does not touch the queue lock; + - umtool's build, capped, with the corpus linked and this worktree's own e2e fixture present: + 48 s wall, max RSS 796 MB, the audio route 463, none under a dot-directory; the post-build + check passes on it; + - the same build with its `BUILD_ID` set back to 2026-09-01 (in place, then put back; nothing + committed): the check skips, `umtool/.next was built 2026-09-01T04:00:00.000Z, before + umtool/next.config.ts (219 changed since); rebuild umtool (…) to check its traces`. With the + mtime put back it passes again. + +#### Found and left + +- **The whole-folder trace in 66 routes** (above). Bounded to umtool's own files; cleaning it means + opt-outs on hundreds of path and fs calls on unknown values, which no static check can find. +- **The guard's nested-call exemption.** Without it, four calls would need an outer opt-out: + `common/lib/paths.ts:311` (`existsSync` of one file), `export/app/changelog/page.tsx:9` and + `homepage/app/changelog/page.tsx:24` (one file each; other slices own them), and + `umtool/report-to-video/brand.mjs:82` (the brand kits, which a standalone build needs). Each + traces the file or files it reads. Left, and the comment says so. +- **Which fs calls Turbopack traces is not established per call.** `existsSync` does (the second + table); the writes were added to the guard without a measurement, since an extra opt-out costs + nothing. +- **The post-build check reads umtool only.** In this worktree the editor's 81 traces pass the same + rule. The review's Info saw `editor/.env` in the primary's; a worktree has none, so it was not + re-measured. +- **The gate command in `implementer-rules.md`, in a worktree that has a `transcripts/` directory,** + makes `transcripts/transcripts` and builds without the corpus where the paths point. This worktree + had one (an `index.mdb` from 2026-09-28), and my first two corpus-linked builds ran like that. The + numbers above are from builds that set it aside and put it back. `ln -sT` would refuse instead. +- **A checkout whose umtool build predates its code skips the post-build check** until umtool is + rebuilt (review M1). The primary's `umtool/.next` is from before this slice, so the check skips + there until the rollout rebuilds it. + +#### Decisions the operator could overturn + +| What I did | The alternative | +|---|---| +| A `cacheFile` helper in `lib/paths.mjs`, used by all three routes that join `CACHE_DIR`. **Ruled at review: keep; one way to name a cache file.** | Opt-outs on the audio route's join and on every fs call on its value, in that route only | +| The `ignoreIssue` stays, narrow, with the measured reason in its comment | Drop it: one warning on every build, naming one route's import trace | +| The post-build check refuses any dot-named path but the build's own and `node_modules/.pnpm` | Refuse only `.e2e-song`, `.next-e2e`, `.env*` and `.git` | +| The post-build check covers umtool only. **Ruled at review: umtool only.** In the primary the editor's traces would fail it (75 of 81, `editor/.env` among them) and the export's `.export-index` entries would be misjudged. | Also read the editor's, the export's and the homepage's builds | +| The static check keeps its nested-call exemption, documented as a simplification. **Ruled at review: it stays; the four sites are safe.** | Require the outer opt-out: four new findings, two in files other slices own | +| A stale build skips the post-build check (review M1) | Fail on it, as first shipped | + +#### Review + +**Verdict: SHIP AFTER FIXES** (`ut-review.md` in the job's scratch). No High. The reviewer found +the fix's nine call-site paths byte-identical, the bisect logs in agreement with the tables, the +excludes' key and shape right, and 0 dot entries across all 70 traces of a fresh build. + +| Finding | Where | +|---|---| +| M1: a stale umtool build turns `test:scripts` red, pointing at a call already fixed | `a4d100b4`: the check skips a build older than `next.config.ts` or any module it scans; the changelog, FACTS and "Found and left" say so; the rollout note below | +| L1: `cache`/`dev` skipped at any depth | `c3e8a2c7`: only directly under `.next`; a test with a route directory named each | +| L2: with the excludes in place the check cannot see the original defect | `c3e8a2c7` (the test's comment), guard (a) above and FACTS: it sees only a name the excludes miss | +| L3: "cleaned 31 / 49" | `c3e8a2c7`, `b4d6607d`, this commit: "left 31 / 49 of the 68 clean" | +| L4: FACTS cited the native binary's string as Next's runtime | this commit: `next-server.js:620` | +| L5: the build-gate command's `ln -s` | The parent's (`implementer-rules.md`) | +| Info: the editor's and export's primary builds carry the same class of widening | Recorded in the decisions table; a later slice's | + +**For the rollout.** Rebuild umtool in the primary first, under the cap +(`timeout -s KILL 240 systemd-run --user --scope -q -p MemoryMax=5G -p MemorySwapMax=0 pnpm +--filter umtool exec next build`), then run `pnpm run test:scripts` there. That is the only proof +on the real fixture, which carries `.env.local`, `.next-shots` and `test-results/.last-run.json`, +two of them names the excludes do not cover. Until that rebuild the post-build check skips in the +primary, saying why. umtool's code changes nothing at run time. + +### Slice DS, as shipped — a stalled drive does not stop the editor answering (2026-09-29) + +Branch `r15/drive-stall` off `main` `ccf90892` (slice IG merged), worktree `~/Projects/r12-paths-fix` +(block #12: editor 4201, test 4211, export 4210), one Opus implementer. Scratch files `ds-*` in the +job's `tmp`. The ruling: a drive that is mounted and not answering must not stop the editor +answering. Two detectors find the stall without the editor waiting on the drive, and pages and polls +do not touch it in-process while it is not answering. The parent's five rulings on the first pass +(Q1–Q5) and the review's fixes (M1–M3, L1–L10) are applied; the tables at the end map each to its +commit. + +**What was wrong.** Node runs every filesystem call on libuv's thread pool, four threads by default. +On a drive that has stalled (an SMR disk in a USB enclosure resetting under a long write) each call +blocks its thread for about 30 s. The home page, `/channels` and the three-second auto-queue status +poll each `stat`ted every relocated channel's target in-process and uncached; the one-second job-list +poll walked the whole `data/` of every channel with a job listed, 64 calls wide; and the recency layer +read metadata tails 32 at a time. Four blocked calls were enough for no page, poll or job log to +answer. The storage watch saw nothing of it: its five-minute pass asks "is the disk here", and a +stalled disk is here. + +- **The health state** (new `common/lib/storageHealth.ts`): one map per process on `globalThis` + (`__yttStorageHealth__`, the house pattern: the watch writes it from instrumentation's module copy + and pages read it from theirs), by location id: state `ok | stalled | absent`, `since`, the last + check, a clean streak, the cause, and the `detector` that gave the last verdict. In memory only. + - One `stalled` answer marks the location stalled at once. Two clean answers in a row clear it (an + `absent` answer is clean). A miss in between starts the count again. A re-pointed root starts the + location over. Locations no longer configured are pruned; every configured one is registered + (`registerLocationHealth`) before a pass asks anything, so the watchdog can find a channel's + location before the first verdict. + - Pure of I/O and without execa, so `lib/channelMedia.ts` can ask it. +- **Detector 1, every 15 s: the block device's own counters** (`detectLocationHealth`, + `common/lib/storageVolumes.ts`; ruling Q1(a)). A child `stat` of the root is answered from the + kernel's inode cache whenever the drive was used lately, so it can say "ok" while the reads that + reach the device wait out a reset loop. Instead, per location per pass: + - the root's device: `findmnt -J -T <root> -o SOURCE,UUID` as a child raced against 3 s, a + `[subvolume]` suffix taken off, `/dev/mapper/*` resolved to its `dm-N` (a read of `/dev`), the + basename. Asked only when the root has no device yet or its device's `/sys` entry stops reading + (a replug under another name): a findmnt per pass would leave one child stuck per pass during a + long stall (L10). A UUID other than the location's recorded one names no device (the root is + then a directory on another filesystem, not the drive). + - `/sys/class/block/<dev>/stat`, which never touches the drive: completed = reads (field 1) + + writes (5) + discards (12) + flushes (16) where the kernel counts them (in_flight counts those + too, so a long SMR media-cache flush alone moves completions; L2), and requests in flight (9). + Against the previous sample for the location (same device, at least `MIN_COUNTER_INTERVAL_MS` = + 10 s earlier): **stalled ⇔ in flight at both AND nothing completed between**; anything else is + clean. The first sample gives no verdict. The samples are on `globalThis` + (`__yttHealthDetector__`), so the pass and `/storage`'s Refresh, in different module copies, + compare against one previous sample (L5). + - **No device** (a container, no findmnt, a tmpfs or network source, no `/sys` entry) falls back to + the child `stat` probe (`probeLocationHealth`): `stat -L -c %F -- <root>` raced against 3 s, the + child SIGKILLed and not waited for; `directory` is `ok`, anything else `absent`, no binary `ok`. + - The verdict names its detector (`"counters" | "stat"`), the health state records it (and the + counters' device, which the watchdog reads), and `/storage`'s line says which watched the drive. +- **Detector 2, on every gated call: a 3 s watchdog** (`onDrive(where, call)`, + `lib/storageHealth.ts`; ruling Q1(b)). The detector that cannot be fooled by a cache: a page or poll + that actually reaches the drive finds out. + - Refused at once, with no call, when the location is stalled. + - Otherwise raced against `DRIVE_CALL_BUDGET_MS` (3 s). A call that has not answered marks its + location stalled (since now, cause "a read in the editor did not answer within 3 s") and throws + `DriveNotAnsweringError`; the caller answers `stalled`. The call is left to settle on its own: + its thread is the stated limit. + - **Slow is not stalled** (L6): on a timeout, when the counters detector has named the location's + device, its counters are read (synchronously, from `/sys`, so the check does not wait behind the + pool it is judging) and compared with a reading taken when the call began. Requests completed + meanwhile: the drive is slow; the call is refused and nothing is marked. + - **The budget covers a whole unit of work** (L8): a video directory's reads, a page's reads of + one video, go through as one call, so a slow drive still answering can be marked by one long + unit (unless the counters show it completing, above). + - **At most `DRIVE_CALLS_IN_FLIGHT` (4) calls per slot key are in flight.** The rest wait in a + queue of its own (not libuv's), so a 64-wide walk that meets a stall puts four calls on the + drive, not 64. A slot is released when its call really returns. The queue (M1): + - every transition to `stalled` refuses the waiting calls at once, whoever decided it (the + pass, the watchdog, a Refresh); + - a wait's deadline follows progress (re-review M4): every call that returns on the key, in + time or late, restarts the deadline of every call waiting on it, and a waiting call is + refused, without marking, only when nothing on the key has returned for the budget plus a + grace of a quarter of it (at most 250 ms; the grace lets the calls it waits behind, whose + timers start a moment later, time out and mark first). A deep queue on a drive that is busy + but answering therefore waits as long as it takes; + - calls past their budget are counted per key (`overdue`), each with the counters reading + taken when it began; when every slot is held by one, a new call is refused at once, and the + location is marked stalled again (even if the pass has since cleared it) unless the disk has + completed requests since the oldest of them began: then it is slow, not stalled, and nothing + is marked (the same test as a timeout's). + - **The slot key** (M2, L7): a configured location's id; a probe of another root under a + location's id is keyed by that root and marks nothing; a path on no configured location (a root + typed by hand) is keyed by the root it is under (`<root>/<slug>/data` → `<root>`), so it holds + at most four threads too, and nothing can mark it. Calls are not nested for one key. The timer + is not `unref`'d: it is cleared the moment the call answers. +- **The cadence** (`common/controller/storageWatch.ts`): `startStorageHealthWatch` arms the 15 s + health pass and runs one at once; **`editor/instrumentation.ts` arms it above the idle gate**, + beside the storage boot probe (M2): it is in memory and writes nothing, and without it an idle + boot has no registered locations and a stall the watchdog marks is never cleared. The five-minute + pass (`startStorageWatch`), which may write an auto-pause, stays below the gate. All locations are + asked concurrently, each bounded by its own timers, and an overrunning pass is not stacked. A + transition is logged (`[storage] "<id>": drive not answering — <cause>; …` / `answering again + (ok)`). A CLI process has no pass, but its gated calls still go through the watchdog. +- **The gate and the watchdog, by caller:** + + | Caller | On a stalled location | Through `onDrive` | + |---|---|---| + | `inspectChannelMedia` (`lib/channelMedia.ts`) | **The relocation marker is read first** (ruling Q2: it is in the channel dir, on the corpus disk), so a channel mid-move on a stalled drive reads `in-transition`. Then the gate: status `stalled`, detail `drive not answering (location "<label>", since HH:MM)`, before the link and the target. With a config in hand a stalled channel costs one call, the marker read. | The target's `stat` (`:445`). | + | `assertChannelMediaReachable` | Refuses it (`ChannelMediaUnreachableError`, status `stalled`), so `runManagedFunction`'s `needsMedia` guard, `generateChannelSnapshot` (also the channel page's **Refresh report**), the operation batch, normalise, keep-videos and the shard action refuse it. | Through inspect. | + | The index and stats builds | Hold it: `HELD_REASON.stalled` is "its drive is not answering (a stalled disk)", and IG's `isMediaHeld` holds every status but `ok` and `in-place`. | Through inspect (both of the index build's looks). | + | The storage watch's five-minute pass | Counts a stalled location, or a channel whose inspect says `stalled` on no location (L9), as down: two passes auto-pause its channels, and the pass after the drive answers restores them, as for an unmount (ruling Q3). The pause record carries `cause: "not-answering"`, and `autoPauseReasonOf` says "on a drive that is not answering … when the drive answers again" (L4; a record without a cause, all written before, reads as not there). | Through inspect and `probeLocation`. | + | `probeLocation` and `probeLocationMemo` (`lib/storageVolumes.ts`) | Probe status `stalled` (`STORAGE_STATUS_LABEL`: "Not answering"), identity unknown, no free space; no `stat`, `statfs` or `findmnt`. The memo is asked after the gate. | The root's `stat` and `statfs` (`:262`, `:279`). | + | `volumeFreeBytes` (`controller/storageLocations.ts`) | Unknown ("—"), with no call. | The root's and the mountpoint's `stat`, and the `statfs` (`:288`, `:325`, `:334`). | + | `generateChannelSnapshot` (`controller/channelSnapshot.ts`), after every download or sync, sixteen video directories wide (M3) | Refused by its start guard. | Its `data/` listing (a refusal is rethrown, never read as an empty channel), the keep-latest keys' metadata reads (`keyedVideosNewestFirst`, now `mapConcurrent` 16 wide instead of an unbounded `Promise.all`) and each video directory's unit (`:751`, `:844`). A throw keeps the last `snapshot.json`, as on any failed refresh. The sequential reconcile pass before it is not raced. | + | `readChannelStat` (`controller/channels.ts`) | `null` (no counts, so no progress bar), with no walk. The one-second job-list poll and the home page ask it for every channel with a job listed. | The `data/` readdir, then each video directory as one call (`:107`); `null` when the drive stops answering mid-walk. The batch jobs' `listChannelStatsFromDisk` passes no drive and is unchanged. | + | The recency tail reads (`controller/recencyIndex.ts`) | Skipped, and NOT remembered as misses: layers 3 and 4 until a refresh with the drive answering reads them. | Each tail read of a relocated channel (`:257`). | + | `relocationRootPresenceProblem` (`controller/relocateChannelMedia.ts`) | A move onto a stalled location is refused before the root's `stat`. | The root's `stat` (`:307`). | + | `inspectSavedVideosStore` (`controller/relocateSavedVideos.ts`) | `unreachable` with the stall's detail; `/storage` skips the store's size walk. | The target's `stat` (`:181`); `/storage`'s store walk too. | + | `listSavedVideos` (`controller/savedVideoInventory.ts`), when a page passes `notAnswering` | The channel is skipped and named (its `data/` link is read, not followed). The backup job passes nothing and is unchanged. | Each read of a relocated channel (`:81`). | + | The videos list, the video page (and its title), the channel page's Cleanup stage | A notice (`aria-label="media not answering"`, `MediaNotAnswering.tsx`) with links to the channel and `/storage`. | The list's `data/` listing and titles, then the selected video's files; the video page's whole directory read, as one unit; its title; the Cleanup stage's saved-video totals. | + | The channel page's Storage stage free space (L1) | "—". | The `statfs` of the relocated target. | + | The media file route (`/api/channels/<slug>/videos/<id>/files/<name>`) | 503 with `Retry-After: 15`. | The file's `stat`; the stream after it is not raced. | + +- **The memo** (`inspectChannelMedia`): five seconds per channel, keyed by channels dir, slug and + the configured `dataDir`, on `globalThis` (`__yttChannelMediaMemo__`). On by default (ruling Q4). + A remembered `in-transition` is given as it is; any other remembered answer is gated first; a + stall is not remembered. `{ fresh: true }` bypasses it and does not store. **The deciders, every + one passing `fresh: true`:** + + | Caller | Where | + |---|---| + | `assertChannelMediaReachable` (so every guard below) | `lib/channelMedia.ts:514` | + | ↳ `runManagedFunction`'s `needsMedia` guard | `jobs/streamCommand.ts:283` | + | ↳ the operation batch | `controller/operationBatch.ts:1593` | + | ↳ `generateChannelSnapshot` | `controller/channelSnapshot.ts:739` | + | ↳ `normalizeAllTranscripts` | `controller/normalizeAll.ts:63` | + | ↳ keep-videos | `controller/keepVideosMatching.ts:165` | + | ↳ the shard action | `editor/app/channels/[slug]/shardActions.ts:88` | + | The channel mover, out and back | `controller/relocateChannelMedia.ts:638`, `:877` | + | The index build, before and after the walk | `controller/buildIndex.ts:348`, `:467` | + | The stats build | `controller/buildStats.ts:291` | + | The storage watch's five-minute pass | `controller/storageWatch.ts:235` | + | Clip-window eviction | `controller/evictClipWindows.ts:99` | + | The re-point preflight | `controller/storageLocations.ts:564` | + | `archilyzer doctor` | `bin/doctor.ts:124` | + + **The memo's readers**, pages and polls: the home page (`editor/app/page.tsx:80`), `/channels` + (`editor/app/channels/page.tsx:219`), the channel page (`[slug]/page.tsx:239`), the ops channel + route (`api/ops/channel/[slug]/route.ts:70`), `channelsOnLocation`'s rollup for `/storage` + (`controller/storageLocations.ts:225`), the runners' `buildChannelWork` (`controller/autoRunner.ts:622`, + the tick and the three-second status poll share it), and the bulk actions' skips + (`editor/app/channels/actions.ts:562`, `bulkStorageActions.ts:105`; the jobs they queue re-check + fresh). `forgetChannelMedia(slug?)` clears it: the channel mover's marker writes and clear (each + phase writes its marker after the link and config it changes), `clearRelocationMarker`, a + completed re-point, `/storage`'s Refresh and the e2e `invalidate-cache` route. +- **Headroom:** `UV_THREADPOOL_SIZE=${UV_THREADPOOL_SIZE:-16}` in `editor/package.json`'s `start` + (which the rollout restart script runs) and in `docker/entrypoint.sh` before the editor's `exec`. + Declared in `envVars.ts` (runtime) and `ENVIRONMENT.md` regenerated. It buys time for calls + already in flight and isolates nothing; the doc line says so. `ports.test.ts` reads every + `${NAME:-N}` in a script as a port, so the same commit names `UV_THREADPOOL_SIZE` as the one + numeric default that is not (ruling Q5). +- **Where it shows:** + + | Surface | What it says | + |---|---| + | `/storage` | The row's status badge reads **Not answering**; a line under the details reads `not answering since HH:MM — <cause>. Pages and polls skip this drive until it answers twice in a row. Watched through its disk's request counters.` (or `… with a stat of its root (no disk could be named here).`), `aria-label="location not answering"`. Re-point and Mount are withheld with the status. **Refresh** asks the location's detector first (one answer, counted like the pass's; the counters give none within 10 s of the last sample) and forgets the channels' remembered answers; on a stalled drive its note says so instead of probing. | + | `/channels` | The volume chip reads `… · not answering since HH:MM` in place of its free space, with a title saying what it means. Each row's badge reads `on <label> — not answering` (accessible name `media location: Media not answering · on <label>`). | + | The channel page | The Storage stage's card is red with the inspector's sentence; its destination list names the location "Not answering"; Move back is withheld for a stalled channel, as for an unreachable one. | + | `/saved-videos` | `Not read, because the drive their media is on is not answering: <slugs>.` (`aria-label="saved videos not read"`). | + | `/review`, the rack, the channel page (an auto-paused channel) | `Auto-paused — its media is on a drive that is not answering since <date>. It returns to <tier> on its own when the drive answers again.` (L4) | + | The runners | `[auto] skipping <slug>: media stalled — drive not answering (…)`, once per state change. | + | The logs | The health pass's transition lines, with the cause; the index and stats builds' hold lines. | + + Labels are contracts: no existing accessible name or test id changed. + +**Commits** + +| Commit | What | +|---|---| +| `c0ecc55c` | `common:` `lib/storageHealth.ts`, `probeLocationHealth`, the 15 s health pass, the `stalled` statuses and the gate in inspect, the probe and its memo, `volumeFreeBytes`, `readChannelStat`, the recency reads, the move-root check and the saved-video store; `HELD_REASON.stalled`; the 5 s memo with `fresh` for every decider and `forgetChannelMedia` in the movers; the badge's and the stage card's words. Tests. | +| `cfe551de` | `editor:` `/storage` (the row's line, Refresh asks the health first), the `/channels` volume chip, the Storage panel's Move back, the videos list and video page notice, the media file route's 503, the e2e `invalidate-cache` route; `refreshLocationHealth`; `views/storage.ts` `notAnswering`. | +| `480f2556` | `editor:` `UV_THREADPOOL_SIZE=16` in the editor's `start` and `docker/entrypoint.sh`; `envVars.ts` + `ENVIRONMENT.md`; `ports.test.ts` names it as a numeric default that is not a port (folded in from the next commit, ruling Q5; no change to the tree at the tip). | +| `4e1f7a90` | `editor:` `listSavedVideos({ notAnswering })` for `/saved-videos` and the Cleanup stage's notice. | +| `63f42ef5` | `plans:` the first version of this section, FACTS, the changelog. | +| `c69ad41a` | `common:` rulings Q2 and Q1(b): the marker before the gate; `onDrive` (the 3 s watchdog and the four-call cap per location) on every common gated call; `registerLocationHealth`. Tests. | +| `f6a25cf5` | `editor:` the pages' reads of a relocated drive through `onDrive` (the videos list, the video page and its title, the file route, the Cleanup totals, `/storage`'s store walk). | +| `b1a30902` | `common:` ruling Q1(a): `detectLocationHealth`, the block-device counters with the child-stat fallback; `detector` in the health state and on `/storage`. Tests. | +| `981e05a0` | `plans:` this section rewritten for the rulings; FACTS; the changelog. | +| `33094c29` | `common:` review M1, M2 (slot keys), L2, L5, L6, L7, L8, L10: the queue's deadline, the overdue count, waiters refused on every transition to stalled; the hand-typed-root and candidate-root keys; slow is not stalled; discards and flushes counted; the detector's samples on `globalThis`; findmnt only when needed. Tests. | +| `bd39579e` | `common, editor:` review M2, L4, L9: the health pass armed above the idle gate; the pause record's `cause`; a `stalled` channel on no location is down. SETTINGS.md. Tests. | +| `ff235c2f` | `common:` review M3: the snapshot walk through `onDrive`; keep-latest keys bounded. Test. | +| `19c5842d` | `editor:` review L1, L3: the Storage stage's `statfs` through `onDrive`; the stale comments. | +| `30df4193` | Merge `main` (`bab894db`: release 14 HS, S1, CF; release 15 UT). Two conflicts, both kept: the changelog's `[Unreleased]` carries UT's bullet then DS's; this file is `main`'s with DS's table row and this section after UT's. | +| `b34a7613` | `plans:` the review's findings to their commits; FACTS; the changelog. | +| `c367a3d7` | `editor:` re-review L11: the health pass's block above the boot probe's comment. | +| `d61bdd93` | `common:` re-review M4: a slot wait's deadline follows progress. Tests. | +| `ee68d225` | `plans:` the re-review's findings to their commits; FACTS. | +| `9a12308e` | `common:` the overdue refusal tells slow from stalled by the counters, as a timeout does. Tests. | +| this commit | `plans:` that ruling as a decision row; FACTS; the report. | + +**Tests** (unit; no test stalls a real drive: a stalled call is a promise that never settles or a +fake `stat` that never answers, and a stalled device is a temp `/sys` whose counters stand still) + +| File | What it pins | +|---|---| +| `lib/storageHealth.test.ts` (28) | The rules: one miss stalls at once, one clean answer after a stall does not clear it and two in a row do, a miss in between starts over, `absent` is clean, a re-pointed root starts over, the path match is `locationOfDataDir`'s, pruning, `globalThis`, the "since" wording, registering without an answer. `onDrive`: an answer passes through (value or error) and frees its slot; a never-settling call is `stalled` on the timer, marks the location (since now) and keeps its slot until it settles; a stalled location is refused with no call; seven calls at once put four in flight and the three that waited are refused without a call when the location stalls; a freed slot runs a waiter; the 3 s default (answered after 3–4.5 s). After the review: a hand-typed root's four slots are shared by its channels and a fifth call is refused at once, with no entry made; a candidate root has its own slots and leaves the location's calls alone; **M1's case** (four hung calls, two clean answers, a fifth refused at once and the location marked again); a transition to stalled from the pass refuses the waiting call; **L6** (counters completing: four calls refused as slow, the location not marked, a waiter's wait runs out unmarked when nothing returns; counters standing still: marked). After the re-review (**M4**): a healthy 64-wide walk of units at half the budget (the last call waits about fifteen units) has no refusals and marks nothing, on a location and on a hand-typed root; a queue behind four hung calls is still refused within the budget (marked on a location, unmarked on a hand-typed root); a call that returns late hands its slot to the calls waiting behind it. Then: four units slower than the budget on a device whose completions move — the fifth call is refused at once and nothing is marked; with completions unchanged, after the pass cleared the location — refused and marked again. | +| `lib/storageHealthCounters.test.ts` (7) | The stat line parser (17 and 11 fields, garbage); the verdict over sample pairs (stuck → stalled; moving, idle or drained → ok); device names (partition, `[subvolume]` suffix, non-`/dev` sources); the detector end to end with a fake findmnt and a temp `/sys`: the first sample gives no verdict, stuck → stalled with its cause, drained → ok, busy and moving → ok; a sample sooner than 10 s gives none and keeps the first; every no-device fallback goes to the child stat and says so (tmpfs, findmnt failing, no `/sys` entry, another volume's UUID, no binary). After the review: discards and flushes are counted (a flush alone is not a stall); a known device is read without running findmnt, and findmnt runs again only when its `/sys` entry stops reading; the samples are on `globalThis`. | +| `lib/storageHealthProbe.test.ts` (5) | The child `stat`: a directory is `ok`, a missing path and a file are `absent`; a fake `stat` asleep for 20 s is `stalled` on the timer without being waited for; the 3 s default; an answer inside the budget is taken; no binary is `ok`. | +| `controller/storageStall.test.ts` (22) | A spy on every `node:fs` and `node:fs/promises` call, with a hang mode that makes a matching promise-API call never settle. The gate: with the location stalled, inspect reads only the marker (with a config) or config.json and the marker (without); a channel mid-move on a stalled drive reads `in-transition` (and is remembered so); the guard, `probeLocation` and its memo (no findmnt run), `volumeFreeBytes`, `readChannelStat`, the recency layer, the move-root check, a snapshot refresh, the saved-video store and the inventory make no call on the drive. The watchdog: with the location answering and one drive call hung, inspect, `readChannelStat` (at most four video directories asked), `probeLocation`, `volumeFreeBytes`, the recency layer (not remembered as a miss) the move-root check and (M3) the snapshot walk each answer `stalled` within the race and mark the location (the snapshot writes no `snapshot.json`, and at most four video directories reach the drive); after it, inspect, the guard and the walk make no call on the drive, and once cleared the drive is asked again. The memo: two inspects inside 5 s stat the target once, `fresh` and a 5 s age ask again, a fresh answer is not stored, another target is another key, `forgetChannelMedia` and `clearRelocationMarker` clear it, and a stall is seen with an `ok` remembered. | +| `controller/storageWatch.test.ts` (+9) | The pass stalls a location on one miss and a page then gets `stalled`; the five-minute pass suspects its channel; two clean passes clear it; a location no longer configured is forgotten; a probe that throws is `ok`; the health pass arms on its own, runs once at once and stops, and the five-minute watch arms no health pass; a Refresh counts as one answer; the pass registers every location, records a verdict's detector, and a verdict with no answer changes nothing; a counters verdict records its device and a stat verdict forgets it; a stall auto-pauses after two passes with `cause: "not-answering"` and its wording, the sanitizer keeps the cause, and a record without one reads as not there. | +| `views/storage.test.ts` (+1) | A stalled row reads "Not answering", carries its line, has no free space, and withholds Re-point and Mount. | + +#### Gates (logs `$T/ds-*.log`) + +- **tsc** was clean before every commit and on the merged tree (65 s). After the re-review: 53 s, + once the worktree's gitignored `export/.next/dev` (a generated `validator.ts` an export dev server + had left truncated) was deleted; it is not a source file. +- **Unit, on the merged tree (`30df4193`):** + + | Suite | Result | + |---|---| + | common | **2,295/2,295**, 58 s: `main`'s 2,229 plus DS's 66 (60 at the rulings, 6 from the review). After the re-review: **2,299/2,299**, 53 s (4 for M4); after the overdue ruling **2,301/2,301**, 75 s (2 more). | + | editor unit | 87/87, and 87/87 after the re-review | + | `test:scripts` | 194 passed, 2 skipped (196). The second skip is UT's post-build trace check, which skips a umtool build older than its config (this worktree's `umtool/.next` predates UT); the first is the `LIVE=1` archive check. | + | mcp | 271/271 at the first pass; no mcp file has changed since. | + +- **Docs:** `docs env --check`, `docs files --check` and `settings example --check` all exit **0** + (SETTINGS.md regenerated in `bd39579e` for the pause record's `cause`). +- **Build:** the editor's `next build` on the merged tree, with the primary's `transcripts/` linked + in and capped at 5 GB with no swap: 63 s, max RSS 1,641,444 KB, exit 0. The link was removed after the build, and + nothing ran through it. (First pass: 104 s, max RSS 1,643,860 KB.) +- **e2e** (editor, detached and queued): + + | Run | Specs | Result | + |---|---|---| + | 1, first pass | the `storage` and `channels` specs, `video-page`, `saved-videos`, `dashboard`, `auto-queue`, and IG's eight (`$T/ds-specs.txt`) | **124 passed, 3 failed, 12 skipped, 16.6 min**: three 30 s timeouts while this slice's own tsc and common suite ran beside the suite. | + | 2, first pass | `auto-queue`, `tags`, `saved-videos`, four cleanup specs, `video-titles`, `video-page` (`$T/ds-specs2.txt`) | **70 passed, 1 failed, 5.9 min**: `auto-queue.spec.ts:352`, the Start-button race its own comment describes. | + | 3, first pass | `auto-queue` alone | **24 passed, 1.3 min**. | + | 4, after the rulings (`b1a30902`) | the 9 `storage` + `channels` specs (`$T/ds-specs4.txt`) | **38 passed, 0 failed, 12 skipped, 2.5 min**. | + | 7, after the overdue ruling (`9a12308e`) | the 9 `storage` + `channels` specs (`$T/ds-specs4.txt`) | **38 passed, 0 failed, 12 skipped, 2.5 min**. | + | 6, after the re-review (`d61bdd93`) | the 9 `storage` + `channels` specs (`$T/ds-specs4.txt`) | **38 passed, 0 failed, 12 skipped, 2.6 min**. | + | 5, after the review, on the merged tree (`30df4193`) | the 9 `storage` + `channels` specs; the snapshot scheduler's (`auto-report-refresh`, `channel-work`, `jobs-batch-tasks-drain`, `incomplete-transcript`); the keep-latest users of the bounded key fan-out (`cleanup-holds`, `saved-videos`); `reconcile`; `review` (the auto-pause wording) — `$T/ds-specs5.txt` | **73 passed, 0 failed, 12 skipped, 5.2 min** (37 min 54 s in the queue behind another session's suite). | + + The 12 skips in each are `channels-rack-audit`, which needs `E2E_RACK_SHOTS`. No e2e fixture has + a stalled drive: these confirm nothing changed for drives that answer. The stall paths are the + unit tests above. +- **Numbers tool:** none. + +#### Found and left + +- **The stated limit.** A call already in flight when the drive stalls holds its thread until the + kernel gives up (about 30 s in the observed reset loop). On one drive, at most four threads wait + that way for the calls that go through `onDrive`: every page and poll path, and the snapshot + walk. A job's own reads that do not go through it are not capped: `measureTree` in the movers, + the index build's processing phase (IG recorded it), the snapshot's sequential reconcile pass + (one read at a time), and the keep-latest reads of the other callers of `computeKeptVideoIds` + (cleanup, persist, prune; now 16 at a time). With 16 threads the editor keeps answering while + those four wait. +- **A root typed by hand** (a channel moved to a root no storage location names): its calls are + capped by the root they are under, four at a time, and raced, but nothing can mark it, so every + page and poll keeps asking it, each answering after 3 s or at once while its four slots are held + by calls the watchdog gave up on. Its channels' `stalled` comes from the watchdog alone, and the + five-minute pass counts it as down (L9). Naming the root as a location on `/storage` gives it the + full treatment. +- **The drive's spin-up** (L6, a question for the operator): the budget is 3 s, and a USB disk that + spins down when idle can take 3–10 s to answer its first read. The watchdog then refuses that + read, and marks the location unless the counters show requests completing meanwhile (spinning + up completes none), for 15–30 s, during which the start-of-work guard refuses the channel's jobs. + **Does the drive spin down when idle?** If it does, a longer budget for the first call after an + idle spell, or a spin-down timer on the drive, would avoid it. +- **A drive slower than the budget per unit is treated as not answering.** A queue's depth no + longer matters (M4): a waiting call is refused only when nothing on the drive has returned for + 3 s. But a single unit that takes longer than 3 s is refused: without a mark when the counters + show the disk completing other requests, with one otherwise. A snapshot refresh with such a unit + throws, so the scheduler keeps the last `snapshot.json` and tries again on its next trigger. Once + four such units hold every slot past the budget, the next call is refused at once (a decision + below says when that also marks). +- **The counters need two samples.** A drive already stalled when the editor starts is seen by the + counters at the second pass (15–30 s), or at once by the watchdog when a page reaches it. + in_flight counts only requests dispatched to the driver: a request requeued during a host reset + is not counted, so a sample in that window can read clean; the watchdog covers it. +- **Ungated request paths**, each a click rather than a page or poll: the channel and video server + actions that read a video's directory in the request (`bulkVideoActions`, `digestActions`, + `videoActions`, `fixIncompleteTranscript`, `pipelineActions`); most of what they do is enqueue + jobs, whose guard is fresh and refuses a stalled channel. `/api/media/fetch-window/<jobId>` (one + `stat` of the fetched file, once the job is done). The media file route's stream after its `stat`. +- **The runners' tick reads the memo.** `buildChannelWork` serves both the tick and the status poll, + so a drive unmounted in the last 5 s can have one unit dispatched, which fails at the dangling + link. The snapshot regeneration and `runManagedFunction`'s guard are fresh. +- **umtool's twin of the reachability check** (`checkChannelReachable` in + `umtool/report-to-video/cues.mjs`) has no stall gate: umtool is its own process with no health + state, and `umtool/**` belongs to another slice. +- **A CLI process** (`archilyzer index`) has no health pass, so nothing is marked in it; its + inspects still go through the watchdog, so a target `stat` that takes over 3 s holds the channel. + The corpus disk itself is not watched. + +#### Rulings (parent, 2026-09-29) and decisions the operator could overturn + +| Question | Ruling | Where | +|---|---|---| +| Q1: the child stat of the root can be answered from the cache | Two detectors: the block device's counters every 15 s (child stat only with no device, and the state says which), and a 3 s watchdog on every gated call | `b1a30902`, `c69ad41a`, `f6a25cf5` | +| Q2: the gate before or after the marker | The marker first; the gate covers everything after it | `c69ad41a` | +| Q3: a stall auto-pauses | Yes, after two five-minute passes; it clears when the drive answers again, as for an unmount | as built | +| Q4: the memo on by default | As built; every decider listed above | as built | +| Q5: a commit that failed `ports.test.ts` alone | The test line folded into the thread-pool commit | `480f2556` | + +| What I assumed | The alternative | +|---|---| +| At most four gated calls per location in flight; the rest queue in JavaScript and are refused on a stall. **Kept at review.** | No cap: the watchdog alone, and a stall mid-walk fills the pool until the kernel gives up. | +| A wait for a slot runs out at the budget plus a quarter of it (at most 250 ms), so simultaneous timeouts of the calls it waits behind mark first. | Exactly the budget: a waiter queued in the same tick then gives up a moment before those calls and is refused unmarked, and the mark lands a few ms later. | +| The keep-latest key reads run 16 at a time for every caller (they were an unbounded `Promise.all`). | Bound them only for the snapshot. | +| **Ruled after the re-review:** when every slot is held by a call past its budget, the next call is refused at once, and marks the location stalled only when the disk has completed nothing since the oldest of those calls began; four slow units on a disk still completing requests are "drive slow", refused and unmarked. With no device named (the stat detector), it marks. | Mark whenever every slot is overdue (the M1 fix as first built): four slow units on a busy disk then marked it stalled for 15–30 s. | +| An auto-pause record with no `cause` reads as not there. | Word both cases for it ("not there or not answering"). | +| Two counter samples closer than 10 s give no verdict (a Refresh just after a pass among them). **Kept at review.** | Compare any two samples (a busy healthy drive can read "in flight, nothing completed" over a few milliseconds). | +| A findmnt that does not answer reuses the last device named for that root; one naming another volume's UUID names none. | Treat a findmnt that does not answer as a stall. | +| An `absent` answer counts as clean toward clearing a stall. | Only `ok` clears it. | +| `/storage`'s Refresh is one answer like the pass's. | Refresh clears a stall outright on one clean answer. | +| `UV_THREADPOOL_SIZE` defaults to 16 in `start` and the entrypoint; a value already set wins. | A fixed 16, or only in the rollout restart script. | +| The videos list, the video page and the Cleanup stage show a notice instead of their content. | Render what the snapshot knows and leave out only the drive's files. | +| `readChannelStat` returns `null` on a stall (the job row draws no progress bar). | Counts marked unknown. | + +**What runs which code, for the rollout.** The health pass lives in the editor's process, armed with +the storage watch, so detection takes effect only when the editor is rebuilt and restarted (and not +on an idle boot). The restart must go through `pnpm run start` in `editor/` (the rollout restart +script does) for `UV_THREADPOOL_SIZE` to apply; a process started another way keeps Node's 4 threads +unless the variable is set. CLI builds have no health state; their inspects are raced. + +#### Review + +**Verdict: SHIP AFTER FIXES** (`ds-review.md` in the job's scratch). No High. The parent's rulings +on each finding were applied as below. + +| Finding | Ruling | Where | +|---|---|---| +| M1: a queued `onDrive` call had no deadline | The review's fix: the `overdue` count, refuse at once and re-mark when every slot is overdue, the slot wait raced against the budget and refused without marking, `refuseWaiters` on every transition to stalled; the named test | `33094c29` | +| M2: nothing protected an idle boot or a hand-typed root | The 15 s health pass armed above the idle gate, the five-minute pass below it; a path on no entry capped by the root it is under; the limit stated above | `bd39579e`, `33094c29`, this commit | +| M3: the snapshot walk was uncapped and 16 wide | Its per-video unit (and its listing and keep-latest reads) through `onDrive(config.dataDir, …)`; the three "at most four threads" sentences (this record's Found and left, FACTS, the `onDrive` header) say what is true after it | `ff235c2f`, `33094c29`, this commit | +| L1: the Storage stage's `statfs` | Through `onDrive`, "—" on a refusal | `19c5842d` | +| L2: discards and flushes; requeued requests | Fields 12 and 16 added to completed; one FACTS line on requeued requests | `33094c29`, this commit | +| L3: stale comments calling the child stat the detector | Fixed (the five named, and the health pass's header) | `33094c29`, `bd39579e`, `19c5842d` | +| L4: "a drive that is not there" for a stalled drive | The pause record carries the cause; both cases worded | `bd39579e` | +| L5: the detector's samples per module copy; "Refresh asks again at once" | Samples on `globalThis`; the changelog's wording | `33094c29`, this commit | +| L6: a spin-up longer than the budget | Slow, not stalled, when the counters moved since the call began; the spin-down question above | `33094c29`, this commit | +| L7: a candidate-root probe shared the location's slots | Keyed by its root | `33094c29` | +| L8: the budget covers a whole unit | Said in the doc and above | `33094c29` | +| L9: `stalled` on no location was not down | Counted as down | `bd39579e` | +| L10: a findmnt every pass | Only on a root change or a failed `/sys` read | `33094c29` | +| The two questions | Keep the four-call cap; keep the 10 s spacing | as built | + +**Re-review: SHIP AFTER FIXES.** Every finding above was confirmed closed (L6 for a busy drive; the +spin-up case stays the operator's question). Two new ones: + +| Finding | Ruling | Where | +|---|---|---| +| M4: a slot wait was timed from when the call queued, so a deep queue on a busy but answering drive was refused as "not answering" (units of about 1.1 s at 16 wide, 0.46 s at 32, 0.2 s at 64) | The deadline follows progress: every return on the key re-arms its waiters, and a wait is refused only when nothing on the key has returned for the budget; the overdue and transition refusals unchanged. The limit statement and the keep-latest comment corrected | `d61bdd93`, this commit | +| L11: the health pass's block sat inside the boot probe's comment, which still called itself the only thing an idle boot runs | Moved above it; the phrase dropped | `c367a3d7` | +| (the implementer's note) four slow units past the budget on a disk whose counters show completions made the overdue refusal mark the location stalled | The same treatment as L6: the overdue refusal compares the counters with those taken when the oldest overdue call began; moved → refused, not marked; unchanged → marked | `9a12308e`, this commit | + ### Slice SS, as shipped — the editor's site scope is a cookie (2026-09-29) Branch `r15/site-scope` off `main` `721ed0eb`, worktree diff --git a/scripts/next-build-trace.test.mjs b/scripts/next-build-trace.test.mjs @@ -15,17 +15,28 @@ // per module; an imported binding is opaque to it): every path or fs call whose // arguments carry a value derived IN THAT FILE from `process.cwd()`, // `import.meta.url`, `import.meta.dirname|filename` or `__dirname` must open its -// argument list with the documented opt-out, `/* turbopackIgnore: true */`. +// argument list with Turbopack's opt-out, `/* turbopackIgnore: true */` (the +// form its own "whole project was traced" warning advises; the Next docs list +// the comment only for import(), require(), require.resolve() and new Worker()). // The comment changes nothing at run time. A function declared in the file whose // body carries a source is a source too (`const ROOT = findMonorepoRoot()`). -// `os.homedir()` is NOT a source: the -// tracer does not follow it (a build with HOME pointed at a synthetic home full -// of out-of-root symlinks inside the project succeeds), and neither is -// `process.env.*`. +// +// A value Turbopack cannot know -- `process.env.*`, `os.homedir()`, a +// parameter, an imported binding -- is not one of these sources, and it is not +// ignored either: it is a dynamic part, and a path or fs call on it becomes a +// PATTERN over the app's own directory. Measured in umtool (release 15, slice +// UT): the path ops on env and home-directory values in its path modules took +// in the app's whole tree outside dot-directories (opting them out left 31 of +// the 68 routes clean), and the clip-audio route's join with a dynamic extension took +// in the dot-directories too, the e2e fixture and `.env.local` among them. +// Neither walk entered a symlinked directory. No static check here can tell +// such a pattern from a harmless one, so the last test reads a build's traces +// back instead. // // Run with: pnpm test:scripts import assert from "node:assert/strict"; -import { readdirSync, readFileSync } from "node:fs"; +import { existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; import path from "node:path"; import test from "node:test"; import { fileURLToPath } from "node:url"; @@ -38,10 +49,13 @@ const MARK = "__TURBOPACK_IGNORE__"; const IGNORE_COMMENT = /\/\*\s*turbopackIgnore\s*:\s*true\s*\*\//g; // path ops, and fs calls either bare (`existsSync(`) or on a namespace -// (`fs.readdir(`, `fsp.stat(`). A method on anything else (`obj.stat(`) is not -// one. +// (`fs.readdir(`, `fsp.stat(`, `fs.promises.readFile(`). A method on anything +// else (`obj.stat(`) is not one. The opens and writes are checked too +// (`open`, `writeFile`, `appendFile`, `createWriteStream`): which fs calls +// Turbopack traces is not documented, and an opt-out on one it does not trace +// costs nothing. const SINK = - /(?:\bpath\.(?:join|resolve|dirname|relative)|(?<![\w$.])(?:fs\.|fsp\.|promises\.)?(?:existsSync|readFileSync|readdirSync|statSync|lstatSync|realpathSync|opendirSync|readFile|readdir|stat|lstat|opendir|createReadStream))\s*\(/g; + /(?:\bpath\.(?:join|resolve|dirname|relative)|(?<![\w$.])(?:fs\.promises\.|fsPromises\.|fs\.|fsp\.|promises\.)?(?:existsSync|readFileSync|readdirSync|statSync|lstatSync|realpathSync|opendirSync|openSync|writeFileSync|appendFileSync|readFile|readdir|stat|lstat|opendir|open|writeFile|appendFile|createReadStream|createWriteStream))\s*\(/g; /** Comments out, except the opt-out, which becomes a marker. Strings stay. */ function prepare(text) { @@ -154,9 +168,15 @@ export function untracedCalls(text) { const carries = (args) => SOURCE.test(args) || [...names].some((n) => new RegExp(`(?<![\\w$.])${n.replace(/\$/g, "\\$")}\\b(?!\\s*:)`).test(args)); - // A call nested in these arguments that opts out is opaque to this one too: + // A call nested in these arguments that opts out is let through: // `readFileSync(path.join(/* turbopackIgnore: true */ HERE, "a.json"))`. Its // own arguments are cut out before asking whether this call carries a source. + // That is a simplification, not how Turbopack reads it: the outer call traces + // the join's value all the same (release 15, slice UT, measured). On a + // cwd-derived value that value is a known path, so the outer call traces the + // one file it names (a changelog), or the files of one known directory (the + // brand kits). Where the join names a directory, or a dynamic part could + // reach past the files the call reads, give the outer call its own opt-out. const withoutOptedOut = (args) => { let out = args; for (let i = out.search(new RegExp(`\\(\\s*${MARK}`)); i !== -1; i = out.search(new RegExp(`\\(\\s*${MARK}`))) { @@ -187,24 +207,75 @@ function modulesUnder(dir, out = []) { return out; } -/** umtool's modules that a Next build can reach: the app, its libs, the pipeline. */ +const MODULE_EXT = [".ts", ".tsx", ".mjs", ".js", ".cjs"]; +const isModule = (p) => MODULE_EXT.some((x) => p.endsWith(x)) && !/\.test\./.test(p) && !p.endsWith(".d.ts"); +const isFile = (p) => { + try { + return statSync(p).isFile(); + } catch { + return false; + } +}; + +/** The module a relative specifier names, the way the bundler resolves it, or null. */ +function resolveRelative(from, spec) { + const base = path.resolve(path.dirname(from), spec); + const candidates = [base, ...MODULE_EXT.map((x) => base + x), ...MODULE_EXT.map((x) => path.join(base, "index" + x))]; + return candidates.find((c) => isModule(c) && isFile(c)) ?? null; +} + +// `import … from "./x"`, `export … from "../x"`, `import "./x"`, `import("./x")`, +// `require("./x")`: the relative specifiers only. A package import (`next`, +// `yt-dlp-transcript-common/…`) is covered by the package's own directory in +// the set, or is not this repo's code. +const RELATIVE_IMPORT = /(?:\bfrom\s*|\bimport\s*\(?\s*|\brequire\s*\(\s*)(["'])(\.{1,2}\/[^"'\n]+)\1/g; + +/** + * `files` plus every module of this repo they reach by relative imports, to + * any depth. A directory list alone misses a module one app imports from a + * folder the list treats as CLI-only (`common/bin/_publicFile.ts`, imported by + * `common/publish/source.ts`; umtool's `song/pitch.mjs`, imported by + * `lib/verdict.ts`) or keeps outside `app/` (`homepage/content/docs.ts`). + */ +export function withRelativeImports(files) { + const seen = new Set(files); + const queue = [...files]; + while (queue.length) { + const file = queue.pop(); + for (const m of readFileSync(file, "utf8").matchAll(RELATIVE_IMPORT)) { + const target = resolveRelative(file, m[2]); + if (!target || seen.has(target) || target.includes(`${path.sep}node_modules${path.sep}`)) continue; + if (path.relative(REPO, target).startsWith("..")) continue; + seen.add(target); + queue.push(target); + } + } + return [...seen]; +} + +/** + * umtool's modules that a Next build can reach: the app, its components and + * libs, the report pipeline, and whatever of `song/` they import (`paths.mjs` + * through `lib/paths.mjs`, `pitch.mjs`, `reasons.mjs` and the rest through the + * libs; the other song scripts are CLIs nothing in the app imports). + */ function umtoolModules() { const out = []; for (const d of ["app", "components", "lib"]) modulesUnder(path.join(UMTOOL, d), out); for (const e of readdirSync(path.join(UMTOOL, "report-to-video"))) { if (e.endsWith(".mjs") && !e.includes(".test.")) out.push(path.join(UMTOOL, "report-to-video", e)); } - // song/paths.mjs is imported by lib/paths.mjs; the other song scripts are CLIs. - out.push(path.join(UMTOOL, "song", "paths.mjs")); - return out; + return withRelativeImports(out); } /** * The editor's, the export's and the homepage's modules a Next build can * reach: each app's `app/` (the editor's `lib/` and `instrumentation.ts` too), - * and every module of common/ but its CLIs (`bin/`, which no app imports). The - * common set is wider than what the apps import today, on purpose: a module - * that starts being imported is already covered. + * every module of common/ but its CLIs in `bin/`, and every module those reach + * by a relative import — which brings in the few `bin/` modules an app does + * import (`_publicFile.ts`) and the homepage's `content/docs.ts`. The common set + * is wider than what the apps import today, on purpose: a module that starts + * being imported is already covered. */ function nextAppModules() { const out = []; @@ -214,7 +285,7 @@ function nextAppModules() { if (!e.isDirectory() || e.name === "bin" || e.name === "node_modules" || e.name.startsWith(".")) continue; modulesUnder(path.join(REPO, "common", e.name), out); } - return out; + return withRelativeImports(out); } function untracedIn(files) { @@ -298,6 +369,135 @@ test("no umtool module the app can import joins a cwd-derived path without optin ); }); +test("the scan set follows relative imports out of the listed folders", () => { + const rel = (files) => new Set(files.map((f) => path.relative(REPO, f))); + const um = rel(umtoolModules()); + for (const m of ["paths", "reasons", "archive-url", "pitch", "flatness", "clipwindow", "deplosive", "orderfeat"]) { + assert.ok(um.has(`umtool/song/${m}.mjs`), `umtool/song/${m}.mjs is not scanned`); + } + // A song CLI nothing in the app imports stays out. + assert.ok(!um.has("umtool/song/build-um.mjs")); + const apps = rel(nextAppModules()); + assert.ok(apps.has("common/bin/_publicFile.ts"), "common/bin/_publicFile.ts is not scanned"); + assert.ok(apps.has("homepage/content/docs.ts"), "homepage/content/docs.ts is not scanned"); + assert.ok(!apps.has("common/bin/compose-site.ts"), "a common CLI no app imports is scanned"); +}); + +/** + * Every `.nft.json` a build wrote under its directory `dist`, its cache and + * dev-server output (`dist/cache`, `dist/dev`) aside. Only those two: a route + * directory named `cache` or `dev` deeper down is read like any other. + */ +function traceFilesUnder(dist, dir = dist, out = []) { + for (const e of readdirSync(dir, { withFileTypes: true })) { + const p = path.join(dir, e.name); + if (e.isDirectory()) { + if (dir !== dist || (e.name !== "cache" && e.name !== "dev")) traceFilesUnder(dist, p, out); + } else if (e.name.endsWith(".nft.json")) out.push(p); + } + return out; +} + +/** + * Why `abs`, a file a build traced, must not be in the trace, or null. + * + * The build's own directory (`dist`) holds the chunks every trace lists, and + * `node_modules/.pnpm` is where pnpm keeps the packages; any other path through + * a directory or file whose name starts with a dot is something no server + * needs at run time — a fixture (`.e2e-song`, where the e2e fixture links the + * song data), another build (`.next-e2e`), a secret (`.env.local`), `.git` — + * and is the mark of a pattern Turbopack could not bound. So are the corpus + * and anything outside the repo. + */ +export function forbiddenTrace(abs, dist) { + const rel = path.relative(REPO, abs); + if (rel === "" || rel.startsWith("..") || path.isAbsolute(rel)) return "outside the repo"; + if (abs.startsWith(dist + path.sep)) return null; + const parts = rel.split(path.sep); + if (parts[0] === "transcripts") return "the corpus"; + for (let i = 0; i < parts.length; i += 1) { + if (!parts[i].startsWith(".")) continue; + if (parts[i] === ".pnpm" && parts[i - 1] === "node_modules") continue; + return `under ${parts.slice(0, i + 1).join("/")}`; + } + return null; +} + +test("traceFilesUnder: only the build's own cache/ and dev/ are left out", () => { + const dist = mkdtempSync(path.join(tmpdir(), "next-build-trace-")); + try { + for (const f of ["cache/a.nft.json", "dev/b.nft.json", "server/app/api/cache/route.js.nft.json", "server/app/dev/page.js.nft.json"]) { + mkdirSync(path.dirname(path.join(dist, f)), { recursive: true }); + writeFileSync(path.join(dist, f), '{"files":[]}'); + } + const found = traceFilesUnder(dist).map((f) => path.relative(dist, f)).sort(); + assert.deepEqual(found, ["server/app/api/cache/route.js.nft.json", "server/app/dev/page.js.nft.json"]); + } finally { + rmSync(dist, { recursive: true, force: true }); + } +}); + +test("forbiddenTrace: a fixture, another build, a secret, the corpus, outside the repo", () => { + const dist = path.join(UMTOOL, ".next"); + const at = (p) => forbiddenTrace(path.join(REPO, p), dist); + assert.equal(at("umtool/.next/server/chunks/ssr/a.js"), null); + assert.equal(at("node_modules/.pnpm/next@16.2.3/node_modules/next/dist/server/next.js"), null); + assert.equal(at("umtool/lib/paths.mjs"), null); + assert.equal(at("umtool/.e2e-song/data/planted/x/config.json"), "under umtool/.e2e-song"); + assert.equal(at("umtool/.next-e2e/dev/server/a.js"), "under umtool/.next-e2e"); + assert.equal(at("umtool/.env.local"), "under umtool/.env.local"); + assert.equal(at(".git/config"), "under .git"); + assert.equal(at("transcripts/channels/x/config.json"), "the corpus"); + assert.equal(forbiddenTrace(path.resolve(REPO, "..", "elsewhere", "a.json"), dist), "outside the repo"); +}); + +// The static checks above cannot see a value Turbopack reads through an +// import: `path.join(CACHE_DIR, `${stamp}.${asMp3 ? "mp3" : "wav"}`)` in +// umtool's clip-audio route took umtool's dot-directories into that route's +// trace, the fixture and the e2e build's directory included (plans/release-15.md, +// slice UT). So the last build's traces are read back, when there is one. +// +// What this can see: umtool/next.config.ts now excludes `.e2e-song`, +// `.next-e2e` and `.env*` from every trace, so a pattern like that one shows +// here only through a name the excludes miss -- `test-results/.last-run.json` +// after an e2e run, `.next-shots`, the corpus, a path outside the repo. A +// checkout with no e2e run behind it is blind to it; the fix at the call is +// what keeps the route clean. +test("umtool's last build traced no dot-directory, no corpus file and nothing outside the repo", (t) => { + const dist = path.join(UMTOOL, ".next"); + const id = path.join(dist, "BUILD_ID"); + if (!existsSync(path.join(dist, "server")) || !existsSync(id)) { + t.skip("no umtool build to read (umtool/.next/server); `pnpm --filter umtool exec next build` makes one"); + return; + } + // A build older than the code that decides its traces judges code that is + // gone: after a merge or a checkout it would fail on a call already fixed. + // umtool runs under `next dev` day to day, so nothing else refreshes it. + const built = statSync(id).mtimeMs; + const newer = [path.join(UMTOOL, "next.config.ts"), ...umtoolModules()].filter((f) => statSync(f).mtimeMs > built); + if (newer.length) { + t.skip( + `umtool/.next was built ${new Date(built).toISOString()}, before ${path.relative(REPO, newer[0])}` + + ` (${newer.length} changed since); rebuild umtool (\`pnpm --filter umtool exec next build\`) to check its traces`, + ); + return; + } + const bad = []; + for (const nft of traceFilesUnder(dist)) { + const { files } = JSON.parse(readFileSync(nft, "utf8")); + for (const f of files) { + const why = forbiddenTrace(path.resolve(path.dirname(nft), f), dist); + if (why) bad.push(`${path.relative(dist, nft)}: ${f} (${why})`); + } + } + assert.deepEqual( + bad.slice(0, 20), + [], + `${bad.length} traced file(s) no server needs; find the fs or path call whose value Turbopack could not bound (this file's header):\n` + + bad.slice(0, 20).join("\n"), + ); +}); + test("no module the editor, the export or the homepage can bundle joins a cwd-derived path without opting out", () => { const files = nextAppModules(); assert.ok(files.length > 500, `only ${files.length} modules found`); diff --git a/umtool/app/api/clip/[key]/audio/route.ts b/umtool/app/api/clip/[key]/audio/route.ts @@ -1,10 +1,9 @@ import { createHash } from "node:crypto"; import { existsSync } from "node:fs"; import { mkdir, readFile, writeFile, rename } from "node:fs/promises"; -import path from "node:path"; import { execFile } from "node:child_process"; import { promisify } from "node:util"; -import { CACHE_DIR } from "@/lib/paths"; +import { CACHE_DIR, cacheFile } from "@/lib/paths"; import { sourceWav } from "@/lib/clips"; import { readWavWindow, encodeWav, peakOver } from "@/lib/wav"; @@ -55,7 +54,8 @@ export async function GET(request: Request, ctx: { params: Promise<{ key: string .update(`${video}|${from.toFixed(3)}|${to.toFixed(3)}|${asMp3 ? "mp3" : "wav"}`) .digest("hex") .slice(0, 16); - const cached = path.join(CACHE_DIR, `${stamp}.${asMp3 ? "mp3" : "wav"}`); + // Through cacheFile, never a join here: see its comment in lib/paths.mjs. + const cached = cacheFile(`${stamp}.${asMp3 ? "mp3" : "wav"}`); const type = asMp3 ? "audio/mpeg" : "audio/wav"; if (existsSync(cached)) { @@ -79,8 +79,8 @@ export async function GET(request: Request, ctx: { params: Promise<{ key: string const wav = encodeWav(x, region.sampleRate); let body: Buffer = wav; if (asMp3) { - const tmpWav = path.join(CACHE_DIR, `${stamp}.in.wav`); - const tmpMp3 = path.join(CACHE_DIR, `${stamp}.out.mp3`); + const tmpWav = cacheFile(`${stamp}.in.wav`); + const tmpMp3 = cacheFile(`${stamp}.out.mp3`); await writeFile(tmpWav, wav); await run("ffmpeg", [ "-nostdin", "-v", "error", "-y", "-i", tmpWav, @@ -89,7 +89,7 @@ export async function GET(request: Request, ctx: { params: Promise<{ key: string body = await readFile(tmpMp3); await rename(tmpMp3, cached).catch(() => {}); } else { - const tmp = `${cached}.tmp`; + const tmp = cacheFile(`${stamp}.wav.tmp`); await writeFile(tmp, wav); await rename(tmp, cached).catch(() => {}); } diff --git a/umtool/app/api/clip/[key]/video/route.ts b/umtool/app/api/clip/[key]/video/route.ts @@ -1,10 +1,9 @@ import { createHash } from "node:crypto"; import { existsSync } from "node:fs"; import { mkdir, readFile, rename } from "node:fs/promises"; -import path from "node:path"; import { execFile } from "node:child_process"; import { promisify } from "node:util"; -import { CACHE_DIR } from "@/lib/paths"; +import { CACHE_DIR, cacheFile } from "@/lib/paths"; import { sourceVideo } from "@/lib/clips"; const run = promisify(execFile); @@ -44,7 +43,8 @@ export async function GET(request: Request, ctx: { params: Promise<{ key: string .update(`v1|${video}|${from.toFixed(3)}|${to.toFixed(3)}`) .digest("hex") .slice(0, 16); - const cached = path.join(CACHE_DIR, `${stamp}.mp4`); + // Through cacheFile, never a join here: see its comment in lib/paths.mjs. + const cached = cacheFile(`${stamp}.mp4`); if (existsSync(cached)) { return new Response(new Uint8Array(await readFile(cached)), { headers: { "content-type": "video/mp4", "cache-control": "no-store" }, @@ -52,7 +52,7 @@ export async function GET(request: Request, ctx: { params: Promise<{ key: string } await mkdir(CACHE_DIR, { recursive: true }); - const tmp = `${cached}.tmp.mp4`; + const tmp = cacheFile(`${stamp}.mp4.tmp.mp4`); // -ss BEFORE -i for the fast seek, then -t for the length. Re-encoded rather // than copied because a stream copy starts at the previous keyframe, which // would slide the picture against the audio by up to several seconds. diff --git a/umtool/app/api/face/frame/route.ts b/umtool/app/api/face/frame/route.ts @@ -1,11 +1,10 @@ import { createHash } from "node:crypto"; import { existsSync } from "node:fs"; import { mkdir, readFile, rename } from "node:fs/promises"; -import path from "node:path"; import { execFile } from "node:child_process"; import { promisify } from "node:util"; import { sourceVideo } from "@/lib/clips"; -import { CACHE_DIR } from "@/lib/paths"; +import { CACHE_DIR, cacheFile } from "@/lib/paths"; const run = promisify(execFile); @@ -55,11 +54,12 @@ export async function GET(request: Request) { .update(`face1|${video}|${at.toFixed(3)}|${w ?? "native"}`) .digest("hex") .slice(0, 16); - const cached = path.join(CACHE_DIR, `${stamp}.jpg`); + // Through cacheFile, never a join here: see its comment in lib/paths.mjs. + const cached = cacheFile(`${stamp}.jpg`); if (!existsSync(cached)) { await mkdir(CACHE_DIR, { recursive: true }); - const tmp = `${cached}.tmp.jpg`; + const tmp = cacheFile(`${stamp}.jpg.tmp.jpg`); // -ss BEFORE -i for the fast seek. When a width is asked for it is scaled to // an even one with the aspect preserved; when it is not, the frame comes out // at the source's own size and no mapping is needed at all. diff --git a/umtool/lib/paths.mjs b/umtool/lib/paths.mjs @@ -18,6 +18,21 @@ export { SONG_DATA, SONG_REPORTS }; // the data, not in the repo, and is safe to delete at any time. export const CACHE_DIR = path.join(SONG_DATA, ".cache", "umtool"); +/** + * A file in CACHE_DIR, by name. A route names its cache files through this + * rather than joining CACHE_DIR itself. Turbopack reads a path it can see as a + * pattern of files to trace, in the join and in every fs call its value + * reaches, and CACHE_DIR is unknown to it (an env var or the home directory). + * So `path.join(CACHE_DIR, `${stamp}.${asMp3 ? "mp3" : "wav"}`)` in the + * clip-audio route was a pattern that reached into umtool's dot-directories: + * that route's trace listed the e2e fixture, the e2e server's build directory + * and `.env.local` (plans/release-15.md, slice UT). A value returned by a + * function from another module is opaque to it, so a call site traces nothing. + */ +export function cacheFile(name) { + return path.join(/* turbopackIgnore: true */ CACHE_DIR, name); +} + // Render scratch: the body render and the cut background sit in the job temp // dir ABOVE SONG_DATA, not inside it, because render-poly.mjs writes them next // to its logs. @@ -105,8 +120,10 @@ const dedupe = (list) => [...new Set(list.map((p) => path.resolve(p)))]; * REFERENCE: `next build` walked the whole corpus (hundreds of GB, `data/` * symlinked to another drive) and was OOM-killed, or died on the first symlink * out of the root. A worktree with no transcripts/ builds fine, which is how it - * shipped. The comment is the documented per-expression opt-out; the values at - * run time are unchanged. scripts/next-build-trace.test.mjs holds the line. + * shipped. The comment is Turbopack's per-expression opt-out (the form its own + * "whole project was traced" warning advises; the Next docs list the comment + * for import(), require(), require.resolve() and new Worker() only); the values + * at run time are unchanged. scripts/next-build-trace.test.mjs holds the line. */ export function findRepoRoot(start) { let dir = path.resolve(/* turbopackIgnore: true */ start); diff --git a/umtool/lib/paths.ts b/umtool/lib/paths.ts @@ -19,6 +19,7 @@ import path from "node:path"; // --------------------------------------------------------------------------- export { CACHE_DIR, + cacheFile, INDEX_DIR, MEDIA_ROOTS, MIX_CACHE, diff --git a/umtool/next.config.ts b/umtool/next.config.ts @@ -18,19 +18,37 @@ const nextConfig: NextConfig = { // fail the whole module graph. Every page importing lib/projects then 500s // with "Can't resolve 'cbor-x'", which names a package nothing here uses. serverExternalPackages: ["lmdb"], + // No route's trace may list the e2e fixture (.e2e-song, where + // e2e/fixtures/make-fixture.mjs links the song data), the e2e server's own + // build directory (.next-e2e) or an env file: none is a run-time input. The + // clip-audio route's trace listed 1,704 such files (plans/release-15.md, slice + // UT). That was fixed at the call (lib/paths.mjs `cacheFile`); this is the + // second line, measured on its own: with the old route it takes the trace + // back to what the sibling routes list. scripts/next-build-trace.test.mjs + // reads the last build's traces back. + outputFileTracingExcludes: { + "/*": ["./.e2e-song/**/*", "./.next-e2e/**/*", "./.env*"], + }, turbopack: { // Same reasoning as editor/next.config.ts: Turbopack infers the workspace // root by walking up for the outermost lockfile, and a stray pnpm-lock.yaml // above the checkout silently relocates it. Nothing here lives above the // monorepo root, so pinning it costs nothing. root: path.join(__dirname, ".."), - // Suppress the harmless "whole project was traced unintentionally" NFT - // warning, exactly as editor/next.config.ts does. It fires because the - // server genuinely does runtime-dynamic fs reads it cannot statically bound - // -- lib/paths.ts resolves SONG_DATA from an env var and the routes read - // wav48/<video>.wav by name. We don't use `output: 'standalone'`, so the - // .nft.json traces are never consumed and the over-tracing is cosmetic. - // Scoped to this exact issue (path + title) so other warnings still surface. + // Silences the "whole project was traced unintentionally" warning, and only + // it (path + title), for one measured reason: 66 of the 68 routes trace + // umtool's own tree, its 361 files outside dot-directories, next.config.ts + // (the file the warning names) among them. A path or fs call on a value + // Turbopack cannot know (an env var, the home directory, a parameter) is a + // pattern over the project, and umtool has hundreds. Opting out every path + // op in the three path modules left 31 of the 68 routes clean; opting out + // all 319 in the 53 modules that have one left 49 clean, and the rest come + // through fs calls (lib/report/snapshots.mjs, among others). That walk skips + // dot-directories and does not enter symlinks, and the traces are not + // consumed while `output: "standalone"` stays off. The warning cannot tell + // that walk from one that does reach a dot-directory (both name + // next.config.ts), so that case is excluded above and checked after the + // build by scripts/next-build-trace.test.mjs instead. ignoreIssue: [ { path: "**/next.config.ts",