Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d1c283701e325f94d51e467900f610f5cda0108b
parent 48343182098bc1eb3570f35680e698e5dc9060dc
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue, 29 Sep 2026 20:51:35 -0400

common: the index build holds a channel whose media cannot be read instead of emptying it — not rescanned, its records and shared pages kept, its availability states carried; a full rebuild with one held refuses unless ARCHILYZER_INDEX_ALLOW_HELD=1; heldChannels in the result and the log; new buildIndex.test.ts

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
MENVIRONMENT.md | 1+
Acommon/controller/buildIndex.test.ts | 585+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/buildIndex.ts | 229+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Mcommon/lib/envVars.ts | 1+
4 files changed, 800 insertions(+), 16 deletions(-)

diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md @@ -76,6 +76,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not | `AUDIO_CHECK_RESUME_DURING_PROBE` | the channel's `audioCheck.resumeDuringProbe` | `1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting. | common/ytdlp/audioCheckedDownload.ts | | `AUDIO_CHECK_BACKOFF_FACTOR` | the built-in factor | The audio check's interval backoff factor, in (0, 1], for a one-off run. | common/ytdlp/audioCheckedDownload.ts | | `ARCHILYZER_STATS_ALLOW_DOWNGRADE` | off | `1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions. | common/controller/buildStats.ts | +| `ARCHILYZER_INDEX_ALLOW_HELD` | off | `1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel. | common/controller/buildIndex.ts | | `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts | | `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool | | `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs | diff --git a/common/controller/buildIndex.test.ts b/common/controller/buildIndex.test.ts @@ -0,0 +1,585 @@ +// Integration: the index build's HOLD, through the REAL buildIndex, over a temp +// corpus. +// +// A channel's `data/` may be an absolute symlink to another drive (AGENTS.md, +// "A channel's `data/` may live on another drive"). With that drive unmounted +// the link dangles, and the index build used to read the channel as having no +// videos: it removed every record the channel had, and the site built next +// published the channel as gone. These cases pin the hold that replaced it: +// the channel is not rescanned, and its records and shared pages are kept as +// they are; a FULL rebuild with a channel held refuses unless +// ARCHILYZER_INDEX_ALLOW_HELD is set; and all of it undoes itself when the +// drive is back. The stats build's twin is buildStats.test.ts case (i). +// +// Run with: node_modules/.bin/tsx --test common/controller/buildIndex.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { createRequire, syncBuiltinESMExports } from "node:module"; +import { + chmodSync, + existsSync, + mkdirSync, + mkdtempSync, + readdirSync, + readFileSync, + renameSync, + rmSync, + statSync, + symlinkSync, + writeFileSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +// EVERY PATH getPaths() CAN RESOLVE TO A PLACE THIS FILE'S CODE MAY WRITE IS +// PINNED UNDER ROOT before anything calls it (the buildStats.test.ts list). The +// last case proves no write this file caused landed outside ROOT. +const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-")); +const PINNED: Record<string, string> = { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), + EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), + CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), + SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), + CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), + ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), +}; +Object.assign(process.env, PINNED); +delete process.env.ARCHILYZER_INDEX_ALLOW_HELD; +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const { getPaths } = await import("../lib/paths"); +const { buildIndex, INDEX_ALLOW_HELD_ENV } = await import("./buildIndex"); +const { writeGlobalTags } = await import("../lib/curatedTagsStore"); +const { META_PAGES_PENDING } = await import("./curatedTagsIndex"); +const { open } = await import("lmdb"); + +const paths = getPaths(); +const CHANNEL = "test-channel"; +const DRIVE_CHANNEL = "drive-channel"; +const SITE = "testsite"; +const MEDIA = path.join(ROOT, "media"); +const AWAY = `${MEDIA}-away`; +const COMMON = fileURLToPath(new URL("..", import.meta.url)); + +// ── a write spy over the whole file ───────────────────────────────────────── +// The buildStats.test.ts spy, writes only: node:fs and node:fs/promises, async, +// sync and callback, synced into the named ESM imports the code under test +// holds. The last case reads it. +const writes: string[] = []; +{ + const req = createRequire(import.meta.url); + const fsCjs = req("node:fs") as Record<string, unknown>; + const fspCjs = req("node:fs/promises") as Record<string, unknown>; + const WRITES = ["writeFile", "appendFile", "rename", "mkdir", "rm", "rmdir", "unlink", "copyFile", "cp", "symlink", "link", "utimes", "truncate", "mkdtemp", "chmod"]; + const TWO_PATHS = new Set(["rename", "copyFile", "cp", "symlink", "link"]); + const opensForWrite = (flags: unknown) => + (typeof flags === "string" && /[wa+]/.test(flags)) || + (typeof flags === "number" && (flags & 3) !== 0); + const asPath = (v: unknown) => + typeof v === "string" ? v : v instanceof URL ? fileURLToPath(v) : Buffer.isBuffer(v) ? v.toString() : null; + const wrap = (mod: Record<string, unknown>, name: string, mode: "write" | "open") => { + const fn = mod[name]; + if (typeof fn !== "function") return; + mod[name] = function (this: unknown, ...args: unknown[]) { + if (mode === "write" || opensForWrite(args[1])) { + const ps = TWO_PATHS.has(name.replace(/Sync$/, "")) ? [args[0], args[1]] : [args[0]]; + for (const a of ps) { + const p = asPath(a); + if (p !== null) writes.push(path.resolve(p)); + } + } + return (fn as (...a: unknown[]) => unknown).apply(this, args); + }; + }; + for (const n of WRITES) { + wrap(fspCjs, n, "write"); + wrap(fsCjs, n, "write"); + wrap(fsCjs, `${n}Sync`, "write"); + } + wrap(fspCjs, "open", "open"); + wrap(fsCjs, "open", "open"); + wrap(fsCjs, "openSync", "open"); + wrap(fsCjs, "createWriteStream", "write"); + syncBuiltinESMExports(); +} + +// ── the corpus ────────────────────────────────────────────────────────────── +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; +const videoDir = (id: string, channel = CHANNEL) => + path.join(paths.channelsDir, channel, "data", id); + +// YouTube's rolling-caption shape: parseVtt keeps only lines carrying inline +// timing tags, so a plain cue would parse to nothing. +const VTT = + "WEBVTT\nKind: captions\nLanguage: en\n\n" + + "00:00:00.000 --> 00:00:05.000 align:start position:0%\n" + + "First<00:00:01.000><c> caption</c><00:00:02.000><c> line.</c>\n\n" + + "00:01:00.000 --> 00:01:50.000 align:start position:0%\n" + + "Second<00:01:10.000><c> caption</c><00:01:20.000><c> line.</c>\n"; + +// Metadata and English captions; `subs` adds a German track, which the index +// publishes in the shared subs tree. +function seedVideo( + id: string, + channel = CHANNEL, + opts: { title?: string; subs?: boolean; dir?: string } = {}, +): void { + const dir = opts.dir ?? videoDir(id, channel); + writeJson(path.join(dir, "metadata.info.json"), { + id, + title: opts.title ?? `Video ${id}`, + channel: channel, + upload_date: "20260601", + duration: 120, + description: "fixture", + webpage_url: `https://www.youtube.com/watch?v=${id}`, + extractor_key: "Youtube", + }); + writeFileSync(path.join(dir, "transcript.en.vtt"), VTT); + if (opts.subs) writeFileSync(path.join(dir, "transcript.de.vtt"), VTT); +} + +function writeChannel(slug: string, extra: Record<string, unknown> = {}): void { + writeJson(path.join(paths.channelsDir, slug, "config.json"), { + handling: "youtube", + name: slug, + url: `https://www.youtube.com/@${slug}/videos`, + ...extra, + }); +} + +// A fresh corpus, LMDB and export tree per test, so every count is exact. The +// drive is a storage location, as /storage records it: a held channel is named +// by its label, never by a path. +function resetCorpus(channels: string[] = [CHANNEL, DRIVE_CHANNEL]): void { + for (const p of [paths.transcriptsDir, PINNED.EXPORT_INDEX_DIR, MEDIA, AWAY]) { + rmSync(p, { recursive: true, force: true }); + } + mkdirSync(paths.transcriptsDir, { recursive: true }); + writeFileSync( + paths.settingsFile, + JSON.stringify({ + storage: { locations: [{ id: "usb", label: "USB drive", root: MEDIA, autoRepoint: false }] }, + }), + ); + writeChannel(CHANNEL); + writeJson(path.join(paths.sitesDir, SITE, "site.json"), { + siteId: SITE, + siteTitle: "Test Site", + siteDescription: "fixture", + headerTitle: "Test Site", + homeTagline: "", + socialLinks: [], + groups: [{ id: "default", name: "All channels", selectedByDefault: true }], + defaultGroupId: "default", + channels: channels.map((slug) => ({ slug, groupId: "default" })), + }); +} + +// A channel whose data/ is a relocated symlink, the way the editor's Storage +// panel leaves it: channels/<slug>/data -> <MEDIA>/<slug>/data, with +// config.dataDir recording the target. d1 carries a subtitle track. +function seedDriveChannel(titles: Record<string, string> = {}): void { + const target = path.join(MEDIA, DRIVE_CHANNEL, "data"); + mkdirSync(target, { recursive: true }); + writeChannel(DRIVE_CHANNEL, { dataDir: target }); + symlinkSync(target, path.join(paths.channelsDir, DRIVE_CHANNEL, "data")); + seedVideo("d1", DRIVE_CHANNEL, { subs: true, title: titles.d1 }); + seedVideo("d2", DRIVE_CHANNEL, { title: titles.d2 }); +} + +// Unmount: the link now dangles, exactly as an absent USB drive leaves it. +const unmount = () => renameSync(MEDIA, AWAY); +const remount = () => renameSync(AWAY, MEDIA); + +async function runIndex(log: string[] = []) { + const res = await buildIndex({ paths, onLog: (s) => log.push(s) }); + return { res, log }; +} + +// ── reading what the build left ───────────────────────────────────────────── +// The index LMDB, opened the way buildIndex opens it, and closed again. +function withIndex<T>(fn: (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>) => T): T { + const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); + try { + return fn((name) => root.openDB({ name, encoding: "msgpack" })); + } finally { + root.close(); + } +} +// Every indexed video, as "<channel>/<dir>". +const indexed = () => + withIndex((db) => + [...db("mtimes").getKeys()].map((k) => (k as unknown as string[]).join("/")).sort(), + ); +const summaryCount = () => withIndex((db) => [...db("sums").getKeys()].length); +const storedSchema = () => withIndex((db) => db("meta").get("schema") as number); +const setStoredSchema = (v: number) => withIndex((db) => db("meta").putSync("schema", v)); +const pagesPending = () => withIndex((db) => db("meta").get(META_PAGES_PENDING)); + +// A directory tree as {relative path: contents}, or null when it is not there. +function tree(dir: string): Record<string, string> | null { + if (!existsSync(dir)) return null; + const out: Record<string, string> = {}; + const walk = (d: string) => { + for (const e of readdirSync(d, { withFileTypes: true })) { + const p = path.join(d, e.name); + if (e.isDirectory()) walk(p); + else out[path.relative(dir, p)] = readFileSync(p, "utf8"); + } + }; + walk(dir); + return out; +} +const sharedTranscripts = (slug = DRIVE_CHANNEL) => + tree(path.join(paths.exportSharedTranscriptsDir, slug)); +const sharedSubs = (slug = DRIVE_CHANNEL) => tree(path.join(paths.exportSharedSubsDir, slug)); + +type Published = { id: string; channelSlug?: string; state?: string; curatedTags?: string[] }; +const siteDir = () => path.join(paths.exportSitesIndexDir, SITE); +const summaries = (): Published[] => + JSON.parse(readFileSync(path.join(siteDir(), "summaries", "page-0000.json"), "utf8")); +const siteManifest = () => + JSON.parse(readFileSync(path.join(siteDir(), "summaries", "manifest.json"), "utf8")) as { + channels: { slug: string; count: number }[]; + }; +const siteSubsManifest = () => + JSON.parse(readFileSync(path.join(siteDir(), "subs", "manifest.json"), "utf8")) as { + channels: { slug: string; videoCount: number }[]; + }; +const stateOf = (id: string) => { + const r = summaries().find((s) => s.id === id); + assert.ok(r, `${id} is published`); + return r.state ?? "available"; +}; +const transcriptRecord = (id: string, slug = DRIVE_CHANNEL): Published => { + const page = JSON.parse( + readFileSync(path.join(paths.exportSharedTranscriptsDir, slug, "page-0000.json"), "utf8"), + ) as Published[]; + const r = page.find((s) => s.id === id); + assert.ok(r, `${id} is on its channel's transcript page`); + return r; +}; + +const DRIVE_VIDEOS = [`${DRIVE_CHANNEL}/d1`, `${DRIVE_CHANNEL}/d2`]; + +// ── the cases ─────────────────────────────────────────────────────────────── + +test("(a) an unmounted drive: the channel's records, transcript pages and subs survive an incremental build", async () => { + resetCorpus(); + seedVideo("local"); + seedDriveChannel(); + const mounted = await runIndex(); + assert.deepEqual(mounted.res.heldChannels, []); + assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]); + const pagesBefore = sharedTranscripts(); + const subsBefore = sharedSubs(); + assert.ok(pagesBefore?.["manifest.json"] && pagesBefore["page-0000.json"], "the drive's transcript pages"); + assert.ok(subsBefore?.["manifest.json"], "the drive's subs pages"); + + unmount(); + // Something else changed too, so the build rewrites the shared trees and + // the site: the hold has to survive a real build, not a no-op one. + seedVideo("local2"); + const { res, log } = await runIndex(); + + assert.deepEqual(res.heldChannels, [DRIVE_CHANNEL]); + assert.equal(res.added, 1); + assert.equal(res.removed, 0, "not read as a channel with no videos"); + assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`, `${CHANNEL}/local2`]); + assert.equal(summaryCount(), 4); + // Byte-identical, manifest included: its generatedAt shows no rewrite. + assert.deepEqual(sharedTranscripts(), pagesBefore); + assert.deepEqual(sharedSubs(), subsBefore); + // The site built from this index still publishes the channel. + for (const id of ["d1", "d2", "local", "local2"]) { + assert.ok(summaries().some((s) => s.id === id), `${id} still published`); + } + assert.equal(siteManifest().channels.find((c) => c.slug === DRIVE_CHANNEL)?.count, 2); + assert.equal(siteSubsManifest().channels.find((c) => c.slug === DRIVE_CHANNEL)?.videoCount, 1); + + // Said, with the location's label and no path. + const line = log.find((l) => l.startsWith(`Channel ${DRIVE_CHANNEL}:`)); + assert.ok(line, log.join("\n")); + assert.match( + line, + /its media is not reachable \(drive not mounted\?\), on location "USB drive"; its 2 indexed video\(s\) are kept as they are, not rescanned/, + ); + assert.ok(!line.includes(ROOT), line); + assert.ok( + log.some((l) => l.startsWith("Diff: +1 added, ~0 changed, -0 removed, 2 total. Held: 1 channel(s), 2 video(s) kept.")), + log.join("\n"), + ); + assert.ok( + log.some((l) => l.startsWith("Done in") && l.endsWith(`Held, their media not readable: ${DRIVE_CHANNEL}.`)), + log.join("\n"), + ); +}); + +test("(b) a held channel's availability states are carried over, not re-read from the missing drive", async () => { + resetCorpus(); + seedVideo("local"); + seedDriveChannel(); + // Both drive videos fell out of the channel's listing. d2 was confirmed + // public after that scan (available); d1 never was (maybe missing). + writeJson(path.join(paths.channelsDir, DRIVE_CHANNEL, "maybe-missing.json"), { + checkedAt: "2026-08-01T12:00:00.000Z", + freshPlaylistCount: 0, + ids: ["d1", "d2"], + }); + writeJson(path.join(videoDir("d2", DRIVE_CHANNEL), "availability.json"), { + checkedAt: "2026-08-02T00:00:00.000Z", + availability: "public", + }); + await runIndex(); + assert.deepEqual([stateOf("d1"), stateOf("d2")], ["maybe_missing", "available"]); + + unmount(); + seedVideo("local2"); // so the site is rebuilt + const { res } = await runIndex(); + assert.deepEqual(res.heldChannels, [DRIVE_CHANNEL]); + // Re-read from the missing drive, d2's confirmation would be gone and both + // would say "maybe missing"; skipped without the carry, d1's would vanish. + assert.deepEqual([stateOf("d1"), stateOf("d2")], ["maybe_missing", "available"]); + const videoState = withIndex((db) => + Object.fromEntries([...db("videoState").getRange()].map(({ key, value }) => [String(key).replace("\x00", "/"), value])), + ); + assert.deepEqual(videoState, { [`${DRIVE_CHANNEL}/d1`]: "maybe_missing" }); +}); + +test("(c) a full rebuild with a channel held refuses without the override, and holds with it", async () => { + resetCorpus(); + seedVideo("local"); + seedDriveChannel(); + await runIndex(); + const current = storedSchema(); + const pagesBefore = sharedTranscripts(); + const subsBefore = sharedSubs(); + + unmount(); + setStoredSchema(current - 1); + await assert.rejects(runIndex(), (err: Error) => { + assert.match( + err.message, + new RegExp( + `must be rebuilt in full \\(index schema ${current - 1} -> ${current}\\), but 1 channel\\(s\\) cannot be read: ` + + `drive-channel \\(its media is not reachable \\(drive not mounted\\?\\), on location "USB drive"\\)`, + ), + ); + // Why, the ways out (mounting first), and the override by name; no path. + assert.match(err.message, /the next site build would publish them as gone/); + assert.match( + err.message, + /For each: mount its media and run this again; or repair or re-point its location on \/storage; or finish or clear its move .*; or, if it is gone for good, delete the channel or set excludeFromBuild/, + ); + assert.match(err.message, new RegExp(`set ${INDEX_ALLOW_HELD_ENV}=1\\.$`)); + assert.ok(!err.message.includes(ROOT), err.message); + return true; + }); + // Refused before the clear: nothing touched. + assert.equal(storedSchema(), current - 1); + assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]); + assert.deepEqual(sharedTranscripts(), pagesBefore); + + // The CLI (the export's build:index, a site build's data phase) exits + // non-zero on it, and leaves the index as it was. + const cli = spawnSync( + path.join(COMMON, "node_modules", ".bin", "tsx"), + ["bin/archilyzer.ts", "index"], + { cwd: COMMON, env: { ...process.env }, encoding: "utf8" }, + ); + assert.notEqual(cli.status, 0, cli.stdout + cli.stderr); + assert.match(cli.stderr, /must be rebuilt in full/); + assert.equal(storedSchema(), current - 1); + assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]); + + // Overridden: the rebuild runs, the held channel's records go with the + // clear (they cannot be re-read), and its pages are left as they are. + process.env[INDEX_ALLOW_HELD_ENV] = "1"; + try { + const { res, log } = await runIndex(); + assert.deepEqual(res.heldChannels, [DRIVE_CHANNEL]); + assert.equal(storedSchema(), current); + assert.deepEqual(indexed(), [`${CHANNEL}/local`]); + assert.deepEqual(sharedTranscripts(), pagesBefore); + assert.deepEqual(sharedSubs(), subsBefore); + assert.ok( + log.some( + (l) => + l.startsWith(`Channel ${DRIVE_CHANNEL}: its media is not reachable`) && + l.includes(`held under ${INDEX_ALLOW_HELD_ENV}: this full rebuild cleared its index records`), + ), + log.join("\n"), + ); + } finally { + delete process.env[INDEX_ALLOW_HELD_ENV]; + } + + // The drive back: an ordinary build takes the channel in again. + remount(); + const back = await runIndex(); + assert.deepEqual(back.res.heldChannels, []); + assert.equal(back.res.added, 2); + assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]); + assert.equal(siteManifest().channels.find((c) => c.slug === DRIVE_CHANNEL)?.count, 2); +}); + +test("(d) a first build, with no index yet, is a full rebuild: it refuses with a channel held too", async () => { + resetCorpus(); + seedVideo("local"); + seedDriveChannel(); + unmount(); + await assert.rejects(runIndex(), /index schema <none> -> \d+\), but 1 channel\(s\) cannot be read: drive-channel/); + remount(); + const { res } = await runIndex(); + assert.deepEqual(res.heldChannels, []); + assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]); +}); + +test("(e) the drive back: the held set is empty and what arrived meanwhile is indexed", async () => { + resetCorpus(); + seedVideo("local"); + seedDriveChannel(); + await runIndex(); + unmount(); + const away = await runIndex(); + assert.deepEqual(away.res.heldChannels, [DRIVE_CHANNEL]); + assert.equal(away.res.removed, 0); + + // Downloaded onto the drive while it was elsewhere. + seedVideo("d3", DRIVE_CHANNEL, { dir: path.join(AWAY, DRIVE_CHANNEL, "data", "d3") }); + remount(); + const { res, log } = await runIndex(); + assert.deepEqual(res.heldChannels, []); + assert.equal(res.added, 1); + assert.equal(res.changed, 0, "the kept records match the disk"); + assert.equal(res.removed, 0); + assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${DRIVE_CHANNEL}/d3`, `${CHANNEL}/local`]); + assert.deepEqual(Object.keys(JSON.parse(sharedTranscripts()!["manifest.json"]).slugToPage).sort(), ["d1", "d2", "d3"]); + assert.ok(!log.some((l) => l.includes("Held")), log.join("\n")); +}); + +test("(f) a channel that is really empty is still emptied, not held", async () => { + resetCorpus(["emptied", "gone", CHANNEL]); + for (const slug of ["emptied", "gone"]) { + writeChannel(slug); + seedVideo("x1", slug); + seedVideo("x2", slug); + } + seedVideo("local"); + await runIndex(); + assert.equal(indexed().length, 5); + + // Its media deleted: an empty data/ in place, and no data/ at all. Neither + // was relocated, so there is no drive to be missing. + for (const id of ["x1", "x2"]) rmSync(videoDir(id, "emptied"), { recursive: true }); + rmSync(path.join(paths.channelsDir, "gone", "data"), { recursive: true }); + const { res, log } = await runIndex(); + assert.deepEqual(res.heldChannels, []); + assert.equal(res.removed, 4); + assert.deepEqual(indexed(), [`${CHANNEL}/local`]); + assert.deepEqual(JSON.parse(sharedTranscripts("emptied")!["manifest.json"]).slugToPage, {}); + // The missing directory is said, not swallowed. + assert.ok( + log.includes("Channel gone: no data/ directory; indexed as a channel with no videos."), + log.join("\n"), + ); +}); + +test("(g) a data directory that cannot be read holds its channel, and says why", async () => { + if (process.getuid?.() === 0) return; // root reads through a mode of 000 + resetCorpus(["locked", "flaky", CHANNEL]); + for (const slug of ["locked", "flaky"]) { + writeChannel(slug); + seedVideo("x1", slug); + seedVideo("x2", slug); + } + await runIndex(); + assert.equal(indexed().length, 4); + + // The whole data/ unreadable; and one video dir unreadable mid-walk. + const locked = path.join(paths.channelsDir, "locked", "data"); + const flaky = videoDir("x2", "flaky"); + chmodSync(locked, 0o000); + chmodSync(flaky, 0o000); + try { + const { res, log } = await runIndex(); + assert.deepEqual([...res.heldChannels].sort(), ["flaky", "locked"]); + assert.equal(res.removed, 0); + assert.equal(indexed().length, 4); + assert.ok( + log.some((l) => l.startsWith("Channel locked: its data directory could not be read (EACCES)")), + log.join("\n"), + ); + assert.ok( + log.some((l) => l.startsWith("Channel flaky: its data directory could not be read (EACCES)")), + log.join("\n"), + ); + } finally { + chmodSync(locked, 0o755); + chmodSync(flaky, 0o755); + } + const { res } = await runIndex(); + assert.deepEqual(res.heldChannels, []); + assert.equal(indexed().length, 4); +}); + +test("(h) a curated-tag change while a channel is held reaches its pages when the drive is back", async () => { + resetCorpus(); + seedVideo("local"); + seedDriveChannel({ d1: "Stream with Elfpire Eva" }); + await runIndex(); + assert.equal("curatedTags" in transcriptRecord("d1"), false); + + unmount(); + // A rule edit moves no mtime. It re-derives d1 in the index (no disk read), + // but d1's page is not rewritten while its channel is held. + writeGlobalTags(paths, { + version: 1, + tags: [ + { + id: "eva-collab", + label: "Collab", + group: "eva", + groupLabel: "Eva", + order: 1, + rules: [{ id: "meta", kind: "metadata" as const, pattern: "elfpire", enabled: true }], + }, + ], + assignments: {}, + }); + const pagesBefore = sharedTranscripts(); + const { log } = await runIndex(); + assert.deepEqual(sharedTranscripts(), pagesBefore); + assert.equal(pagesPending(), true, "the page debt is kept"); + assert.ok(log.some((l) => l.startsWith("curated tags: the pages of 1 held channel(s) are not rewritten while held")), log.join("\n")); + + remount(); + const back = await runIndex(); + assert.equal(back.res.added + back.res.changed + back.res.removed, 0, "no mtime moved"); + assert.deepEqual(transcriptRecord("d1").curatedTags, ["eva-collab"]); + assert.equal(pagesPending(), false); +}); + +test("(z) no write this file caused landed outside its temp root", () => { + // LMDB writes natively, past the spy: its file must be under the root too. + assert.ok(paths.lmdbPath.startsWith(ROOT + path.sep), paths.lmdbPath); + assert.ok(statSync(ROOT).isDirectory()); + const outside = writes.filter((p) => p !== ROOT && !p.startsWith(ROOT + path.sep)); + assert.deepEqual(outside, []); + assert.ok(writes.length > 0, "the spy saw the writes"); +}); diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -8,6 +8,22 @@ // // Per-channel config.json selects the transcript parser ("youtube" → VTT, // "transcribe" → whisper.cpp JSON). Short-circuits when mtimes already match. +// +// A CHANNEL WHOSE MEDIA IS NOT REACHABLE is HELD, not emptied: a relocated +// `data/` on an unmounted drive, one mid-relocation, a link and a config that +// disagree (inspectChannelMedia), or a data dir that fails to read. It is not +// rescanned; its index records are kept as they are and its shared page trees +// (transcripts, subs, digests) are left as they are, so the next site build +// still publishes it. Until this hold the scan read such a channel as having +// no videos, removed every record it had, and the site built next published +// the channel as gone. The stats build has the same hold (buildStats.ts); the +// words are shared (lib/channelMediaHold.ts). +// +// A FULL REBUILD (a schema change, or no index yet) with a channel held +// REFUSES: it clears every channel's records, and a held channel cannot be +// re-read, so it would come out empty. ARCHILYZER_INDEX_ALLOW_HELD=1 lets it +// proceed; the held channel is then out of the index until its media is back +// and the index is built again. import path from "node:path"; import { createHash } from "node:crypto"; @@ -78,6 +94,13 @@ import { type ChannelHandling, } from "../lib/channelConfig"; import { readChannelConfigFile } from "./channels"; +import { inspectChannelMedia } from "../lib/channelMedia"; +import { + HELD_WAYS_OUT, + describeHeld, + heldReason, + isMediaHeld, +} from "../lib/channelMediaHold"; import { resolveChannelGroupId } from "../lib/channelGroups"; import type { Paths } from "../lib/paths"; import { @@ -275,21 +298,32 @@ async function exists(p: string): Promise<boolean> { } } +function errCode(err: unknown): string { + const code = (err as NodeJS.ErrnoException | null)?.code; + return typeof code === "string" ? code : String(err); +} + +// `held` maps each channel whose media could not be read to why, in words with +// no path in them. A held channel contributes no live entries; the caller keeps +// its records and pages (see the file header). async function scanSource( channelsDir: string, log: (msg: string) => void, ): Promise<{ live: LiveEntry[]; channels: Map<string, ChannelConfig>; + held: Map<string, string>; }> { + const locations = getSettings().storage.locations; const channels = new Map<string, ChannelConfig>(); const live: LiveEntry[] = []; + const held = new Map<string, string>(); let channelEntries: Dirent[]; try { channelEntries = await readdir(channelsDir, { withFileTypes: true }); } catch { // Fresh transcripts dir with no channels yet. - return { live, channels }; + return { live, channels, held }; } for (const ch of channelEntries) { if (!ch.isDirectory()) continue; @@ -308,13 +342,33 @@ async function scanSource( // and routing it through the video scan would only ever produce noise. Its // posts tree is built from the JSONL shards further down. if (isSocialChannel(cfg)) continue; + // An unmounted drive is not an empty channel (lib/channelMedia.ts): the + // readdir below would fail, and every record the channel has would be + // removed as gone. + const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg); + if (isMediaHeld(media.status)) { + held.set(ch.name, heldReason(media, cfg.dataDir, locations)); + continue; + } const dataDir = path.join(channelDir, "data"); let videoEntries: Dirent[]; try { videoEntries = await readdir(dataDir, { withFileTypes: true }); - } catch { + } catch (err) { + const code = errCode(err); + // No data/ at all on a channel whose media was never moved: it has + // downloaded nothing yet (or its media was deleted), and it IS empty. + if (code === "ENOENT" && media.status === "in-place") { + log(`Channel ${ch.name}: no data/ directory; indexed as a channel with no videos.`); + continue; + } + // Anything else — a relocated drive gone between the check and the + // read, a permission or I/O error — is a channel that could not be read. + held.set(ch.name, `its data directory could not be read (${code})`); continue; } + const channelLive: LiveEntry[] = []; + let readFailure: string | null = null; for (const v of videoEntries) { if (!v.isDirectory()) continue; const videoDir = v.name; @@ -323,7 +377,16 @@ async function scanSource( let metaMs: number; try { metaMs = (await stat(metaPath)).mtimeMs; - } catch { + } catch (err) { + // ENOENT is a video dir with no metadata yet (a download in flight, a + // partial one): skipped, as always. Any other error is the channel's + // media failing mid-scan; the channel is held below rather than read + // as missing this video and every one after it. + const code = errCode(err); + if (code !== "ENOENT" && code !== "ENOTDIR") { + readFailure = code; + break; + } continue; } // Hybrid: a single channel may contain both YouTube auto-subs (.vtt) @@ -373,7 +436,7 @@ async function scanSource( // Sidecar absent — the common case (102 of ~76,000 videos have one). } } - live.push({ + channelLive.push({ channelSlug: ch.name, handling: cfg.handling, configName: cfg.name, @@ -389,8 +452,21 @@ async function scanSource( digestMs, }); } + if (readFailure !== null) { + held.set(ch.name, `its data directory could not be read (${readFailure})`); + continue; + } + // Asked again after the walk: a drive that went away DURING it leaves the + // videos after that point missing from this scan, which would remove them. + // Three syscalls a channel. + const after = await inspectChannelMedia({ channelsDir }, ch.name, cfg); + if (isMediaHeld(after.status)) { + held.set(ch.name, heldReason(after, cfg.dataDir, locations)); + continue; + } + for (const e of channelLive) live.push(e); } - return { live, channels }; + return { live, channels, held }; } function pathKeyId(k: PathKey): string { @@ -418,6 +494,9 @@ export type BuildIndexResult = { changed: number; removed: number; shortCircuited: boolean; + // Channels whose media could not be read, so they were not rescanned: their + // index records and shared pages were kept as they were (see the header). + heldChannels: string[]; }; export type BuildIndexOptions = { @@ -425,6 +504,17 @@ export type BuildIndexOptions = { onLog?: (msg: string) => void; }; +// Set to 1 (or true/yes/on) to let a FULL rebuild proceed with a channel held. +// Declared in lib/envVars.ts. +export const INDEX_ALLOW_HELD_ENV = "ARCHILYZER_INDEX_ALLOW_HELD"; +const TRUTHY = new Set(["1", "true", "yes", "on"]); +function allowsHeldFullRebuild( + env: Record<string, string | undefined> = process.env, +): boolean { + const raw = env.ARCHILYZER_INDEX_ALLOW_HELD; + return typeof raw === "string" && TRUTHY.has(raw.trim().toLowerCase()); +} + export async function buildIndex({ paths, onLog, @@ -535,6 +625,36 @@ export async function buildIndex({ const storedSchema = meta.get("schema") as number | undefined; const schemaBumped = storedSchema !== SCHEMA_VERSION; + + // Recorded as INDEX_SCANNED_AT_KEY only when this build completes, so + // buildStats can tell apart a video with no `mtimes` record: metadata newer + // than this is "not indexed yet"; older, and this build saw it and skipped it + // (no upload_date, or processing failed) or its channel was held. + // + // The scan reads only the source tree, so it runs BEFORE a schema clear: a + // full rebuild must know which channels it cannot read before it drops them. + const scanStartedAt = Date.now(); + const { + live, + channels: channelConfigs, + held, + } = await scanSource(channelsDir, log); + + // A full rebuild clears every channel's records, and a held channel cannot be + // re-read: it would come out of this build empty, and the site built next + // would publish it as gone. Refuse, unless told to go on without it. + const heldThroughClear = schemaBumped && held.size > 0; + if (heldThroughClear && !allowsHeldFullRebuild()) { + await root.close(); + throw new Error( + `The index must be rebuilt in full (index schema ${storedSchema ?? "<none>"} -> ${SCHEMA_VERSION}), ` + + `but ${held.size} channel(s) cannot be read: ${describeHeld(held)}. ` + + `A full rebuild clears every channel's index records, so these would come out empty and the next site build would publish them as gone. ` + + `${HELD_WAYS_OUT} ` + + `To rebuild without them anyway (each stays out of the index until its media is back and the index is built again), set ${INDEX_ALLOW_HELD_ENV}=1.`, + ); + } + if (schemaBumped) { log( `Schema change (${storedSchema ?? "<none>"} -> ${SCHEMA_VERSION}); invalidating LMDB cache.`, @@ -557,12 +677,6 @@ export async function buildIndex({ await meta.put("schema", SCHEMA_VERSION); } - // Recorded as INDEX_SCANNED_AT_KEY only when this build completes, so - // buildStats can tell apart a video with no `mtimes` record: metadata newer - // than this is "not indexed yet"; older, and this build saw it and skipped it - // (no upload_date, or processing failed). - const scanStartedAt = Date.now(); - const { live, channels: channelConfigs } = await scanSource(channelsDir, log); const livePathIds = new Set<string>(); const liveByPathId = new Map<string, LiveEntry>(); for (const s of live) { @@ -590,12 +704,29 @@ export async function buildIndex({ changed.push(s); } } + // A held channel has no live entries, and its records are not "gone": they + // are kept, and counted for the log. + const keptHeld = new Map<string, number>(); for (const { key, value } of mtimes.getRange()) { const k = key as PathKey; + if (held.has(k[0])) { + keptHeld.set(k[0], (keptHeld.get(k[0]) ?? 0) + 1); + continue; + } if (!livePathIds.has(pathKeyId(k))) { removed.push({ pathKey: k, indexKey: (value as MtimeRecord).indexKey }); } } + let keptHeldTotal = 0; + for (const [slug, why] of held) { + const kept = keptHeld.get(slug) ?? 0; + keptHeldTotal += kept; + log( + heldThroughClear + ? `Channel ${slug}: ${why}; held under ${INDEX_ALLOW_HELD_ENV}: this full rebuild cleared its index records, so it is out of the index until its media is back and the index is built again. Its transcript, subtitle and digest pages are left as they are.` + : `Channel ${slug}: ${why}; its ${kept} indexed video(s) are kept as they are, not rescanned, and its transcript, subtitle and digest pages are left as they are.`, + ); + } const anyMutations = added.length > 0 || changed.length > 0 || removed.length > 0; @@ -608,6 +739,8 @@ export async function buildIndex({ // from LMDB further below so they reflect current site config. const sharedManifestsPresent = async (): Promise<boolean> => { for (const channelSlug of channelConfigs.keys()) { + // A held channel's pages are not written this build either way. + if (held.has(channelSlug)) continue; const mPath = path.join(transcriptsOutDir, channelSlug, "manifest.json"); const raw = await readFile(mPath, "utf8").catch(() => null); if (!raw) return false; @@ -638,7 +771,10 @@ export async function buildIndex({ const curatedFresh = new Set<string>(); log( - `Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`, + `Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.` + + (held.size > 0 + ? ` Held: ${held.size} channel(s), ${keptHeldTotal} video(s) kept.` + : ""), ); for (const channelSlug of channelConfigs.keys()) { @@ -1056,6 +1192,10 @@ export async function buildIndex({ if (sharedNeedsBuild) { for (const channelSlug of Array.from(channelConfigs.keys()).sort()) { + // A held channel's pages are left exactly as the last build wrote them: + // not rewritten, not pruned, and (below) not removed. After a full rebuild + // its records are gone, and a rewrite would publish it empty. + if (held.has(channelSlug)) continue; const channelDir = path.join(transcriptsOutDir, channelSlug); await mkdir(channelDir, { recursive: true }); @@ -1172,6 +1312,10 @@ export async function buildIndex({ // availability.json for just those ids is cheap and exact. let maybeMissingCount = 0; for (const slug of channelConfigs.keys()) { + // A held channel's availability.json files are on the media that cannot be + // read, and a missing one reads as "maybe missing": its states are carried + // over from the last build instead (below). + if (held.has(slug)) continue; const record = await loadMaybeMissing(paths, slug); if (!record?.ids.length) continue; const scannedAtMs = Date.parse(record.checkedAt); @@ -1199,6 +1343,25 @@ export async function buildIndex({ ); } + // A held channel keeps the states the last build published for it (the + // confirmed ones above come from its kept records; this adds the overlay's), + // read back before the wholesale rewrite below. Nothing to carry after a full + // rebuild: the clear took them, with the records they described. + if (held.size > 0) { + for (const { key, value } of videoState.getRange()) { + const id = key as string; + const cut = id.indexOf("\x00"); + if (cut < 0) continue; + const slug = id.slice(0, cut); + if (!held.has(slug)) continue; + const rec = mtimes.get([slug, id.slice(cut + 1)]); + if (!rec) continue; + const st = value as VideoState; + stateByIndexKey.set(indexKeyId(rec.indexKey), st); + stateByPath.set(id, st); + } + } + // Publish the sparse map for buildStats, which runs after us against the same // LMDB file and would otherwise have to re-read ~76k availability.json files // to build the status chart. Rewritten wholesale each build: the map is small @@ -1219,6 +1382,16 @@ export async function buildIndex({ if (sharedNeedsBuild) { for (const channelSlug of Array.from(channelConfigs.keys()).sort()) { + // Held: its subs dir is left as it is, and its stats carried over so the + // site manifests still list it (none to carry after a full rebuild). + if (held.has(channelSlug)) { + const prev = channelStatsDb.get(channelSlug); + if (prev) { + channelStats.set(channelSlug, prev); + subsTotalCount += prev.videoCount; + } + continue; + } const cfg = channelConfigs.get(channelSlug)!; const subsChannelDir = path.join(subsOutDir, channelSlug); const tracksInChannel = new Set<string>(); @@ -1331,7 +1504,7 @@ export async function buildIndex({ const subsChannelSlugSet = new Set(channelStats.keys()); for (const e of topSubsEntries) { if (e.isDirectory()) { - if (!subsChannelSlugSet.has(e.name)) { + if (!subsChannelSlugSet.has(e.name) && !held.has(e.name)) { await rm(path.join(subsOutDir, e.name), { recursive: true, force: true, @@ -1365,7 +1538,17 @@ export async function buildIndex({ // curated-tag page debt is settled. Deliberately AFTER the page build and not // beside the hashes: an interrupt anywhere above must leave the flag standing // so the next build rewrites the shards. - clearCuratedPagesPending(meta); + // + // Except for a held channel's pages, which were not written: the re-apply + // pass re-derives its records in LMDB like any other, and its pages owe them. + // The flag stays, so the first build with its media back rewrites them. + if (held.size > 0 && curatedReapply.pagesPending) { + log( + `curated tags: the pages of ${held.size} held channel(s) are not rewritten while held; the re-derived tags reach them on the first build with their media back.`, + ); + } else { + clearCuratedPagesPending(meta); + } await meta.flushed; // --------------------------------------------------------------------------- @@ -1578,6 +1761,16 @@ export async function buildIndex({ if (sharedNeedsBuild) { for (const channelSlug of Array.from(channelConfigs.keys()).sort()) { + // Held: as with subs, its digest dir is left as it is and its stats + // carried over. + if (held.has(channelSlug)) { + const prev = channelDigestStatsDb.get(channelSlug); + if (prev) { + channelDigestStats.set(channelSlug, prev); + digestTotalCount += prev.digestCount; + } + continue; + } const cfg = channelConfigs.get(channelSlug)!; const digestChannelDir = path.join(digestsOutDir, channelSlug); let digestCount = 0; @@ -1677,7 +1870,7 @@ export async function buildIndex({ }).catch(() => [] as Dirent[]); for (const e of topDigestEntries) { if (e.isDirectory()) { - if (!channelDigestStats.has(e.name)) { + if (!channelDigestStats.has(e.name) && !held.has(e.name)) { await rm(path.join(digestsOutDir, e.name), { recursive: true, force: true, @@ -2010,7 +2203,10 @@ export async function buildIndex({ const durationMs = Date.now() - t0; log( - `Done in ${(durationMs / 1000).toFixed(2)}s. ${sites.length} site(s): ${sitesBuilt} built, ${sitesSkipped} up to date; ${live.length} transcripts in pool.`, + `Done in ${(durationMs / 1000).toFixed(2)}s. ${sites.length} site(s): ${sitesBuilt} built, ${sitesSkipped} up to date; ${live.length} transcripts in pool.` + + (held.size > 0 + ? ` Held, their media not readable: ${[...held.keys()].join(", ")}.` + : ""), ); return { totalCount: aggregateSummaries, @@ -2024,5 +2220,6 @@ export async function buildIndex({ changed: changed.length, removed: removed.length, shortCircuited: !sharedNeedsBuild && sitesBuilt === 0, + heldChannels: [...held.keys()], }; } diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts @@ -112,6 +112,7 @@ const DECLARED: EnvVarDecl[] = [ { name: "AUDIO_CHECK_RESUME_DURING_PROBE", audience: "runtime", default: "the channel's `audioCheck.resumeDuringProbe`", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "`1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting." }, { name: "AUDIO_CHECK_BACKOFF_FACTOR", audience: "runtime", default: "the built-in factor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "The audio check's interval backoff factor, in (0, 1], for a one-off run." }, { name: "ARCHILYZER_STATS_ALLOW_DOWNGRADE", audience: "runtime", default: "off", readBy: "common/controller/buildStats.ts", doc: "`1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions." }, + { name: "ARCHILYZER_INDEX_ALLOW_HELD", audience: "runtime", default: "off", readBy: "common/controller/buildIndex.ts", doc: "`1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel." }, { name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." }, { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." }, { name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },