commit 33d5b9b21bedfa44939ec1f5d485430a105a08a9
parent c8ff4c7e7b85427ce5eddd5f2221c896a5420df3
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 29 Sep 2026 21:57:52 -0400
Merge main (release 15 IG, release 14 HS) into r14/first-search — the two record conflicts kept both sides: the export changelog's [Unreleased] carries HS's bullet then S1's; release-14.md carries the HS and S1 rows, HS's "as shipped" section then S1's before the Rollout, and HS's Rollout edits with S1's live checks after them
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
32 files changed, 2070 insertions(+), 111 deletions(-)
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -76,6 +76,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| `AUDIO_CHECK_RESUME_DURING_PROBE` | the channel's `audioCheck.resumeDuringProbe` | `1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting. | common/ytdlp/audioCheckedDownload.ts |
| `AUDIO_CHECK_BACKOFF_FACTOR` | the built-in factor | The audio check's interval backoff factor, in (0, 1], for a one-off run. | common/ytdlp/audioCheckedDownload.ts |
| `ARCHILYZER_STATS_ALLOW_DOWNGRADE` | off | `1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions. | common/controller/buildStats.ts |
+| `ARCHILYZER_INDEX_ALLOW_HELD` | off | `1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel. | common/controller/buildIndex.ts |
| `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts |
| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool |
| `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs |
diff --git a/SITE.md b/SITE.md
@@ -23,6 +23,7 @@ Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/f
| [`cloudflareProject`](#cloudflareproject) | absent |
| [`accent`](#accent) | absent |
| [`siteUrl`](#siteurl) | absent |
+| [`listed`](#listed) | `true` |
| [`relatedSites`](#relatedsites) | `[]` |
| [`pwa`](#pwa) | `false` |
| [`archives`](#archives) | `true` |
@@ -157,6 +158,12 @@ Absolute public URL of this site's deployment, e.g. `https://jeralyzer.pages.dev
Default: absent
+## `listed`
+
+Whether the family lists this site. Opt-OUT: absent/true = listed, only an explicit `false` is written. An unlisted site still builds and deploys as before, and its own pages are unchanged; it is left out of the homepage (cards, chart, `/stats`), the hub (members, federated search, `/corpus.json`, `/llms.txt`), every other site's footer, and the published `channel-sites.json` and pooled `stats/`. A channel only unlisted sites expose is in none of the family's public totals; a channel a listed site also exposes is credited to the listed one.
+
+Default: `true`
+
## `relatedSites`
Pulls specific siblings to the front of the footer's cross-site list, in named groups. Siblings not named here fall into a trailing "Other sites" group. Absent/empty = one flat list of every sibling.
diff --git a/common/bin/compose-homepage.ts b/common/bin/compose-homepage.ts
@@ -8,6 +8,9 @@
// public/channel-sites.json <- channel slug -> [siteId, ...]
// public/homepage-summary.json <- small cross-site landing summary
//
+// None of the three names an unlisted site (site.json `listed: false`) or holds
+// a channel only unlisted sites expose (lib/siteSchema.ts isListedSite).
+//
// Requires build:index to have populated the cues LMDB first (the homepage
// prebuild chains it), same as the export pipeline.
@@ -15,6 +18,7 @@ import path from "node:path";
import { getPaths, type Paths } from "../lib/paths";
import { writeJsonAtomic as writeJsonAtomicShared } from "../lib/jsonFile-server";
import { buildPoolSummary } from "../controller/poolSummary";
+import { isListedSite } from "../lib/site";
import { runIfEntryPoint } from "./_cli";
// Where the homepage Next.js app serves static assets from. Overridable for e2e
@@ -45,7 +49,7 @@ export async function main(opts: { paths?: Paths } = {}): Promise<void> {
statsDir,
});
- // channel slug -> the ids of the content sites that expose it. Drives
+ // channel slug -> the ids of the listed content sites that expose it. Drives
// `groupBy: "site"` in the hub's dashboard (see channelSites.tsx).
await writeJsonAtomic(
path.join(publicDir, "channel-sites.json"),
@@ -60,8 +64,11 @@ export async function main(opts: { paths?: Paths } = {}): Promise<void> {
summary,
);
+ const unlisted = sites.filter((s) => !isListedSite(s)).length;
console.log(
- `compose-homepage: ${Object.keys(channelSites).length} channel(s) mapped across ${sites.length} site(s); ` +
+ `compose-homepage: ${Object.keys(channelSites).length} channel(s) mapped across ${sites.length - unlisted} listed site(s)` +
+ (unlisted > 0 ? ` (${unlisted} unlisted left out)` : "") +
+ "; " +
`summary covers ${summary.totals.transcripts} transcription(s) / ${summary.totals.downloads} download(s) ` +
`across ${summary.sites.length} public site(s) into ${publicDir}.`,
);
diff --git a/common/bin/compose-hub.test.ts b/common/bin/compose-hub.test.ts
@@ -130,3 +130,43 @@ test("in a worktree, compose-hub writes its own files and never through the link
rmSync(root, { recursive: true, force: true });
}
});
+
+// Release 14 slice HS: an unlisted site builds and deploys, and the hub does
+// not list it — not a member, so not in federated search, corpus.json or
+// llms.txt.
+test("an unlisted site is in none of the hub's files; a listed one is in each", async () => {
+ const root = mkdtempSync(path.join(tmpdir(), "compose-hub-"));
+ const log = console.log;
+ try {
+ const paths = fixturePaths(root);
+ const write = (id: string, site: Record<string, unknown>) => {
+ mkdirSync(path.join(paths.sitesDir, id), { recursive: true });
+ writeFileSync(path.join(paths.sitesDir, id, "site.json"), JSON.stringify(site));
+ };
+ write("fixture-shown", { siteTitle: "Shown Fixture", siteUrl: "https://fixture-shown.example" });
+ write("fixture-unlisted", {
+ siteTitle: "Unlisted Fixture",
+ siteUrl: "https://fixture-unlisted.example",
+ listed: false,
+ });
+ const lines: string[] = [];
+ console.log = (...a: unknown[]) => void lines.push(a.join(" "));
+ await main({ paths });
+ console.log = log;
+ const pool = JSON.parse(
+ readFileSync(path.join(paths.exportPublicDir, "hub-sites.json"), "utf8"),
+ ) as Array<{ siteId: string }>;
+ assert.deepEqual(pool.map((s) => s.siteId), ["fixture-shown"]);
+ for (const f of ["hub-sites.json", "corpus.json", "llms.txt"]) {
+ const text = readFileSync(path.join(paths.exportPublicDir, f), "utf8");
+ assert.ok(text.includes("fixture-shown.example"), `${f} lists the listed site`);
+ for (const needle of ["fixture-unlisted", "Unlisted Fixture"]) {
+ assert.ok(!text.includes(needle), `${f} names ${needle}`);
+ }
+ }
+ assert.match(lines.join("\n"), /compose-hub: 1 built-in pool site\(s\)/);
+ } finally {
+ console.log = log;
+ rmSync(root, { recursive: true, force: true });
+ }
+});
diff --git a/common/bin/compose-hub.ts b/common/bin/compose-hub.ts
@@ -4,7 +4,8 @@
// there is no SITE_ID and no per-site data: the hub is a federating shell that
// reads every archive cross-origin at runtime. It emits:
//
-// public/hub-sites.json <- the built-in trusted pool (listSites with a siteUrl)
+// public/hub-sites.json <- the built-in trusted pool (listSites with a siteUrl,
+// listed — site.json `listed`, isListedSite)
// public/hub-summary.json <- the official instances' numbers, the homepage's
// own (lib/hubSummary.ts) — OPTIONAL: skipped when
// there is no index to walk
@@ -19,7 +20,7 @@ import { existsSync } from "node:fs";
import { cp, rm, access } from "node:fs/promises";
import { getPaths, type Paths } from "../lib/paths";
import { accentHex } from "../lib/accent";
-import { listSites, resolveHubUrl } from "../lib/site";
+import { isListedSite, listSites, resolveHubUrl } from "../lib/site";
import { getHomepageConfig } from "../lib/homepage";
import { SITE_DESCRIPTOR_VERSION } from "../lib/siteDescriptor";
import {
@@ -79,7 +80,10 @@ export async function main(opts: { paths?: Paths } = {}): Promise<void> {
const paths = opts.paths ?? getPaths();
const publicDir = paths.exportPublicDir;
- // Built-in pool: every configured site that publishes a public URL. The entry
+ // Built-in pool: every configured site that publishes a public URL and is
+ // listed. An unlisted site (`listed: false`) still builds and deploys, but the
+ // hub does not list it: not a member, not in federated search, not in the
+ // hub's corpus.json or llms.txt (both are built from this list). The entry
// the hub registry loads at boot to seed its trusted built-in pool, and the
// input buildHubCorpus maps — ONE type for both (lib/archive/contract.ts
// HubMemberInput), where this file used to restate it as a local
@@ -88,7 +92,7 @@ export async function main(opts: { paths?: Paths } = {}): Promise<void> {
// a siteUrl can't be federated and is dropped — by both consumers.
const builtins: HubMemberInput[] = [];
for (const site of listSites(paths)) {
- if (!site.siteUrl) continue;
+ if (!site.siteUrl || !isListedSite(site)) continue;
builtins.push({
siteId: site.siteId,
siteTitle: site.siteTitle,
diff --git a/common/controller/buildIndex.test.ts b/common/controller/buildIndex.test.ts
@@ -0,0 +1,645 @@
+// Integration: the index build's HOLD, through the REAL buildIndex, over a temp
+// corpus.
+//
+// A channel's `data/` may be an absolute symlink to another drive (AGENTS.md,
+// "A channel's `data/` may live on another drive"). With that drive unmounted
+// the link dangles, and the index build used to read the channel as having no
+// videos: it removed every record the channel had, and the site built next
+// published the channel as gone. These cases pin the hold that replaced it:
+// the channel is not rescanned, and its records and shared pages are kept as
+// they are; a FULL rebuild with a channel held refuses unless
+// ARCHILYZER_INDEX_ALLOW_HELD is set; and all of it undoes itself when the
+// drive is back. The stats build's twin is buildStats.test.ts case (i).
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildIndex.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { spawnSync } from "node:child_process";
+import { createRequire, syncBuiltinESMExports } from "node:module";
+import {
+ chmodSync,
+ existsSync,
+ mkdirSync,
+ mkdtempSync,
+ readdirSync,
+ readFileSync,
+ renameSync,
+ rmSync,
+ statSync,
+ symlinkSync,
+ writeFileSync,
+} from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+// EVERY PATH getPaths() CAN RESOLVE TO A PLACE THIS FILE'S CODE MAY WRITE IS
+// PINNED UNDER ROOT before anything calls it (the buildStats.test.ts list). The
+// last case proves no write this file caused landed outside ROOT.
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_INDEX_ALLOW_HELD;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex, INDEX_ALLOW_HELD_ENV } = await import("./buildIndex");
+const { writeGlobalTags } = await import("../lib/curatedTagsStore");
+const { META_PAGES_PENDING } = await import("./curatedTagsIndex");
+const { open } = await import("lmdb");
+
+const paths = getPaths();
+const CHANNEL = "test-channel";
+const DRIVE_CHANNEL = "drive-channel";
+const SITE = "testsite";
+const MEDIA = path.join(ROOT, "media");
+const AWAY = `${MEDIA}-away`;
+const COMMON = fileURLToPath(new URL("..", import.meta.url));
+
+// ── a write spy over the whole file ─────────────────────────────────────────
+// The buildStats.test.ts spy, writes only: node:fs and node:fs/promises, async,
+// sync and callback, synced into the named ESM imports the code under test
+// holds. The last case reads it.
+const writes: string[] = [];
+let afterStat: ((p: string) => void) | null = null;
+{
+ const req = createRequire(import.meta.url);
+ const fsCjs = req("node:fs") as Record<string, unknown>;
+ const fspCjs = req("node:fs/promises") as Record<string, unknown>;
+ const WRITES = ["writeFile", "appendFile", "rename", "mkdir", "rm", "rmdir", "unlink", "copyFile", "cp", "symlink", "link", "utimes", "truncate", "mkdtemp", "chmod"];
+ const TWO_PATHS = new Set(["rename", "copyFile", "cp", "symlink", "link"]);
+ const opensForWrite = (flags: unknown) =>
+ (typeof flags === "string" && /[wa+]/.test(flags)) ||
+ (typeof flags === "number" && (flags & 3) !== 0);
+ const asPath = (v: unknown) =>
+ typeof v === "string" ? v : v instanceof URL ? fileURLToPath(v) : Buffer.isBuffer(v) ? v.toString() : null;
+ const wrap = (mod: Record<string, unknown>, name: string, mode: "write" | "open") => {
+ const fn = mod[name];
+ if (typeof fn !== "function") return;
+ mod[name] = function (this: unknown, ...args: unknown[]) {
+ if (mode === "write" || opensForWrite(args[1])) {
+ const ps = TWO_PATHS.has(name.replace(/Sync$/, "")) ? [args[0], args[1]] : [args[0]];
+ for (const a of ps) {
+ const p = asPath(a);
+ if (p !== null) writes.push(path.resolve(p));
+ }
+ }
+ return (fn as (...a: unknown[]) => unknown).apply(this, args);
+ };
+ };
+ for (const n of WRITES) {
+ wrap(fspCjs, n, "write");
+ wrap(fsCjs, n, "write");
+ wrap(fsCjs, `${n}Sync`, "write");
+ }
+ wrap(fspCjs, "open", "open");
+ wrap(fsCjs, "open", "open");
+ wrap(fsCjs, "openSync", "open");
+ wrap(fsCjs, "createWriteStream", "write");
+ // Case (i)'s hook: the drive is lost DURING the scan's walk. node:fs/promises
+ // `stat` calls `afterStat` with each path it has just answered for.
+ const stat = fspCjs.stat as (...a: unknown[]) => Promise<unknown>;
+ fspCjs.stat = async function (this: unknown, ...args: unknown[]) {
+ const result = await stat.apply(this, args);
+ const p = asPath(args[0]);
+ if (p !== null) afterStat?.(path.resolve(p));
+ return result;
+ };
+ syncBuiltinESMExports();
+}
+
+// ── the corpus ──────────────────────────────────────────────────────────────
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const videoDir = (id: string, channel = CHANNEL) =>
+ path.join(paths.channelsDir, channel, "data", id);
+
+// YouTube's rolling-caption shape: parseVtt keeps only lines carrying inline
+// timing tags, so a plain cue would parse to nothing.
+const VTT =
+ "WEBVTT\nKind: captions\nLanguage: en\n\n" +
+ "00:00:00.000 --> 00:00:05.000 align:start position:0%\n" +
+ "First<00:00:01.000><c> caption</c><00:00:02.000><c> line.</c>\n\n" +
+ "00:01:00.000 --> 00:01:50.000 align:start position:0%\n" +
+ "Second<00:01:10.000><c> caption</c><00:01:20.000><c> line.</c>\n";
+
+// Metadata and English captions; `subs` adds a German track, which the index
+// publishes in the shared subs tree.
+function seedVideo(
+ id: string,
+ channel = CHANNEL,
+ opts: { title?: string; subs?: boolean; dir?: string } = {},
+): void {
+ const dir = opts.dir ?? videoDir(id, channel);
+ writeJson(path.join(dir, "metadata.info.json"), {
+ id,
+ title: opts.title ?? `Video ${id}`,
+ channel: channel,
+ upload_date: "20260601",
+ duration: 120,
+ description: "fixture",
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ writeFileSync(path.join(dir, "transcript.en.vtt"), VTT);
+ if (opts.subs) writeFileSync(path.join(dir, "transcript.de.vtt"), VTT);
+}
+
+function writeChannel(slug: string, extra: Record<string, unknown> = {}): void {
+ writeJson(path.join(paths.channelsDir, slug, "config.json"), {
+ handling: "youtube",
+ name: slug,
+ url: `https://www.youtube.com/@${slug}/videos`,
+ ...extra,
+ });
+}
+
+// A fresh corpus, LMDB and export tree per test, so every count is exact. The
+// drive is a storage location, as /storage records it: a held channel is named
+// by its label, never by a path.
+function resetCorpus(channels: string[] = [CHANNEL, DRIVE_CHANNEL]): void {
+ for (const p of [paths.transcriptsDir, PINNED.EXPORT_INDEX_DIR, MEDIA, AWAY]) {
+ rmSync(p, { recursive: true, force: true });
+ }
+ mkdirSync(paths.transcriptsDir, { recursive: true });
+ writeFileSync(
+ paths.settingsFile,
+ JSON.stringify({
+ storage: { locations: [{ id: "usb", label: "USB drive", root: MEDIA, autoRepoint: false }] },
+ }),
+ );
+ writeChannel(CHANNEL);
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: channels.map((slug) => ({ slug, groupId: "default" })),
+ });
+}
+
+// A channel whose data/ is a relocated symlink, the way the editor's Storage
+// panel leaves it: channels/<slug>/data -> <MEDIA>/<slug>/data, with
+// config.dataDir recording the target. d1 carries a subtitle track.
+function seedDriveChannel(titles: Record<string, string> = {}): void {
+ const target = path.join(MEDIA, DRIVE_CHANNEL, "data");
+ mkdirSync(target, { recursive: true });
+ writeChannel(DRIVE_CHANNEL, { dataDir: target });
+ symlinkSync(target, path.join(paths.channelsDir, DRIVE_CHANNEL, "data"));
+ seedVideo("d1", DRIVE_CHANNEL, { subs: true, title: titles.d1 });
+ seedVideo("d2", DRIVE_CHANNEL, { title: titles.d2 });
+}
+
+// Unmount: the link now dangles, exactly as an absent USB drive leaves it.
+const unmount = () => renameSync(MEDIA, AWAY);
+const remount = () => renameSync(AWAY, MEDIA);
+
+async function runIndex(log: string[] = []) {
+ const res = await buildIndex({ paths, onLog: (s) => log.push(s) });
+ return { res, log };
+}
+
+// ── reading what the build left ─────────────────────────────────────────────
+// The index LMDB, opened the way buildIndex opens it, and closed again.
+function withIndex<T>(fn: (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>) => T): T {
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ try {
+ return fn((name) => root.openDB({ name, encoding: "msgpack" }));
+ } finally {
+ root.close();
+ }
+}
+// Every indexed video, as "<channel>/<dir>".
+const indexed = () =>
+ withIndex((db) =>
+ [...db("mtimes").getKeys()].map((k) => (k as unknown as string[]).join("/")).sort(),
+ );
+const summaryCount = () => withIndex((db) => [...db("sums").getKeys()].length);
+const storedSchema = () => withIndex((db) => db("meta").get("schema") as number);
+const setStoredSchema = (v: number) => withIndex((db) => db("meta").putSync("schema", v));
+const pagesPending = () => withIndex((db) => db("meta").get(META_PAGES_PENDING));
+
+// A directory tree as {relative path: contents}, or null when it is not there.
+function tree(dir: string): Record<string, string> | null {
+ if (!existsSync(dir)) return null;
+ const out: Record<string, string> = {};
+ const walk = (d: string) => {
+ for (const e of readdirSync(d, { withFileTypes: true })) {
+ const p = path.join(d, e.name);
+ if (e.isDirectory()) walk(p);
+ else out[path.relative(dir, p)] = readFileSync(p, "utf8");
+ }
+ };
+ walk(dir);
+ return out;
+}
+const sharedTranscripts = (slug = DRIVE_CHANNEL) =>
+ tree(path.join(paths.exportSharedTranscriptsDir, slug));
+const sharedSubs = (slug = DRIVE_CHANNEL) => tree(path.join(paths.exportSharedSubsDir, slug));
+
+type Published = { id: string; channelSlug?: string; state?: string; curatedTags?: string[] };
+const siteDir = () => path.join(paths.exportSitesIndexDir, SITE);
+const summaries = (): Published[] =>
+ JSON.parse(readFileSync(path.join(siteDir(), "summaries", "page-0000.json"), "utf8"));
+const siteManifest = () =>
+ JSON.parse(readFileSync(path.join(siteDir(), "summaries", "manifest.json"), "utf8")) as {
+ channels: { slug: string; count: number }[];
+ };
+const siteSubsManifest = () =>
+ JSON.parse(readFileSync(path.join(siteDir(), "subs", "manifest.json"), "utf8")) as {
+ channels: { slug: string; videoCount: number }[];
+ };
+const stateOf = (id: string) => {
+ const r = summaries().find((s) => s.id === id);
+ assert.ok(r, `${id} is published`);
+ return r.state ?? "available";
+};
+const transcriptRecord = (id: string, slug = DRIVE_CHANNEL): Published => {
+ const page = JSON.parse(
+ readFileSync(path.join(paths.exportSharedTranscriptsDir, slug, "page-0000.json"), "utf8"),
+ ) as Published[];
+ const r = page.find((s) => s.id === id);
+ assert.ok(r, `${id} is on its channel's transcript page`);
+ return r;
+};
+
+const DRIVE_VIDEOS = [`${DRIVE_CHANNEL}/d1`, `${DRIVE_CHANNEL}/d2`];
+
+// ── the cases ───────────────────────────────────────────────────────────────
+
+test("(a) an unmounted drive: the channel's records, transcript pages and subs survive an incremental build", async () => {
+ resetCorpus();
+ seedVideo("local");
+ seedDriveChannel();
+ const mounted = await runIndex();
+ assert.deepEqual(mounted.res.heldChannels, []);
+ assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]);
+ const pagesBefore = sharedTranscripts();
+ const subsBefore = sharedSubs();
+ assert.ok(pagesBefore?.["manifest.json"] && pagesBefore["page-0000.json"], "the drive's transcript pages");
+ assert.ok(subsBefore?.["manifest.json"], "the drive's subs pages");
+
+ unmount();
+ // Something else changed too, so the build rewrites the shared trees and
+ // the site: the hold has to survive a real build, not a no-op one.
+ seedVideo("local2");
+ const { res, log } = await runIndex();
+
+ assert.deepEqual(res.heldChannels, [DRIVE_CHANNEL]);
+ assert.equal(res.added, 1);
+ assert.equal(res.removed, 0, "not read as a channel with no videos");
+ assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`, `${CHANNEL}/local2`]);
+ assert.equal(summaryCount(), 4);
+ // Byte-identical, manifest included: its generatedAt shows no rewrite.
+ assert.deepEqual(sharedTranscripts(), pagesBefore);
+ assert.deepEqual(sharedSubs(), subsBefore);
+ // The site built from this index still publishes the channel.
+ for (const id of ["d1", "d2", "local", "local2"]) {
+ assert.ok(summaries().some((s) => s.id === id), `${id} still published`);
+ }
+ assert.equal(siteManifest().channels.find((c) => c.slug === DRIVE_CHANNEL)?.count, 2);
+ assert.equal(siteSubsManifest().channels.find((c) => c.slug === DRIVE_CHANNEL)?.videoCount, 1);
+
+ // Said, with the location's label and no path.
+ const line = log.find((l) => l.startsWith(`Channel ${DRIVE_CHANNEL}:`));
+ assert.ok(line, log.join("\n"));
+ assert.match(
+ line,
+ /its media is not reachable \(drive not mounted\?\), on location "USB drive"; its 2 indexed video\(s\) are kept as they are, not rescanned/,
+ );
+ assert.ok(!line.includes(ROOT), line);
+ assert.ok(
+ log.some((l) => l.startsWith("Diff: +1 added, ~0 changed, -0 removed, 2 total. Held: 1 channel(s), 2 video(s) kept.")),
+ log.join("\n"),
+ );
+ assert.ok(
+ log.some((l) => l.startsWith("Done in") && l.endsWith(`Held, their media not readable: ${DRIVE_CHANNEL}.`)),
+ log.join("\n"),
+ );
+});
+
+test("(b) a held channel's availability states are carried over, not re-read from the missing drive", async () => {
+ resetCorpus();
+ seedVideo("local");
+ seedDriveChannel();
+ // Both drive videos fell out of the channel's listing. d2 was confirmed
+ // public after that scan (available); d1 never was (maybe missing).
+ writeJson(path.join(paths.channelsDir, DRIVE_CHANNEL, "maybe-missing.json"), {
+ checkedAt: "2026-08-01T12:00:00.000Z",
+ freshPlaylistCount: 0,
+ ids: ["d1", "d2"],
+ });
+ writeJson(path.join(videoDir("d2", DRIVE_CHANNEL), "availability.json"), {
+ checkedAt: "2026-08-02T00:00:00.000Z",
+ availability: "public",
+ });
+ await runIndex();
+ assert.deepEqual([stateOf("d1"), stateOf("d2")], ["maybe_missing", "available"]);
+
+ unmount();
+ seedVideo("local2"); // so the site is rebuilt
+ const { res } = await runIndex();
+ assert.deepEqual(res.heldChannels, [DRIVE_CHANNEL]);
+ // Re-read from the missing drive, d2's confirmation would be gone and both
+ // would say "maybe missing"; skipped without the carry, d1's would vanish.
+ assert.deepEqual([stateOf("d1"), stateOf("d2")], ["maybe_missing", "available"]);
+ const videoState = withIndex((db) =>
+ Object.fromEntries([...db("videoState").getRange()].map(({ key, value }) => [String(key).replace("\x00", "/"), value])),
+ );
+ assert.deepEqual(videoState, { [`${DRIVE_CHANNEL}/d1`]: "maybe_missing" });
+});
+
+test("(c) a full rebuild with a channel held refuses without the override, and holds with it", async () => {
+ resetCorpus();
+ seedVideo("local");
+ seedDriveChannel();
+ await runIndex();
+ const current = storedSchema();
+ const pagesBefore = sharedTranscripts();
+ const subsBefore = sharedSubs();
+
+ unmount();
+ setStoredSchema(current - 1);
+ await assert.rejects(runIndex(), (err: Error) => {
+ assert.match(
+ err.message,
+ new RegExp(
+ `must be rebuilt in full \\(index schema ${current - 1} -> ${current}\\), but 1 channel\\(s\\) cannot be read: ` +
+ `drive-channel \\(its media is not reachable \\(drive not mounted\\?\\), on location "USB drive"\\)`,
+ ),
+ );
+ // Why, the ways out (mounting first), and the override by name; no path.
+ assert.match(err.message, /the next site build would publish them as gone/);
+ assert.match(
+ err.message,
+ /For each: mount its media and run this again; or repair or re-point its location on \/storage; or finish or clear its move .*; or, if it is gone for good, delete the channel or set excludeFromBuild/,
+ );
+ // The override by name, and where it is set for each way a build runs.
+ assert.match(
+ err.message,
+ new RegExp(
+ `set ${INDEX_ALLOW_HELD_ENV}=1 in the environment of the process that runs the build: ` +
+ `for the command line, the command's own \\(\`${INDEX_ALLOW_HELD_ENV}=1 pnpm archilyzer index\`\\); ` +
+ `for the editor's Build index job, or a site build started from the editor, the editor's own environment, which takes a restart of the editor\\.$`,
+ ),
+ );
+ assert.ok(!err.message.includes(ROOT), err.message);
+ return true;
+ });
+ // Refused before the clear: nothing touched.
+ assert.equal(storedSchema(), current - 1);
+ assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]);
+ assert.deepEqual(sharedTranscripts(), pagesBefore);
+
+ // The CLI (the export's build:index, a site build's data phase) exits
+ // non-zero on it, and leaves the index as it was.
+ const cli = spawnSync(
+ path.join(COMMON, "node_modules", ".bin", "tsx"),
+ ["bin/archilyzer.ts", "index"],
+ { cwd: COMMON, env: { ...process.env }, encoding: "utf8" },
+ );
+ assert.notEqual(cli.status, 0, cli.stdout + cli.stderr);
+ assert.match(cli.stderr, /must be rebuilt in full/);
+ assert.equal(storedSchema(), current - 1);
+ assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]);
+
+ // Overridden: the rebuild runs, the held channel's records go with the
+ // clear (they cannot be re-read), and its pages are left as they are.
+ process.env[INDEX_ALLOW_HELD_ENV] = "1";
+ try {
+ const { res, log } = await runIndex();
+ assert.deepEqual(res.heldChannels, [DRIVE_CHANNEL]);
+ assert.equal(storedSchema(), current);
+ assert.deepEqual(indexed(), [`${CHANNEL}/local`]);
+ assert.deepEqual(sharedTranscripts(), pagesBefore);
+ assert.deepEqual(sharedSubs(), subsBefore);
+ assert.ok(
+ log.some(
+ (l) =>
+ l.startsWith(`Channel ${DRIVE_CHANNEL}: its media is not reachable`) &&
+ l.includes(`held under ${INDEX_ALLOW_HELD_ENV}: this full rebuild cleared its index records`),
+ ),
+ log.join("\n"),
+ );
+ } finally {
+ delete process.env[INDEX_ALLOW_HELD_ENV];
+ }
+
+ // The drive back: an ordinary build takes the channel in again.
+ remount();
+ const back = await runIndex();
+ assert.deepEqual(back.res.heldChannels, []);
+ assert.equal(back.res.added, 2);
+ assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]);
+ assert.equal(siteManifest().channels.find((c) => c.slug === DRIVE_CHANNEL)?.count, 2);
+});
+
+test("(d) a first build, with no index yet, is a full rebuild: it refuses with a channel held too", async () => {
+ resetCorpus();
+ seedVideo("local");
+ seedDriveChannel();
+ unmount();
+ await assert.rejects(runIndex(), /index schema <none> -> \d+\), but 1 channel\(s\) cannot be read: drive-channel/);
+ remount();
+ const { res } = await runIndex();
+ assert.deepEqual(res.heldChannels, []);
+ assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${CHANNEL}/local`]);
+});
+
+test("(e) the drive back: the held set is empty and what arrived meanwhile is indexed", async () => {
+ resetCorpus();
+ seedVideo("local");
+ seedDriveChannel();
+ await runIndex();
+ unmount();
+ const away = await runIndex();
+ assert.deepEqual(away.res.heldChannels, [DRIVE_CHANNEL]);
+ assert.equal(away.res.removed, 0);
+
+ // Downloaded onto the drive while it was elsewhere.
+ seedVideo("d3", DRIVE_CHANNEL, { dir: path.join(AWAY, DRIVE_CHANNEL, "data", "d3") });
+ remount();
+ const { res, log } = await runIndex();
+ assert.deepEqual(res.heldChannels, []);
+ assert.equal(res.added, 1);
+ assert.equal(res.changed, 0, "the kept records match the disk");
+ assert.equal(res.removed, 0);
+ assert.deepEqual(indexed(), [...DRIVE_VIDEOS, `${DRIVE_CHANNEL}/d3`, `${CHANNEL}/local`]);
+ assert.deepEqual(Object.keys(JSON.parse(sharedTranscripts()!["manifest.json"]).slugToPage).sort(), ["d1", "d2", "d3"]);
+ assert.ok(!log.some((l) => l.includes("Held")), log.join("\n"));
+});
+
+test("(f) a channel that is really empty is still emptied, not held", async () => {
+ resetCorpus(["emptied", "gone", CHANNEL]);
+ for (const slug of ["emptied", "gone"]) {
+ writeChannel(slug);
+ seedVideo("x1", slug);
+ seedVideo("x2", slug);
+ }
+ seedVideo("local");
+ await runIndex();
+ assert.equal(indexed().length, 5);
+
+ // Its media deleted: an empty data/ in place, and no data/ at all. Neither
+ // was relocated, so there is no drive to be missing.
+ for (const id of ["x1", "x2"]) rmSync(videoDir(id, "emptied"), { recursive: true });
+ rmSync(path.join(paths.channelsDir, "gone", "data"), { recursive: true });
+ const { res, log } = await runIndex();
+ assert.deepEqual(res.heldChannels, []);
+ assert.equal(res.removed, 4);
+ assert.deepEqual(indexed(), [`${CHANNEL}/local`]);
+ assert.deepEqual(JSON.parse(sharedTranscripts("emptied")!["manifest.json"]).slugToPage, {});
+ // The missing directory is said, not swallowed.
+ assert.ok(
+ log.includes("Channel gone: no data/ directory; indexed as a channel with no videos."),
+ log.join("\n"),
+ );
+});
+
+test("(g) a data directory that cannot be read holds its channel, and says why", async () => {
+ if (process.getuid?.() === 0) return; // root reads through a mode of 000
+ resetCorpus(["locked", "flaky", CHANNEL]);
+ for (const slug of ["locked", "flaky"]) {
+ writeChannel(slug);
+ seedVideo("x1", slug);
+ seedVideo("x2", slug);
+ }
+ await runIndex();
+ assert.equal(indexed().length, 4);
+
+ // The whole data/ unreadable; and one video dir unreadable mid-walk.
+ const locked = path.join(paths.channelsDir, "locked", "data");
+ const flaky = videoDir("x2", "flaky");
+ chmodSync(locked, 0o000);
+ chmodSync(flaky, 0o000);
+ try {
+ const { res, log } = await runIndex();
+ assert.deepEqual([...res.heldChannels].sort(), ["flaky", "locked"]);
+ assert.equal(res.removed, 0);
+ assert.equal(indexed().length, 4);
+ assert.ok(
+ log.some((l) => l.startsWith("Channel locked: its data directory could not be read (EACCES)")),
+ log.join("\n"),
+ );
+ assert.ok(
+ // One video's directory, said apart from the whole data/ above.
+ log.some((l) => l.startsWith("Channel flaky: a video in its data directory could not be read (EACCES)")),
+ log.join("\n"),
+ );
+ } finally {
+ chmodSync(locked, 0o755);
+ chmodSync(flaky, 0o755);
+ }
+ const { res } = await runIndex();
+ assert.deepEqual(res.heldChannels, []);
+ assert.equal(indexed().length, 4);
+});
+
+test("(h) a curated-tag change while a channel is held reaches its pages when the drive is back", async () => {
+ resetCorpus();
+ seedVideo("local");
+ seedDriveChannel({ d1: "Stream with Elfpire Eva" });
+ await runIndex();
+ assert.equal("curatedTags" in transcriptRecord("d1"), false);
+
+ unmount();
+ // A rule edit moves no mtime. It re-derives d1 in the index (no disk read),
+ // but d1's page is not rewritten while its channel is held.
+ writeGlobalTags(paths, {
+ version: 1,
+ tags: [
+ {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ rules: [{ id: "meta", kind: "metadata" as const, pattern: "elfpire", enabled: true }],
+ },
+ ],
+ assignments: {},
+ });
+ const pagesBefore = sharedTranscripts();
+ const { log } = await runIndex();
+ assert.deepEqual(sharedTranscripts(), pagesBefore);
+ assert.equal(pagesPending(), true, "the page debt is kept");
+ assert.ok(log.some((l) => l.startsWith("curated tags: the pages of 1 held channel(s) are not rewritten while held")), log.join("\n"));
+
+ remount();
+ const back = await runIndex();
+ assert.equal(back.res.added + back.res.changed + back.res.removed, 0, "no mtime moved");
+ assert.deepEqual(transcriptRecord("d1").curatedTags, ["eva-collab"]);
+ assert.equal(pagesPending(), false);
+});
+
+test("(i) the drive lost MID-WALK: the second look holds the channel instead of dropping the rest of it", async () => {
+ resetCorpus();
+ seedVideo("local");
+ seedDriveChannel();
+ for (const id of ["d3", "d4"]) seedVideo(id, DRIVE_CHANNEL);
+ const drive = [...DRIVE_VIDEOS, `${DRIVE_CHANNEL}/d3`, `${DRIVE_CHANNEL}/d4`];
+ await runIndex();
+ assert.deepEqual(indexed(), [...drive, `${CHANNEL}/local`]);
+ const pagesBefore = sharedTranscripts();
+ const subsBefore = sharedSubs();
+
+ // The first look before the walk finds the drive, and readdir lists all four
+ // videos. Then the drive goes, right after the walk's first metadata stat:
+ // every later stat in the channel is ENOENT, which on its own reads as "no
+ // metadata yet" and would drop the rest of the channel as gone.
+ let lost = false;
+ afterStat = (p) => {
+ if (lost || !p.endsWith(`${path.sep}metadata.info.json`)) return;
+ if (!p.includes(`${path.sep}${DRIVE_CHANNEL}${path.sep}data${path.sep}`)) return;
+ lost = true;
+ unmount();
+ };
+ seedVideo("local2"); // so the build is not a no-op
+ const { res, log } = await runIndex().finally(() => {
+ afterStat = null;
+ });
+ assert.ok(lost, "the drive went away inside the walk");
+ assert.deepEqual(res.heldChannels, [DRIVE_CHANNEL]);
+ assert.equal(res.removed, 0, log.join("\n"));
+ assert.equal(res.added, 1);
+ assert.deepEqual(indexed(), [...drive, `${CHANNEL}/local`, `${CHANNEL}/local2`]);
+ assert.deepEqual(sharedTranscripts(), pagesBefore);
+ assert.deepEqual(sharedSubs(), subsBefore);
+ assert.ok(
+ log.some((l) =>
+ l.startsWith(`Channel ${DRIVE_CHANNEL}: its media is not reachable (drive not mounted?), on location "USB drive"; its 4 indexed video(s) are kept`),
+ ),
+ log.join("\n"),
+ );
+});
+
+test("(z) no write this file caused landed outside its temp root", () => {
+ // LMDB writes natively, past the spy: its file must be under the root too.
+ assert.ok(paths.lmdbPath.startsWith(ROOT + path.sep), paths.lmdbPath);
+ assert.ok(statSync(ROOT).isDirectory());
+ const outside = writes.filter((p) => p !== ROOT && !p.startsWith(ROOT + path.sep));
+ assert.deepEqual(outside, []);
+ assert.ok(writes.length > 0, "the spy saw the writes");
+});
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -8,6 +8,22 @@
//
// Per-channel config.json selects the transcript parser ("youtube" → VTT,
// "transcribe" → whisper.cpp JSON). Short-circuits when mtimes already match.
+//
+// A CHANNEL WHOSE MEDIA IS NOT REACHABLE is HELD, not emptied: a relocated
+// `data/` on an unmounted drive, one mid-relocation, a link and a config that
+// disagree (inspectChannelMedia), or a data dir that fails to read. It is not
+// rescanned; its index records are kept as they are and its shared page trees
+// (transcripts, subs, digests) are left as they are, so the next site build
+// still publishes it. Until this hold the scan read such a channel as having
+// no videos, removed every record it had, and the site built next published
+// the channel as gone. The stats build has the same hold (buildStats.ts); the
+// words are shared (lib/channelMediaHold.ts).
+//
+// A FULL REBUILD (a schema change, or no index yet) with a channel held
+// REFUSES: it clears every channel's records, and a held channel cannot be
+// re-read, so it would come out empty. ARCHILYZER_INDEX_ALLOW_HELD=1 lets it
+// proceed; the held channel is then out of the index until its media is back
+// and the index is built again.
import path from "node:path";
import { createHash } from "node:crypto";
@@ -78,6 +94,13 @@ import {
type ChannelHandling,
} from "../lib/channelConfig";
import { readChannelConfigFile } from "./channels";
+import { inspectChannelMedia } from "../lib/channelMedia";
+import {
+ HELD_WAYS_OUT,
+ describeHeld,
+ heldReason,
+ isMediaHeld,
+} from "../lib/channelMediaHold";
import { resolveChannelGroupId } from "../lib/channelGroups";
import type { Paths } from "../lib/paths";
import {
@@ -275,21 +298,32 @@ async function exists(p: string): Promise<boolean> {
}
}
+function errCode(err: unknown): string {
+ const code = (err as NodeJS.ErrnoException | null)?.code;
+ return typeof code === "string" ? code : String(err);
+}
+
+// `held` maps each channel whose media could not be read to why, in words with
+// no path in them. A held channel contributes no live entries; the caller keeps
+// its records and pages (see the file header).
async function scanSource(
channelsDir: string,
log: (msg: string) => void,
): Promise<{
live: LiveEntry[];
channels: Map<string, ChannelConfig>;
+ held: Map<string, string>;
}> {
+ const locations = getSettings().storage.locations;
const channels = new Map<string, ChannelConfig>();
const live: LiveEntry[] = [];
+ const held = new Map<string, string>();
let channelEntries: Dirent[];
try {
channelEntries = await readdir(channelsDir, { withFileTypes: true });
} catch {
// Fresh transcripts dir with no channels yet.
- return { live, channels };
+ return { live, channels, held };
}
for (const ch of channelEntries) {
if (!ch.isDirectory()) continue;
@@ -308,13 +342,33 @@ async function scanSource(
// and routing it through the video scan would only ever produce noise. Its
// posts tree is built from the JSONL shards further down.
if (isSocialChannel(cfg)) continue;
+ // An unmounted drive is not an empty channel (lib/channelMedia.ts): the
+ // readdir below would fail, and every record the channel has would be
+ // removed as gone.
+ const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg);
+ if (isMediaHeld(media.status)) {
+ held.set(ch.name, heldReason(media, cfg.dataDir, locations));
+ continue;
+ }
const dataDir = path.join(channelDir, "data");
let videoEntries: Dirent[];
try {
videoEntries = await readdir(dataDir, { withFileTypes: true });
- } catch {
+ } catch (err) {
+ const code = errCode(err);
+ // No data/ at all on a channel whose media was never moved: it has
+ // downloaded nothing yet (or its media was deleted), and it IS empty.
+ if (code === "ENOENT" && media.status === "in-place") {
+ log(`Channel ${ch.name}: no data/ directory; indexed as a channel with no videos.`);
+ continue;
+ }
+ // Anything else — a relocated drive gone between the check and the
+ // read, a permission or I/O error — is a channel that could not be read.
+ held.set(ch.name, `its data directory could not be read (${code})`);
continue;
}
+ const channelLive: LiveEntry[] = [];
+ let readFailure: string | null = null;
for (const v of videoEntries) {
if (!v.isDirectory()) continue;
const videoDir = v.name;
@@ -323,7 +377,16 @@ async function scanSource(
let metaMs: number;
try {
metaMs = (await stat(metaPath)).mtimeMs;
- } catch {
+ } catch (err) {
+ // ENOENT is a video dir with no metadata yet (a download in flight, a
+ // partial one): skipped, as always. Any other error is the channel's
+ // media failing mid-scan; the channel is held below rather than read
+ // as missing this video and every one after it.
+ const code = errCode(err);
+ if (code !== "ENOENT" && code !== "ENOTDIR") {
+ readFailure = code;
+ break;
+ }
continue;
}
// Hybrid: a single channel may contain both YouTube auto-subs (.vtt)
@@ -373,7 +436,7 @@ async function scanSource(
// Sidecar absent — the common case (102 of ~76,000 videos have one).
}
}
- live.push({
+ channelLive.push({
channelSlug: ch.name,
handling: cfg.handling,
configName: cfg.name,
@@ -389,8 +452,24 @@ async function scanSource(
digestMs,
});
}
+ if (readFailure !== null) {
+ // One video, not the directory: said apart from the readdir failure
+ // above, so the log points at the right place. The whole channel is
+ // held all the same.
+ held.set(ch.name, `a video in its data directory could not be read (${readFailure})`);
+ continue;
+ }
+ // Asked again after the walk: a drive that went away DURING it leaves the
+ // videos after that point missing from this scan, which would remove them.
+ // Three syscalls a channel.
+ const after = await inspectChannelMedia({ channelsDir }, ch.name, cfg);
+ if (isMediaHeld(after.status)) {
+ held.set(ch.name, heldReason(after, cfg.dataDir, locations));
+ continue;
+ }
+ for (const e of channelLive) live.push(e);
}
- return { live, channels };
+ return { live, channels, held };
}
function pathKeyId(k: PathKey): string {
@@ -418,6 +497,9 @@ export type BuildIndexResult = {
changed: number;
removed: number;
shortCircuited: boolean;
+ // Channels whose media could not be read, so they were not rescanned: their
+ // index records and shared pages were kept as they were (see the header).
+ heldChannels: string[];
};
export type BuildIndexOptions = {
@@ -425,6 +507,17 @@ export type BuildIndexOptions = {
onLog?: (msg: string) => void;
};
+// Set to 1 (or true/yes/on) to let a FULL rebuild proceed with a channel held.
+// Declared in lib/envVars.ts.
+export const INDEX_ALLOW_HELD_ENV = "ARCHILYZER_INDEX_ALLOW_HELD";
+const TRUTHY = new Set(["1", "true", "yes", "on"]);
+function allowsHeldFullRebuild(
+ env: Record<string, string | undefined> = process.env,
+): boolean {
+ const raw = env.ARCHILYZER_INDEX_ALLOW_HELD;
+ return typeof raw === "string" && TRUTHY.has(raw.trim().toLowerCase());
+}
+
export async function buildIndex({
paths,
onLog,
@@ -535,6 +628,39 @@ export async function buildIndex({
const storedSchema = meta.get("schema") as number | undefined;
const schemaBumped = storedSchema !== SCHEMA_VERSION;
+
+ // Recorded as INDEX_SCANNED_AT_KEY only when this build completes, so
+ // buildStats can tell apart a video with no `mtimes` record: metadata newer
+ // than this is "not indexed yet"; older, and this build saw it and skipped it
+ // (no upload_date, or processing failed) or its channel was held.
+ //
+ // The scan reads only the source tree, so it runs BEFORE a schema clear: a
+ // full rebuild must know which channels it cannot read before it drops them.
+ const scanStartedAt = Date.now();
+ const {
+ live,
+ channels: channelConfigs,
+ held,
+ } = await scanSource(channelsDir, log);
+
+ // A full rebuild clears every channel's records, and a held channel cannot be
+ // re-read: it would come out of this build empty, and the site built next
+ // would publish it as gone. Refuse, unless told to go on without it.
+ const heldThroughClear = schemaBumped && held.size > 0;
+ if (heldThroughClear && !allowsHeldFullRebuild()) {
+ await root.close();
+ throw new Error(
+ `The index must be rebuilt in full (index schema ${storedSchema ?? "<none>"} -> ${SCHEMA_VERSION}), ` +
+ `but ${held.size} channel(s) cannot be read: ${describeHeld(held)}. ` +
+ `A full rebuild clears every channel's index records, so these would come out empty and the next site build would publish them as gone. ` +
+ `${HELD_WAYS_OUT} ` +
+ `To rebuild without them anyway (each stays out of the index until its media is back and the index is built again), set ${INDEX_ALLOW_HELD_ENV}=1 ` +
+ `in the environment of the process that runs the build: for the command line, the command's own ` +
+ `(\`${INDEX_ALLOW_HELD_ENV}=1 pnpm archilyzer index\`); for the editor's Build index job, or a site build started from the editor, ` +
+ `the editor's own environment, which takes a restart of the editor.`,
+ );
+ }
+
if (schemaBumped) {
log(
`Schema change (${storedSchema ?? "<none>"} -> ${SCHEMA_VERSION}); invalidating LMDB cache.`,
@@ -557,12 +683,6 @@ export async function buildIndex({
await meta.put("schema", SCHEMA_VERSION);
}
- // Recorded as INDEX_SCANNED_AT_KEY only when this build completes, so
- // buildStats can tell apart a video with no `mtimes` record: metadata newer
- // than this is "not indexed yet"; older, and this build saw it and skipped it
- // (no upload_date, or processing failed).
- const scanStartedAt = Date.now();
- const { live, channels: channelConfigs } = await scanSource(channelsDir, log);
const livePathIds = new Set<string>();
const liveByPathId = new Map<string, LiveEntry>();
for (const s of live) {
@@ -590,12 +710,29 @@ export async function buildIndex({
changed.push(s);
}
}
+ // A held channel has no live entries, and its records are not "gone": they
+ // are kept, and counted for the log.
+ const keptHeld = new Map<string, number>();
for (const { key, value } of mtimes.getRange()) {
const k = key as PathKey;
+ if (held.has(k[0])) {
+ keptHeld.set(k[0], (keptHeld.get(k[0]) ?? 0) + 1);
+ continue;
+ }
if (!livePathIds.has(pathKeyId(k))) {
removed.push({ pathKey: k, indexKey: (value as MtimeRecord).indexKey });
}
}
+ let keptHeldTotal = 0;
+ for (const [slug, why] of held) {
+ const kept = keptHeld.get(slug) ?? 0;
+ keptHeldTotal += kept;
+ log(
+ heldThroughClear
+ ? `Channel ${slug}: ${why}; held under ${INDEX_ALLOW_HELD_ENV}: this full rebuild cleared its index records, so it is out of the index until its media is back and the index is built again. Its transcript, subtitle and digest pages are left as they are.`
+ : `Channel ${slug}: ${why}; its ${kept} indexed video(s) are kept as they are, not rescanned, and its transcript, subtitle and digest pages are left as they are.`,
+ );
+ }
const anyMutations =
added.length > 0 || changed.length > 0 || removed.length > 0;
@@ -608,6 +745,8 @@ export async function buildIndex({
// from LMDB further below so they reflect current site config.
const sharedManifestsPresent = async (): Promise<boolean> => {
for (const channelSlug of channelConfigs.keys()) {
+ // A held channel's pages are not written this build either way.
+ if (held.has(channelSlug)) continue;
const mPath = path.join(transcriptsOutDir, channelSlug, "manifest.json");
const raw = await readFile(mPath, "utf8").catch(() => null);
if (!raw) return false;
@@ -638,7 +777,10 @@ export async function buildIndex({
const curatedFresh = new Set<string>();
log(
- `Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`,
+ `Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.` +
+ (held.size > 0
+ ? ` Held: ${held.size} channel(s), ${keptHeldTotal} video(s) kept.`
+ : ""),
);
for (const channelSlug of channelConfigs.keys()) {
@@ -1056,6 +1198,10 @@ export async function buildIndex({
if (sharedNeedsBuild) {
for (const channelSlug of Array.from(channelConfigs.keys()).sort()) {
+ // A held channel's pages are left exactly as the last build wrote them:
+ // not rewritten, not pruned, and (below) not removed. After a full rebuild
+ // its records are gone, and a rewrite would publish it empty.
+ if (held.has(channelSlug)) continue;
const channelDir = path.join(transcriptsOutDir, channelSlug);
await mkdir(channelDir, { recursive: true });
@@ -1172,6 +1318,10 @@ export async function buildIndex({
// availability.json for just those ids is cheap and exact.
let maybeMissingCount = 0;
for (const slug of channelConfigs.keys()) {
+ // A held channel's availability.json files are on the media that cannot be
+ // read, and a missing one reads as "maybe missing": its states are carried
+ // over from the last build instead (below).
+ if (held.has(slug)) continue;
const record = await loadMaybeMissing(paths, slug);
if (!record?.ids.length) continue;
const scannedAtMs = Date.parse(record.checkedAt);
@@ -1199,6 +1349,25 @@ export async function buildIndex({
);
}
+ // A held channel keeps the states the last build published for it (the
+ // confirmed ones above come from its kept records; this adds the overlay's),
+ // read back before the wholesale rewrite below. Nothing to carry after a full
+ // rebuild: the clear took them, with the records they described.
+ if (held.size > 0) {
+ for (const { key, value } of videoState.getRange()) {
+ const id = key as string;
+ const cut = id.indexOf("\x00");
+ if (cut < 0) continue;
+ const slug = id.slice(0, cut);
+ if (!held.has(slug)) continue;
+ const rec = mtimes.get([slug, id.slice(cut + 1)]);
+ if (!rec) continue;
+ const st = value as VideoState;
+ stateByIndexKey.set(indexKeyId(rec.indexKey), st);
+ stateByPath.set(id, st);
+ }
+ }
+
// Publish the sparse map for buildStats, which runs after us against the same
// LMDB file and would otherwise have to re-read ~76k availability.json files
// to build the status chart. Rewritten wholesale each build: the map is small
@@ -1219,6 +1388,16 @@ export async function buildIndex({
if (sharedNeedsBuild) {
for (const channelSlug of Array.from(channelConfigs.keys()).sort()) {
+ // Held: its subs dir is left as it is, and its stats carried over so the
+ // site manifests still list it (none to carry after a full rebuild).
+ if (held.has(channelSlug)) {
+ const prev = channelStatsDb.get(channelSlug);
+ if (prev) {
+ channelStats.set(channelSlug, prev);
+ subsTotalCount += prev.videoCount;
+ }
+ continue;
+ }
const cfg = channelConfigs.get(channelSlug)!;
const subsChannelDir = path.join(subsOutDir, channelSlug);
const tracksInChannel = new Set<string>();
@@ -1331,7 +1510,7 @@ export async function buildIndex({
const subsChannelSlugSet = new Set(channelStats.keys());
for (const e of topSubsEntries) {
if (e.isDirectory()) {
- if (!subsChannelSlugSet.has(e.name)) {
+ if (!subsChannelSlugSet.has(e.name) && !held.has(e.name)) {
await rm(path.join(subsOutDir, e.name), {
recursive: true,
force: true,
@@ -1365,7 +1544,17 @@ export async function buildIndex({
// curated-tag page debt is settled. Deliberately AFTER the page build and not
// beside the hashes: an interrupt anywhere above must leave the flag standing
// so the next build rewrites the shards.
- clearCuratedPagesPending(meta);
+ //
+ // Except for a held channel's pages, which were not written: the re-apply
+ // pass re-derives its records in LMDB like any other, and its pages owe them.
+ // The flag stays, so the first build with its media back rewrites them.
+ if (held.size > 0 && curatedReapply.pagesPending) {
+ log(
+ `curated tags: the pages of ${held.size} held channel(s) are not rewritten while held; the re-derived tags reach them on the first build with their media back.`,
+ );
+ } else {
+ clearCuratedPagesPending(meta);
+ }
await meta.flushed;
// ---------------------------------------------------------------------------
@@ -1578,6 +1767,16 @@ export async function buildIndex({
if (sharedNeedsBuild) {
for (const channelSlug of Array.from(channelConfigs.keys()).sort()) {
+ // Held: as with subs, its digest dir is left as it is and its stats
+ // carried over.
+ if (held.has(channelSlug)) {
+ const prev = channelDigestStatsDb.get(channelSlug);
+ if (prev) {
+ channelDigestStats.set(channelSlug, prev);
+ digestTotalCount += prev.digestCount;
+ }
+ continue;
+ }
const cfg = channelConfigs.get(channelSlug)!;
const digestChannelDir = path.join(digestsOutDir, channelSlug);
let digestCount = 0;
@@ -1677,7 +1876,7 @@ export async function buildIndex({
}).catch(() => [] as Dirent[]);
for (const e of topDigestEntries) {
if (e.isDirectory()) {
- if (!channelDigestStats.has(e.name)) {
+ if (!channelDigestStats.has(e.name) && !held.has(e.name)) {
await rm(path.join(digestsOutDir, e.name), {
recursive: true,
force: true,
@@ -2010,7 +2209,10 @@ export async function buildIndex({
const durationMs = Date.now() - t0;
log(
- `Done in ${(durationMs / 1000).toFixed(2)}s. ${sites.length} site(s): ${sitesBuilt} built, ${sitesSkipped} up to date; ${live.length} transcripts in pool.`,
+ `Done in ${(durationMs / 1000).toFixed(2)}s. ${sites.length} site(s): ${sitesBuilt} built, ${sitesSkipped} up to date; ${live.length} transcripts in pool.` +
+ (held.size > 0
+ ? ` Held, their media not readable: ${[...held.keys()].join(", ")}.`
+ : ""),
);
return {
totalCount: aggregateSummaries,
@@ -2024,5 +2226,6 @@ export async function buildIndex({
changed: changed.length,
removed: removed.length,
shortCircuited: !sharedNeedsBuild && sitesBuilt === 0,
+ heldChannels: [...held.keys()],
};
}
diff --git a/common/controller/buildStats.test.ts b/common/controller/buildStats.test.ts
@@ -65,6 +65,7 @@ const { buildIndex } = await import("./buildIndex");
const { buildStats, STATS_DOWNGRADE_ENV } = await import("./buildStats");
const { normalizeTranscript } = await import("./normalizeTranscript");
const { readStatsPages } = await import("./poolSummary");
+const { siteStatsDir } = await import("../lib/site");
const { STATS_SCHEMA_VERSION } = await import("../lib/stats");
const { open } = await import("lmdb");
@@ -594,6 +595,62 @@ test("(j) the schema guard: an older cache is cleared, a newer one is refused un
}
});
+// Release 14 slice HS: the whole-pool bundle is published as the homepage's
+// `stats/`, and an unlisted site's content is in no public total.
+test("(k) the whole-pool bundle leaves out a channel only an unlisted site exposes; the site's own bundle keeps it", async () => {
+ resetCorpus();
+ const seedChannel = (slug: string, ids: string[]) => {
+ writeJson(path.join(paths.channelsDir, slug, "config.json"), {
+ handling: "youtube",
+ name: slug,
+ url: `https://www.youtube.com/@${slug}/videos`,
+ });
+ for (const id of ids) {
+ seedVideo(id, "2026-07-11T11:00:00Z", {}, slug);
+ addCaptions(id, "2026-07-11T12:00:00Z", slug);
+ }
+ };
+ seedVideo("listed-1");
+ seedChannel("unlisted-channel", ["u1", "u2"]);
+ seedChannel("shared-channel", ["s1"]);
+ seedChannel("pool-channel", ["p1"]);
+ // The listed site also exposes the shared channel; the unlisted site exposes
+ // its own channel and the shared one. The pool channel is on no site.
+ const siteFile = path.join(paths.sitesDir, SITE, "site.json");
+ const listed = JSON.parse(readFileSync(siteFile, "utf8"));
+ listed.channels.push({ slug: "shared-channel", groupId: "default" });
+ writeJson(siteFile, listed);
+ writeJson(path.join(paths.sitesDir, "fixture-unlisted", "site.json"), {
+ ...listed,
+ siteId: "fixture-unlisted",
+ siteTitle: "Unlisted",
+ headerTitle: "Unlisted",
+ siteUrl: "https://unlisted.example",
+ listed: false,
+ channels: [
+ { slug: "unlisted-channel", groupId: "default" },
+ { slug: "shared-channel", groupId: "default" },
+ ],
+ });
+ await runIndex();
+ const log: string[] = [];
+ const { byId } = await runStats(log);
+ assert.deepEqual([...byId.keys()].sort(), ["listed-1", "p1", "s1"]);
+ const manifest = JSON.parse(readFileSync(path.join(POOL, "manifest.json"), "utf8"));
+ assert.equal(manifest.totalCount, 3);
+ assert.deepEqual(
+ manifest.channels.map((c: { slug: string }) => c.slug),
+ ["pool-channel", "shared-channel", CHANNEL],
+ );
+ assert.ok(
+ log.includes("Stats whole-pool: 3 videos, 1 page(s); 1 channel(s) only unlisted sites expose left out."),
+ log.join("\n"),
+ );
+ // The unlisted site still builds as before: its own bundle has its videos.
+ const own = await readStatsPages(siteStatsDir(paths, "fixture-unlisted"));
+ assert.deepEqual(own.map((s) => s.id).sort(), ["s1", "u1", "u2"]);
+});
+
test("(z) no write this file caused landed outside its temp root", () => {
// LMDB writes natively, past the spy: its file must be under the root too.
assert.ok(paths.lmdbPath.startsWith(ROOT + path.sep), paths.lmdbPath);
diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts
@@ -63,14 +63,20 @@ import {
import type { VideoStatus } from "../lib/stats";
import type { ChannelConfig } from "../lib/channelConfig";
import { readChannelConfigFile } from "./channels";
+import { inspectChannelMedia } from "../lib/channelMedia";
import {
- inspectChannelMedia,
- type ChannelMediaStatus,
-} from "../lib/channelMedia";
+ HELD_WAYS_OUT,
+ describeHeld,
+ heldReason,
+ isMediaHeld,
+} from "../lib/channelMediaHold";
import { getSettings } from "../lib/settings";
-import { locationLabelOfDataDir } from "../lib/storageLocations";
import type { Paths } from "../lib/paths";
-import { listSites, siteStatsDir } from "../lib/site";
+import {
+ channelsOnlyOnUnlistedSites,
+ listSites,
+ siteStatsDir,
+} from "../lib/site";
import {
INDEX_SCANNED_AT_KEY,
STATS_SCHEMA_VERSION,
@@ -169,11 +175,12 @@ export type BuildStatsOptions = {
paths: Paths;
onLog?: (msg: string) => void;
signal?: AbortSignal;
- // When set, also write an UNFILTERED whole-pool stats bundle (every non-
- // excluded channel) into this dir as {manifest,page-NNNN}.json. Used by the
- // Archilyzer hub (compose-homepage), whose cross-site charts need the full
- // dataset rather than any one site's filtered slice. Forces collection of the
- // full dataset even when no per-site bundle needs a rebuild.
+ // When set, also write a whole-pool stats bundle (every non-excluded
+ // channel, but for those only unlisted sites expose) into this dir as
+ // {manifest,page-NNNN}.json. Used by the Archilyzer hub (compose-homepage),
+ // whose cross-site charts need the full dataset rather than any one site's
+ // filtered slice. Forces collection of the full dataset even when no per-site
+ // bundle needs a rebuild.
wholePoolStatsDir?: string;
};
@@ -248,16 +255,6 @@ async function resolveAcquisitionDates(
return { downloadedDate, transcribedDate };
}
-// Why a channel is held, without the paths inspectChannelMedia's `detail`
-// carries (/storage shows those).
-const HELD_REASON: Record<ChannelMediaStatus, string> = {
- unreachable: "its media is not reachable (drive not mounted?)",
- "in-transition": "a move of its media is in progress or was interrupted",
- inconsistent: "its data link and its config disagree",
- ok: "reachable",
- "in-place": "reachable",
-};
-
// `held` maps each channel whose media is not reachable to why, in words with
// no path in them: it is not scanned, and the caller keeps its cached stats
// (see the file header).
@@ -292,12 +289,8 @@ async function scanSource(
// An unmounted drive is not an empty channel (lib/channelMedia.ts): the
// readdir below would fail and every one of its stats would be removed.
const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg);
- if (media.status !== "ok" && media.status !== "in-place") {
- const label = locationLabelOfDataDir(cfg.dataDir ?? media.target, locations);
- held.set(
- ch.name,
- `${HELD_REASON[media.status]}${label ? `, on location "${label}"` : ""}`,
- );
+ if (isMediaHeld(media.status)) {
+ held.set(ch.name, heldReason(media, cfg.dataDir, locations));
continue;
}
const dataDir = path.join(channelDir, "data");
@@ -444,11 +437,7 @@ export async function buildStats({
await root.close();
throw new Error(
`The stats cache must be rebuilt (stats schema ${storedSchema ?? "<none>"} -> ${STATS_SCHEMA_VERSION}), ` +
- `but ${held.size} channel(s) cannot be read: ` +
- [...held].map(([slug, why]) => `${slug} (${why})`).join("; ") +
- `. For each: mount its media and run this again; or repair or re-point its location on /storage; ` +
- `or finish or clear its move (the channel's Storage panel); or, if it is gone for good, ` +
- `delete the channel or set excludeFromBuild in its config.`,
+ `but ${held.size} channel(s) cannot be read: ${describeHeld(held)}. ${HELD_WAYS_OUT}`,
);
}
log(
@@ -723,16 +712,25 @@ export async function buildStats({
log(`Stats site ${plan.site.siteId}: ${filtered.length} videos, ${pageCount} page(s).`);
}
- // Whole-pool bundle for the hub: every non-excluded channel, unfiltered. Built
+ // Whole-pool bundle for the hub: every non-excluded channel, except a channel
+ // only unlisted sites expose (site.json `listed: false`): the bundle is
+ // published as the homepage's `stats/`, and an unlisted site's content is in
+ // no public total. Its own per-site bundle above is built as before. Built
// from the same in-memory dataset so it stays consistent with the per-site
// bundles. Always rewritten when requested (stats records are small).
if (wholePoolStatsDir) {
+ const unlistedOnly = channelsOnlyOnUnlistedSites(sites);
+ const pooled =
+ unlistedOnly.size > 0
+ ? all.filter((s) => !unlistedOnly.has(s.channelSlug))
+ : all;
const { pageCount } = await writePages(
wholePoolStatsDir,
- all,
+ pooled,
STATS_MAX_PAGE_BYTES,
);
const channelEntries: StatsChannelEntry[] = [...channels.keys()]
+ .filter((slug) => !unlistedOnly.has(slug))
.map((slug) => ({
slug,
name: channels.get(slug)?.name ?? slug,
@@ -742,13 +740,18 @@ export async function buildStats({
const manifest: StatsManifest = {
version: STATS_MANIFEST_VERSION,
generatedAt: new Date().toISOString(),
- totalCount: all.length,
+ totalCount: pooled.length,
pageCount,
maxPageBytes: STATS_MAX_PAGE_BYTES,
channels: channelEntries,
};
await writeJsonAtomic(path.join(wholePoolStatsDir, "manifest.json"), manifest);
- log(`Stats whole-pool: ${all.length} videos, ${pageCount} page(s).`);
+ log(
+ `Stats whole-pool: ${pooled.length} videos, ${pageCount} page(s)` +
+ (unlistedOnly.size > 0
+ ? `; ${unlistedOnly.size} channel(s) only unlisted sites expose left out.`
+ : "."),
+ );
}
// Prune fingerprints for sites that no longer exist (their staging dirs are
diff --git a/common/controller/poolSummary.test.ts b/common/controller/poolSummary.test.ts
@@ -0,0 +1,28 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { parseSite } from "../lib/siteSchema";
+import { channelSitesOf } from "./poolSummary";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// channelSitesOf is the published `channel-sites.json` (compose-homepage) and
+// the map the summary attributes channels by. Release 14 slice HS: an unlisted
+// site (site.json `listed: false`) is in neither.
+
+test("channel-sites.json names listed sites only; a channel only an unlisted site exposes is absent", () => {
+ const sites = [
+ parseSite("fixture-a", { channels: [{ slug: "a1" }, { slug: "shared" }] }),
+ parseSite("fixture-b", { channels: [{ slug: "b1" }, { slug: "shared" }] }),
+ parseSite("fixture-unlisted", {
+ listed: false,
+ channels: [{ slug: "shared" }, { slug: "own" }],
+ }),
+ ];
+ assert.deepEqual(channelSitesOf(sites), {
+ a1: ["fixture-a"],
+ shared: ["fixture-a", "fixture-b"],
+ b1: ["fixture-b"],
+ });
+ assert.ok(!JSON.stringify(channelSitesOf(sites)).includes("fixture-unlisted"));
+});
diff --git a/common/controller/poolSummary.ts b/common/controller/poolSummary.ts
@@ -11,7 +11,7 @@ import path from "node:path";
import { mkdir, readFile } from "node:fs/promises";
import type { Paths } from "../lib/paths";
import { buildStats } from "./buildStats";
-import { listSites, type Site } from "../lib/site";
+import { isListedSite, listSites, type Site } from "../lib/site";
import {
statsPageFileName,
type StatsManifest,
@@ -45,11 +45,14 @@ export async function readStatsPages(statsDir: string): Promise<VideoStat[]> {
return out;
}
-// channel slug -> the ids of the content sites that expose it. A channel on
-// multiple sites maps to all of them; a pool-only channel is simply absent.
-export function channelSitesOf(sites: Site[]): ChannelSitesMap {
+// channel slug -> the ids of the LISTED content sites that expose it: the
+// published `channel-sites.json`. A channel on multiple sites maps to all of
+// them; a pool-only channel is simply absent, and so is an unlisted site
+// (site.json `listed: false`) and a channel only unlisted sites expose.
+export function channelSitesOf(sites: readonly Site[]): ChannelSitesMap {
const channelSites: ChannelSitesMap = {};
for (const site of sites) {
+ if (!isListedSite(site)) continue;
for (const c of site.channels) {
(channelSites[c.slug] ??= []).push(site.siteId);
}
@@ -71,8 +74,9 @@ export async function buildPoolSummary(opts: {
}): Promise<PoolSummary> {
const { paths, statsDir } = opts;
await mkdir(statsDir, { recursive: true });
- // Whole-pool stats dataset (every non-excluded channel). buildStats also
- // refreshes the per-site bundles as a side effect, which is harmless.
+ // Whole-pool stats dataset (every non-excluded channel but those only
+ // unlisted sites expose). buildStats also refreshes the per-site bundles as a
+ // side effect, which is harmless.
await buildStats({ paths, wholePoolStatsDir: statsDir });
const sites = listSites(paths);
const channelSites = channelSitesOf(sites);
diff --git a/common/lib/channelMediaHold.ts b/common/lib/channelMediaHold.ts
@@ -0,0 +1,61 @@
+import type {
+ ChannelMediaLocation,
+ ChannelMediaStatus,
+} from "./channelMedia";
+import {
+ locationLabelOfDataDir,
+ type StorageLocation,
+} from "./storageLocations";
+
+// THE HOLD, in the words both pool-wide builds use.
+//
+// The index build (controller/buildIndex.ts) and the stats build
+// (controller/buildStats.ts) each walk every channel's `data/`. A channel whose
+// media cannot be read — a relocated `data/` on an unmounted drive, a move in
+// progress, a link and a config that disagree — is HELD by both: not rescanned,
+// and what the last build knew of it kept, rather than read as a channel with
+// no videos and removed (lib/channelMedia.ts says why that reading is the
+// dangerous one). This module is only the shared vocabulary: which statuses
+// hold, why, in words with no path in them (/storage shows the paths), and the
+// ways out a refusal names. Each build decides for itself what "kept" means.
+//
+// Pure: no I/O. The caller asks inspectChannelMedia and passes the answer in.
+
+// "ok" and "in-place" are read; every other status holds, including any a later
+// inspectChannelMedia adds.
+export function isMediaHeld(status: ChannelMediaStatus): boolean {
+ return status !== "ok" && status !== "in-place";
+}
+
+// Why a channel is held, without the paths inspectChannelMedia's `detail`
+// carries.
+export const HELD_REASON: Record<ChannelMediaStatus, string> = {
+ unreachable: "its media is not reachable (drive not mounted?)",
+ "in-transition": "a move of its media is in progress or was interrupted",
+ inconsistent: "its data link and its config disagree",
+ ok: "reachable",
+ "in-place": "reachable",
+};
+
+// The reason, and the storage location's label when the channel's media is on
+// one. `dataDir` is the channel config's; the inspector's target stands in when
+// the config names none (a move in flight).
+export function heldReason(
+ media: Pick<ChannelMediaLocation, "status" | "target">,
+ dataDir: string | undefined,
+ locations: StorageLocation[],
+): string {
+ const label = locationLabelOfDataDir(dataDir ?? media.target, locations);
+ return `${HELD_REASON[media.status]}${label ? `, on location "${label}"` : ""}`;
+}
+
+// A held channel named in a refusal, as `slug (why)`, joined.
+export function describeHeld(held: Map<string, string>): string {
+ return [...held].map(([slug, why]) => `${slug} (${why})`).join("; ");
+}
+
+// What a refusal tells the operator to do, mounting first.
+export const HELD_WAYS_OUT =
+ `For each: mount its media and run this again; or repair or re-point its location on /storage; ` +
+ `or finish or clear its move (the channel's Storage panel); or, if it is gone for good, ` +
+ `delete the channel or set excludeFromBuild in its config.`;
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -112,6 +112,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "AUDIO_CHECK_RESUME_DURING_PROBE", audience: "runtime", default: "the channel's `audioCheck.resumeDuringProbe`", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "`1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting." },
{ name: "AUDIO_CHECK_BACKOFF_FACTOR", audience: "runtime", default: "the built-in factor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "The audio check's interval backoff factor, in (0, 1], for a one-off run." },
{ name: "ARCHILYZER_STATS_ALLOW_DOWNGRADE", audience: "runtime", default: "off", readBy: "common/controller/buildStats.ts", doc: "`1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions." },
+ { name: "ARCHILYZER_INDEX_ALLOW_HELD", audience: "runtime", default: "off", readBy: "common/controller/buildIndex.ts", doc: "`1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel." },
{ name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." },
{ name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." },
{ name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },
diff --git a/common/lib/homepageSummary.test.ts b/common/lib/homepageSummary.test.ts
@@ -277,3 +277,70 @@ test("a site's wordmark lead travels when it is a proper prefix of the title; ot
assert.ok(plain.sites.every((x) => !("wordmarkLead" in x)));
assert.equal(s.version, HOMEPAGE_SUMMARY_VERSION);
});
+
+// Release 14 slice HS: site.json `listed: false`. The site still builds and
+// deploys; the family's public pages do not list it, and no public total counts
+// the channels only it exposes.
+test("an unlisted site is in no array and no total; a channel it shares is the listed site's", () => {
+ const unlisted = { ...site("zeta", ["q1", "a1"], "https://zeta.example"), listed: false } as Site;
+ const sites = [...SITES, unlisted];
+ const channelSites = { ...CHANNEL_SITES, a1: ["alpha", "zeta"], q1: ["zeta"] };
+ const own = [
+ stat({ channelSlug: "q1", id: "q-1", uploadDate: "20251101", duration: 7200 }),
+ stat({ channelSlug: "q1", id: "q-2", uploadDate: "20260201", status: "deleted" }),
+ stat({ channelSlug: "q1", id: "q-3", hasTranscript: false, transcribedDate: null }),
+ ];
+ const s = buildHomepageSummary([...STATS, ...own], channelSites, sites, NOW);
+ const base = buildHomepageSummary(STATS, CHANNEL_SITES, SITES, NOW);
+ // Byte for byte the summary without the unlisted site: every array (sites,
+ // channels, series, recent, monthly) and every total (totals, official,
+ // availability, monthlyUnplaced).
+ assert.deepEqual(s, base);
+ const text = JSON.stringify(s);
+ for (const needle of ["zeta", "ZETA", "q1", "Q1", "q-1"]) {
+ assert.ok(!text.includes(needle), needle);
+ }
+ // The shared channel stays credited to alpha.
+ assert.equal(s.sites.find((x) => x.siteId === "alpha")!.channels, 2);
+ assert.equal(s.version, 6);
+});
+
+test("an unlisted site with no siteUrl, or with every channel shared, changes nothing either", () => {
+ const base = buildHomepageSummary(STATS, CHANNEL_SITES, SITES, NOW);
+ // No siteUrl: never public, and its own channel is still in no total (unlike
+ // a pool-only channel, which `totals` counts).
+ const urlless = { ...site("zeta", ["r1"]), listed: false } as Site;
+ assert.deepEqual(
+ buildHomepageSummary(
+ [...STATS, stat({ channelSlug: "r1", id: "r-1" })],
+ { ...CHANNEL_SITES, r1: ["zeta"] },
+ [...SITES, urlless],
+ NOW,
+ ),
+ base,
+ );
+ const sharedOnly = { ...site("zeta", ["a1", "b1"], "https://zeta.example"), listed: false } as Site;
+ assert.deepEqual(
+ buildHomepageSummary(STATS, { ...CHANNEL_SITES, a1: ["alpha", "zeta"], b1: ["beta", "zeta"] }, [...SITES, sharedOnly], NOW),
+ base,
+ );
+ // Listed explicitly is the default: the same summary as no key at all.
+ const listedTrue = SITES.map((x) => ({ ...x, listed: true })) as Site[];
+ assert.deepEqual(buildHomepageSummary(STATS, CHANNEL_SITES, listedTrue, NOW), base);
+});
+
+test("a shared channel is the listed site's even when the unlisted site's id sorts first", () => {
+ const base = buildHomepageSummary(STATS, CHANNEL_SITES, SITES, NOW);
+ // "aaa-hidden" < "beta": the primary pick sorts ids, so only the listed
+ // filter keeps b1 with beta.
+ const unlisted = { ...site("aaa-hidden", ["b1", "q1"], "https://aaa-hidden.example"), listed: false } as Site;
+ const s = buildHomepageSummary(
+ [...STATS, stat({ channelSlug: "q1", id: "q-1", uploadDate: "20251101" })],
+ { ...CHANNEL_SITES, b1: ["aaa-hidden", "beta"], q1: ["aaa-hidden"] },
+ [unlisted, ...SITES],
+ NOW,
+ );
+ assert.deepEqual(s, base);
+ assert.ok(s.recent.filter((r) => r.slug.startsWith("b1/")).every((r) => r.siteId === "beta"));
+ assert.ok(!JSON.stringify(s).includes("aaa-hidden"));
+});
diff --git a/common/lib/homepageSummary.ts b/common/lib/homepageSummary.ts
@@ -1,6 +1,7 @@
import type { Platform } from "./platform";
import type { VideoStat } from "./stats";
import type { Site } from "./site";
+import { channelsOnlyOnUnlistedSites, isListedSite } from "./siteSchema";
import { accentHex, accentIdOf } from "./accent";
import { wordmarkLeadFor, type AccentId } from "./brand";
import { VIDEO_STATES, type VideoState } from "./availability";
@@ -10,11 +11,15 @@ import { VIDEO_STATES, type VideoState } from "./availability";
// HTML, so the landing renders instantly without the browser fetching the
// multi-MB whole-pool stats dataset.
//
-// Scope: the chart "universe" is PUBLIC sites only (those with a siteUrl), and
-// every video is attributed to a single PRIMARY public site (the first, by
-// sorted id, exposing its channel) so the Site and Channel breakdowns partition
-// the same set and combined totals stay honest. The KPI `totals` are instance-
-// wide (count pool-only / URL-less content too) — a deliberate scope difference.
+// Scope: the chart "universe" is PUBLIC sites only (those with a siteUrl that
+// are listed — site.json `listed`, lib/siteSchema.ts isListedSite), and every
+// video is attributed to a single PRIMARY public site (the first, by sorted id,
+// exposing its channel) so the Site and Channel breakdowns partition the same
+// set and combined totals stay honest. The KPI `totals` (and `availability`)
+// are instance-wide (count pool-only / URL-less content too) — a deliberate
+// scope difference — EXCEPT a channel only unlisted sites expose, which no
+// part of the summary counts (channelsOnlyOnUnlistedSites). An unlisted site is
+// in no array here; a channel it shares with a listed site is the listed one's.
//
// Two metrics are pre-binned at two granularities; Cumulative and Share (100%)
// are derived client-side from these, so no extra precompute is needed.
@@ -33,7 +38,12 @@ import { VIDEO_STATES, type VideoState } from "./availability";
// the same way — a summary without it paints its sites' hex, as before. So is
// the per-site `wordmarkLead` (release 14): a summary without it shows each
// card's title plain, as before. Nothing reads this number to accept a file.
-export const HOMEPAGE_SUMMARY_VERSION = 5;
+//
+// v6 (release 14): an unlisted site (site.json `listed: false`) is in no array,
+// and a channel only unlisted sites expose is in no total — `totals` and
+// `availability` included. No field was added or removed; the number says the
+// totals' scope moved.
+export const HOMEPAGE_SUMMARY_VERSION = 6;
// Day buckets are capped to this many trailing days so the embedded summary stays
// small regardless of archive age (daily detail is only useful recently).
@@ -166,7 +176,8 @@ export type HomepageOfficialTotals = {
export type HomepageSummary = {
version: number;
generatedAt: string; // ISO timestamp
- // Instance-wide headline numbers (count everything, not just public sites).
+ // Instance-wide headline numbers (count everything, not just public sites —
+ // but never a channel only unlisted sites expose).
totals: {
transcripts: number;
downloads: number;
@@ -351,12 +362,21 @@ export function buildHomepageSummary(
const nowWeek = weekOf(nowDay.replace(/-/g, ""))!;
const dayFloor = isoDate(new Date(now.getTime() - (DAY_WINDOW - 1) * 86400000));
- // Public sites only (need a link target + a stable place on the chart).
- const publicSites = sites.filter((s): s is Site & { siteUrl: string } => !!s.siteUrl);
+ // Public sites only (need a link target + a stable place on the chart), and
+ // only the listed ones.
+ const publicSites = sites.filter(
+ (s): s is Site & { siteUrl: string } => !!s.siteUrl && isListedSite(s),
+ );
const siteById = new Map(publicSites.map((s) => [s.siteId, s]));
+ // An unlisted site's own channels: counted nowhere below, totals included.
+ const unlistedOnly = channelsOnlyOnUnlistedSites(sites);
+ const inScope = unlistedOnly.size > 0
+ ? stats.filter((s) => !unlistedOnly.has(s.channelSlug))
+ : stats;
// channel slug -> primary public site (first by sorted id). Channels with no
- // public site are out of the chart universe entirely.
+ // public site are out of the chart universe entirely. Over listed sites only,
+ // so a channel an unlisted site shares with a listed one is the listed one's.
const primarySiteOf = new Map<string, string>();
for (const slug of Object.keys(channelSites)) {
const primary = [...channelSites[slug]]
@@ -369,7 +389,7 @@ export function buildHomepageSummary(
const channelName = new Map<string, string>();
const transcribedItems: Attributed[] = [];
const downloadedItems: Attributed[] = [];
- // Instance-wide KPI accumulators (count everything, not just public).
+ // Instance-wide KPI accumulators (count everything in scope, not just public).
let transcripts = 0;
let downloads = 0;
let hoursSeconds = 0;
@@ -405,7 +425,7 @@ export function buildHomepageSummary(
const uploadItems: { month: string; siteId: string }[] = [];
let monthlyUnplaced = 0;
- for (const s of stats) {
+ for (const s of inScope) {
// A transcript COUNTS whether or not it carries a date: transcripts,
// channels, hours, upload-month placement. Only the transcribed time series
// (and "this month", and the recent rail) need `transcribedDate`. buildStats
@@ -496,7 +516,7 @@ export function buildHomepageSummary(
.sort((a, b) => a.name.localeCompare(b.name));
// Recent feed (public universe), attributed to the primary site for its link.
- const recent: HomepageRecentItem[] = stats
+ const recent: HomepageRecentItem[] = inScope
.filter((s): s is VideoStat & { transcribedDate: string } =>
Boolean(s.hasTranscript && s.transcribedDate && primarySiteOf.get(s.channelSlug)),
)
@@ -522,16 +542,17 @@ export function buildHomepageSummary(
};
});
- // State census over every record, matching `totals`' instance-wide scope
- // (pool-only channels included) rather than the charts' public-site universe.
+ // State census over every in-scope record, matching `totals`' instance-wide
+ // scope (pool-only channels included, a channel only unlisted sites expose
+ // not) rather than the charts' public-site universe.
// Seeded with every state at zero so the shape is stable across corpora.
const byState = Object.fromEntries(
VIDEO_STATES.map((s) => [s, 0]),
) as Record<VideoState, number>;
- for (const s of stats) byState[s.status] = (byState[s.status] ?? 0) + 1;
+ for (const s of inScope) byState[s.status] = (byState[s.status] ?? 0) + 1;
const availability: HomepageAvailability = {
byState,
- counted: stats.length,
+ counted: inScope.length,
};
// Monthly back-catalogue series, keyed by the sites that survived the
diff --git a/common/lib/site.ts b/common/lib/site.ts
@@ -13,6 +13,7 @@ import {
import { socialLinksForSave } from "./socialLinks";
import { readJsonFileSync, writeJsonAtomic } from "./jsonFile-server";
import {
+ isListedSite,
isValidSiteId,
parseSite,
parseSiteUrl,
@@ -138,7 +139,9 @@ export type CrossSiteLink = { siteId: string; title: string; url: string };
export type CrossSiteGroup = { label?: string; sites: CrossSiteLink[] };
// Resolve the footer's cross-site list for `current` against the full pool.
-// Siblings that lack a siteUrl (or are `current`) are not linkable and dropped.
+// Siblings that lack a siteUrl (or are `current`) are not linkable and dropped,
+// and so is an unlisted sibling (`listed: false`, isListedSite) — even one a
+// featured group names. An unlisted `current` still lists its siblings.
// `current.relatedSites` groups render first, in order, each filtered to known
// linkable ids (unknown/used/self skipped, empty groups dropped). Every still-
// unused sibling lands in a trailing remainder group — unlabeled when there
@@ -149,7 +152,7 @@ export function resolveRelatedSites(
): CrossSiteGroup[] {
const byId = new Map<string, CrossSiteLink>();
for (const s of all) {
- if (s.siteId === current.siteId || !s.siteUrl) continue;
+ if (s.siteId === current.siteId || !s.siteUrl || !isListedSite(s)) continue;
byId.set(s.siteId, { siteId: s.siteId, title: s.siteTitle, url: s.siteUrl });
}
const used = new Set<string>();
diff --git a/common/lib/siteSchema.test.ts b/common/lib/siteSchema.test.ts
@@ -9,12 +9,14 @@ import type { z } from "zod";
import {
SITE_FIELD_DOCS,
SITE_KEYS,
+ channelsOnlyOnUnlistedSites,
+ isListedSite,
parseSite,
siteFieldsSchema,
siteToDisk,
type Site,
} from "./siteSchema";
-import { getSite, siteConfigFile, writeSite } from "./site";
+import { getSite, resolveRelatedSites, siteConfigFile, writeSite } from "./site";
import type { Paths } from "./paths";
const HERE = path.dirname(fileURLToPath(import.meta.url));
@@ -60,6 +62,7 @@ test("empty, null, [] and a number all read as the defaults, every key emitted",
assert.equal(want.duplicates, true);
assert.equal(want.transcriptDownloads, true);
assert.equal(want.pwa, false);
+ assert.equal(want.listed, true);
assert.deepEqual(want.relatedSites, []);
assert.equal(want.socialLinks, undefined);
assert.ok("socialLinks" in want);
@@ -231,6 +234,7 @@ function fixtures(): Array<[string, unknown]> {
channels: [{ slug: "c1", groupId: "a", order: 3 }, { slug: "c2", groupId: "q" }],
cloudflareProject: "p",
siteUrl: "https://s.example//",
+ listed: false,
relatedSites: [{ siteIds: ["x", "x", "BAD"] }, { label: " ", siteIds: [] }],
pwa: true,
archives: false,
@@ -285,6 +289,59 @@ test("writeSite throws on no groups, a default outside the groups, and an unsafe
assert.equal(fs.existsSync(siteConfigFile(paths, "s")), false);
});
+test("listed: absent reads listed, only an explicit false unlists, and only false is written", async () => {
+ for (const v of [undefined, true, 0, "false", null]) {
+ assert.equal(parseSite("s", { listed: v }).listed, true, String(v));
+ assert.equal("listed" in siteToDisk(parseSite("s", { listed: v })), false, String(v));
+ }
+ assert.equal(parseSite("s", { listed: false }).listed, false);
+ // `true` in a caller's Site is the default, so it is not written either.
+ assert.equal("listed" in siteToDisk({ ...parseSite("s", {}), listed: true }), false);
+
+ // Through the real writer and reader: false survives, absent reads listed.
+ const paths = scratchPaths(await mkdtemp(path.join(os.tmpdir(), "site-")));
+ const hidden = parseSite("s", { siteTitle: "Hidden", siteUrl: "https://h.example", listed: false });
+ await writeSite(hidden, paths);
+ assert.equal(JSON.parse(await readFile(siteConfigFile(paths, "s"), "utf8")).listed, false);
+ assert.deepEqual(getSite("s", paths), hidden);
+ await writeSite({ ...hidden, listed: true }, paths);
+ const disk = JSON.parse(await readFile(siteConfigFile(paths, "s"), "utf8"));
+ assert.equal("listed" in disk, false);
+ assert.equal(getSite("s", paths).listed, true);
+});
+
+test("isListedSite is the key's default; channelsOnlyOnUnlistedSites keeps a shared channel with the listed site", () => {
+ assert.equal(isListedSite({}), true);
+ assert.equal(isListedSite({ listed: true }), true);
+ assert.equal(isListedSite({ listed: false }), false);
+ const sites = [
+ parseSite("shown", { channels: [{ slug: "shared" }, { slug: "mine" }] }),
+ parseSite("hidden", { listed: false, channels: [{ slug: "shared" }, { slug: "secret" }] }),
+ parseSite("hidden2", { listed: false, channels: [{ slug: "secret" }, { slug: "secret2" }] }),
+ ];
+ assert.deepEqual([...channelsOnlyOnUnlistedSites(sites)].sort(), ["secret", "secret2"]);
+ assert.deepEqual([...channelsOnlyOnUnlistedSites([sites[0]])], []);
+});
+
+test("the footer never links an unlisted sibling, and an unlisted site's own footer still lists the rest", () => {
+ const current = parseSite("cur", {
+ siteUrl: "https://cur.example",
+ relatedSites: [{ label: "Friends", siteIds: ["hidden", "shown"] }],
+ });
+ const shown = parseSite("shown", { siteTitle: "Shown", siteUrl: "https://shown.example" });
+ const hidden = parseSite("hidden", { siteTitle: "Hidden", siteUrl: "https://hidden.example", listed: false });
+ const other = parseSite("other", { siteTitle: "Other", siteUrl: "https://other.example" });
+ const ids = (groups: ReturnType<typeof resolveRelatedSites>) =>
+ groups.flatMap((g) => g.sites.map((s) => s.siteId));
+ // Named in a featured group or not, the unlisted site is not linked.
+ assert.deepEqual(ids(resolveRelatedSites(current, [current, shown, hidden, other])), ["shown", "other"]);
+ // The unlisted site is still built as before: its footer lists its siblings.
+ assert.deepEqual(
+ ids(resolveRelatedSites({ ...hidden, relatedSites: [] }, [current, shown, hidden, other])),
+ ["cur", "shown", "other"],
+ );
+});
+
test("writeSite → getSite round-trips, and the file holds only non-defaults", async () => {
const paths = scratchPaths(await mkdtemp(path.join(os.tmpdir(), "site-")));
const site = parseSite("s", { siteTitle: "Mine", archives: false });
@@ -295,4 +352,5 @@ test("writeSite → getSite round-trips, and the file holds only non-defaults",
assert.equal("duplicates" in disk, false);
assert.equal("transcriptDownloads" in disk, false);
assert.equal("pwa" in disk, false);
+ assert.equal("listed" in disk, false);
});
diff --git a/common/lib/siteSchema.ts b/common/lib/siteSchema.ts
@@ -84,6 +84,7 @@ export type Site = {
cloudflareProject?: string;
accent?: string;
siteUrl?: string;
+ listed?: boolean;
relatedSites?: RelatedSiteGroup[];
pwa?: boolean;
archives?: boolean;
@@ -116,6 +117,8 @@ export const SITE_FIELD_DOCS: FieldDocs<Site> = {
'Per-site brand accent: a named accent id (`signal`, `brass`, `vermilion`, `violet`, `sakura`, `blue`, `green`) or a custom `"#rrggbb"`. It is the site\'s accent on every page; a reader does not pick one. Absent = `signal`, the family default. A custom hex is darkened or lightened per base until it reaches 4.5:1. The public `/site.json` always carries a hex: an id is published as its on-dark value. Any other spelling is dropped.',
siteUrl:
"Absolute public URL of this site's deployment, e.g. `https://jeralyzer.pages.dev` (trimmed, trailing slashes removed; anything not absolute http(s) is dropped). Drives the cross-site footer: a site with no siteUrl is omitted from every other site's list.",
+ listed:
+ "Whether the family lists this site. Opt-OUT: absent/true = listed, only an explicit `false` is written. An unlisted site still builds and deploys as before, and its own pages are unchanged; it is left out of the homepage (cards, chart, `/stats`), the hub (members, federated search, `/corpus.json`, `/llms.txt`), every other site's footer, and the published `channel-sites.json` and pooled `stats/`. A channel only unlisted sites expose is in none of the family's public totals; a channel a listed site also exposes is credited to the listed one.",
relatedSites:
"Pulls specific siblings to the front of the footer's cross-site list, in named groups. Siblings not named here fall into a trailing \"Other sites\" group. Absent/empty = one flat list of every sibling.",
pwa:
@@ -139,6 +142,36 @@ export function isValidSiteId(id: unknown): id is string {
return typeof id === "string" && SITE_ID_RE.test(id);
}
+// THE ONE PREDICATE for `listed` (site.json's opt-out; absent = listed). Every
+// public output that enumerates the family's sites filters through it: the
+// homepage summary (lib/homepageSummary.ts), channel-sites.json and the pooled
+// stats (controller/poolSummary.ts, controller/buildStats.ts), the hub's
+// member list (bin/compose-hub.ts) and the footer's siblings
+// (lib/site.ts resolveRelatedSites). The editor's own pages list every site.
+// Here, beside the key, and exported from lib/site like isValidSiteId, so the
+// pure summary builder can use it without importing file I/O.
+export function isListedSite(site: Pick<Site, "listed">): boolean {
+ return site.listed !== false;
+}
+
+// The channels whose content belongs to unlisted sites alone: exposed by at
+// least one site, and by no listed one. No public total counts them. A channel
+// a listed site also exposes is not here (it is credited to the listed site),
+// and a channel no site exposes (pool-only) is not here either — the family's
+// instance-wide totals have always counted it.
+export function channelsOnlyOnUnlistedSites(
+ sites: readonly Pick<Site, "listed" | "channels">[],
+): Set<string> {
+ const onListed = new Set<string>();
+ const onUnlisted = new Set<string>();
+ for (const site of sites) {
+ const into = isListedSite(site) ? onListed : onUnlisted;
+ for (const c of site.channels) into.add(c.slug);
+ }
+ for (const slug of onListed) onUnlisted.delete(slug);
+ return onUnlisted;
+}
+
export const SITE_DEFAULT_TITLE = "Transcript Browser";
export const SITE_DEFAULT_DESCRIPTION = "Browse and search video transcripts";
@@ -246,6 +279,8 @@ export const siteFieldsSchema = z.object({
).describe(d.cloudflareProject),
accent: settingsField(parseAccentSetting).describe(d.accent),
siteUrl: settingsField(parseSiteUrl).describe(d.siteUrl),
+ // Opt-out: only an explicit false unlists. Absent/true stays listed.
+ listed: settingsField((v): boolean => v !== false).describe(d.listed),
relatedSites: settingsField(parseRelatedSites).describe(d.relatedSites),
pwa: settingsField((v): boolean => v === true).describe(d.pwa),
// Opt-out: only an explicit false disables. Absent/true stays on.
@@ -335,6 +370,8 @@ export function siteToDisk(site: Site): Site {
: {}),
...(accent ? { accent } : {}),
...(siteUrl ? { siteUrl } : {}),
+ // Listed is the default: only the opt-out is persisted.
+ ...(site.listed === false ? { listed: false } : {}),
...(relatedSites.length > 0 ? { relatedSites } : {}),
...(site.pwa ? { pwa: true } : {}),
// Persist only the non-default: archives is on unless explicitly disabled.
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -3,6 +3,7 @@
## [Unreleased]
- **Transcripts that arrived after a video was first seen are counted.** The stats behind the homepage, the hub and every site's charts were cached per video and refreshed only when the video's metadata changed, so a transcript that came later — a Whisper run days after the download, or a video downloaded after the last index build — never reached them, and a video with YouTube captions alone had no transcription date. Counts and charts were low; the homepage could show a site with 0 transcripts, 0 channels and 0 hours while it served its videos. A stat is now also redone whenever the index re-reads the video, every transcript has a date, and a captioned video is dated by when its captions arrived rather than by a later Normalize run, so its place on "Transcribed over time" can move. **After updating, rebuild and restart the editor before anything else:** until then, **Build stats dataset** runs the old code and would undo the new stats, while a site, hub or homepage build already runs the new code — and the first stats build of any kind re-reads every video once (about 10–30 minutes on a large archive; it can be stopped and picks up where it stopped). Then build the index, the stats, the homepage, the hub, and the sites.
- **A stats build keeps the stats of a channel whose drive is not mounted, and will not undo a newer version's stats.** A channel whose media is on a drive that is not mounted (or is being moved) is left as it was instead of being read as a channel with no videos; a stats rebuild that has to start over refuses until the drive is back. A stats build refuses to clear stats written by a newer version of the editor; set `ARCHILYZER_STATS_ALLOW_DOWNGRADE=1` to roll back on purpose. Its log also says apart how many videos were downloaded since the last index build (they catch up after the next one) and how many the index skipped (no upload date, or it failed on them).
+- **An index build keeps a channel whose drive is not mounted, instead of dropping it from the sites.** **Build index**, a site build's data phase and `archilyzer index` read a channel whose media is on a drive that is not mounted (or is being moved, or whose link and config disagree) as a channel with no videos: they removed its videos from the index, and the next site build published the channel as gone. Such a channel is now left as the last build had it — its videos stay in the index, its pages stay as they were, and the sites built next still list it — and the log names it, with its storage location: one line per channel, ` Held: N channel(s), K video(s) kept.` at the end of the `Diff:` line, and the channels again on the last line. A data folder that fails to read is held the same way, and a channel with no data folder at all is said in the log instead of passed over. An index rebuild that has to start over (after an update that changes the index's format, or with no index yet) refuses while any channel is held and says which; mount the drive first, or set `ARCHILYZER_INDEX_ALLOW_HELD=1` to rebuild without that channel until its drive is back and the index is built again — on the command for a command-line build (`ARCHILYZER_INDEX_ALLOW_HELD=1 pnpm archilyzer index`), or in the editor's own environment, with a restart, for **Build index** and the site builds started from the editor.
- **Building the homepage now publishes the source: a read-only git mirror, its raw tree and a fresh tarball, behind a gate.** `archilyzer build homepage`, the `/sites` Homepage jobs and `pnpm ops build-homepage` run `archilyzer source publish` between compose and `next build`. It makes a fresh clone of the private `main` (the repository itself is never rewritten), rewrites that copy with git-filter-repo using your scrub rules (file contents and commit messages; your home directory becomes `/home/user` without a rule), and publishes it under `homepage/public` for `git clone https://archilyzer.pages.dev/source/archilyzer.git`, beside `/source/tree/` and the Downloads tarball. Before anything is written, every object of the rewritten history and every file about to be published is searched for every string you have denied; **one hit refuses the build**, and its log names the string only by where you wrote it (`denylist line 3 (len 5)`) and each hit by its object, field and byte offset — never a byte of the object. **A refusal withdraws the source**: the last publish is removed from `homepage/public` and the last build's copy from `homepage/out`, and **Deploy homepage refuses** a build whose source was not audited under today's rules and today's `main` ("run `archilyzer build homepage`, then deploy"). The rules live outside the repo, in `~/.config/archilyzer/source-scrub.txt` and `source-denylist.txt` (`ARCHILYZER_CONFIG_DIR`, `SOURCE_SCRUB_FILE`, `SOURCE_DENYLIST_FILE`); **without them the build refuses**, naming the missing file. **Put everything private in the denylist before any deploy, a preview included**: previews are public, and every deployment stays reachable at its own address until you delete it. Install git-filter-repo once (`pipx install git-filter-repo`; the editor's process needs `~/.local/bin` on its `PATH` to find it) — without it the build fetches it through `pipx run`, which needs the network — and gitleaks if you want its secret scan too. An unchanged `main` with unchanged rules is skipped, so a rebuild costs about 20 seconds only when something moved. A checkout with no git repository (the docker image, a tarball install) builds with the /source page's empty state. `archilyzer source publish --check` audits without writing, `archilyzer source audit <clone>/.git` checks any clone, `archilyzer build homepage --no-source` removes the published source instead, and `archilyzer doctor` reports the tools, the two files (rule counts and permissions, never their contents) and the last publish. `create-archives.sh` is gone. See PUBLISH.md, "The source mirror (homepage)".
- **umtool reads the corpus from its checkout (or `TRANSCRIPTS_DIR`), and the song project's data defaults to `~/.local/share/archilyzer/song`.** If yours is elsewhere, link it there before restarting umtool: `mkdir -p ~/.local/share/archilyzer && ln -s <where the data is> ~/.local/share/archilyzer/song` (the data stays where it is). With no `CHANNELS_DIR`, umtool reads the corpus at `$TRANSCRIPTS_DIR/channels`, else the checkout's own `transcripts/channels`; it used to fall back to an absolute path that existed on one machine only. The song project's videos default to `~/reports/quartering-uh-song/videos`; `SONG_DIR` and `VIDEO_ROOT` still win. The song project's tracked manifests record their paths relative to the song folders, and the twenty one-off `umtool/song/*.sh` run logs, which only ever ran on the machine that wrote them, are gone.
- **umtool's production build no longer reads the corpus folder.** Since umtool began finding the corpus from its checkout (the bullet above), `next build` treated the checkout's whole `transcripts/channels` as files to bundle. On a real archive it ran out of memory and was killed, so umtool could not be rebuilt. The build now ignores that folder and finishes in about 25 s at under 1 GB, the same as a checkout with no corpus. Nothing changes when umtool runs.
@@ -10,6 +11,7 @@
- **A social icon is checked by what it may contain, when it is saved new or edited and every time it is shown, and a refused one says why.** An icon must be one well-formed `<svg>` of shapes, groups, gradients, clips, masks, filters, text and simple animation, with SVG presentation attributes: no script, `style` block, `foreignObject`, link, embedded image, `title`/`desc` with anything but text (text-only ones are removed), or HTML element; no event handler, however it is written; a `style` attribute of presentation properties only; a reference only to something inside the icon, written plainly; and nothing that could load from elsewhere (a CSS escape or comment, `image-set(`, `image(`, `cross-fade(`, `element(`, `src(`, `paint(`, `@import`). Comments, a leading XML declaration and a plain DOCTYPE are removed. A refused save ends with the reason ("… has an invalid SVG: it has an event handler attribute.", "… it links to something outside the icon.") and never repeats the markup; for a drawing program's file it says to export it with presentation attributes rather than a style block (in Inkscape, save as Plain SVG). **Upgrading:** an icon an older build stored is kept as it is when a save does not change it — a pause, a priority or a title still saves — but a page shows it as its label until its SVG is replaced; `archilyzer doctor`'s new "social icons" line names every stored icon that fails, by file and label, with the reason.
- **Two grounds, Light and Dark, and no accent picker in the header.** The editor's header keeps its theme toggle, which cycles System, Light and Dark; the theme menu (Base and Accent) is gone, and the editor wears its own accent, Signal. A stored choice of the retired third ground loads as Light and is rewritten once; a stored accent is not read and is left in storage. A site's accent is still set in its form; the form's hint no longer says a reader can pick another.
- **The hub URL hints say what the setting does now.** Settings' **Family hub URL** and a site's **Hub URL** no longer promise a Hub link in the header (it was removed): the value is published as `hubUrl` in each site's `/site.json` and `/corpus.json`, so the hub can tell its member sites. `SETTINGS.md` and `SITE.md` say the same.
+- **A site can be left off the homepage and the hub.** A site's settings have a new checkbox, **List on the Archilyzer homepage and hub**, on by default (`listed` in `site.json`; only `false` is written). Turned off, the site still builds and deploys at its own URL as before, but the homepage has no card, chart series, `/stats` entry or recent item for it; the hub does not list it as a member, search it, or name it in its `corpus.json` and `llms.txt`; no other site's footer links it; and `channel-sites.json` and the homepage's `stats/` leave it out. A channel only unlisted sites carry is in none of the published totals, the homepage's headline numbers included; a channel a listed site also carries is counted under the listed site. The editor's own pages still show every site. It takes effect at the next homepage, hub and site builds.
## [0.10.0] - 2026-09-28
- **The homepage can be built and deployed from `/sites`.** Under a new **Homepage** section, after Hub, there is **Build homepage** (tick **Deploy after build** to ship it in the same job, only if the build succeeds) and **Deploy homepage**, which ships the build already in `homepage/out`. A **Preview branch** box beside them sends either deploy to a Cloudflare Pages preview of the `archilyzer` project instead of production, and shows the preview's address as you type; a name Cloudflare would refuse or rewrite, or `main`, greys the deploy buttons out and says why. A line under the buttons says what a deploy would ship: when `homepage/out` was built (or that it holds no build yet), and where it goes, with the live URL. Deploy homepage with nothing built is refused before any job starts. The homepage reads the search index as it stands, so run **Build index** first when its numbers should move. The jobs run the same code as `archilyzer build homepage` / `deploy homepage`, and show on `/jobs` as `build-homepage`, `deploy-homepage` and `build-deploy-homepage`. The Hub section no longer describes the homepage.
diff --git a/editor/app/sites/actions.ts b/editor/app/sites/actions.ts
@@ -92,6 +92,9 @@ export async function saveSiteAction(
};
}
const siteUrl = parseSiteUrl(siteUrlRaw);
+ // Listed on the homepage and hub by default: the same opt-out idiom as
+ // archives below (an unchecked box sends no key → persisted as false).
+ const listed = formData.get("listed") === "on";
// Hub parent (per-site override of the family default) + PWA opt-in.
const hubUrlRaw = String(formData.get("hubUrl") ?? "").trim();
@@ -217,6 +220,8 @@ export async function saveSiteAction(
...(accent ? { accent } : {}),
...(cloudflareProject ? { cloudflareProject } : {}),
...(siteUrl ? { siteUrl } : {}),
+ // The Site is rebuilt from the form: a key missing here is dropped on save.
+ ...(listed ? {} : { listed: false }),
...(hubUrl ? { hubUrl } : {}),
...(pwa ? { pwa: true } : {}),
...(archives ? {} : { archives: false }),
diff --git a/editor/app/sites/components/SiteForm.tsx b/editor/app/sites/components/SiteForm.tsx
@@ -327,6 +327,22 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) {
defaultValue={initial.siteUrl ?? ""}
hint="Absolute URL this site is served at (e.g. https://jeralyzer.com). Used so other sites can link to it in their footer. Leave blank to omit this site from cross-site lists."
/>
+ <label className="flex items-center gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="listed"
+ defaultChecked={initial.listed !== false}
+ className="accent-brand"
+ />
+ List on the Archilyzer homepage and hub
+ </label>
+ <p className="-mt-2 text-xs text-muted-foreground">
+ On by default. Turn off to leave this site out of the homepage (its
+ card, chart and stats), the hub (its members, search, corpus.json and
+ llms.txt) and every other site's footer, and to count its own
+ channels in none of the published totals. The site still builds and
+ deploys as before, at its own URL.
+ </p>
<Field
label="Hub URL"
name="hubUrl"
diff --git a/editor/e2e/helpers.ts b/editor/e2e/helpers.ts
@@ -152,6 +152,7 @@ export async function writeSite(
? { cloudflareProject: site.cloudflareProject }
: {}),
...(site.siteUrl ? { siteUrl: site.siteUrl } : {}),
+ ...(site.listed === false ? { listed: false } : {}),
...(site.relatedSites ? { relatedSites: site.relatedSites } : {}),
};
await writeFile(
diff --git a/editor/e2e/sites-crud.spec.ts b/editor/e2e/sites-crud.spec.ts
@@ -345,6 +345,62 @@ test("archives + per-video transcript downloads opt-outs round-trip", async ({
await expect(downloads).toBeChecked();
});
+test("the listed opt-out round-trips, and a save of another field keeps it", async ({
+ page,
+}) => {
+ await resetData("empty");
+ // An invented fixture id: no real site is named in a test.
+ await writeSite("fixture-unlisted", {
+ siteTitle: "Unlisted Fixture",
+ siteUrl: "https://fixture-unlisted.example",
+ listed: false,
+ });
+
+ type ListedSiteFile = { siteTitle?: string; listed?: boolean };
+ const file = "test-transcripts/sites/fixture-unlisted/site.json";
+ const listed = page.getByRole("checkbox", {
+ name: "List on the Archilyzer homepage and hub",
+ });
+ const save = async () => {
+ await page.getByRole("button", { name: /save site/i }).click();
+ await expect(
+ page.getByRole("status").filter({ hasText: "Saved" }),
+ ).toBeVisible();
+ };
+
+ // `false` on disk opens unticked, and a save that changes only the title
+ // keeps it: the action rebuilds the site from the form.
+ await page.goto("/sites/fixture-unlisted");
+ await expect(listed).not.toBeChecked();
+ await page.getByLabel(/site title/i).fill("Unlisted Fixture Renamed");
+ await save();
+ await expect(async () => {
+ const site = await readJson<ListedSiteFile>(file);
+ expect(site.siteTitle).toBe("Unlisted Fixture Renamed");
+ expect(site.listed).toBe(false);
+ }).toPass({ timeout: 10_000 });
+
+ // Ticking it removes the key — listed is the default.
+ await page.goto("/sites/fixture-unlisted");
+ await expect(listed).not.toBeChecked();
+ await listed.check();
+ await save();
+ await expect(async () => {
+ const site = await readJson<ListedSiteFile>(file);
+ expect("listed" in site).toBe(false);
+ }).toPass({ timeout: 10_000 });
+
+ // And unticking writes the explicit false again.
+ await page.goto("/sites/fixture-unlisted");
+ await expect(listed).toBeChecked();
+ await listed.uncheck();
+ await save();
+ await expect(async () => {
+ const site = await readJson<ListedSiteFile>(file);
+ expect(site.listed).toBe(false);
+ }).toPass({ timeout: 10_000 });
+});
+
test("the hub form's per-video transcript downloads opt-out round-trips to homepage.json", async ({
page,
}) => {
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -6,6 +6,7 @@
- **A chart's stacked bars are separated by a 2 px gap in the chart card's colour.** A stacked bar's segments were drawn touching; they now have a 2 px gap in the card's colour between them, and in high-contrast mode the system's background colour. Stacked areas keep their line in each series' colour along the top, charts of one series, line charts and side-by-side bars are unchanged. Needs a rebuild and deploy of each site.
- **Two grounds, Light and Dark, and each site in its own accent.** The third ground, the warm paper one, is gone: the header's toggle cycles System, Light and Dark. A reader who had chosen it gets Light, before the page first paints and with no other ground on the way, and the stored choice becomes Light (the old paper theme's `archive` + `light` too). The theme menu's accent picker is gone from the header and the slide-out menu: every page wears the site's own accent (`site.json` `accent`), and a reader's stored pick from before is not read and is left in storage. Needs a rebuild and deploy of each site.
- **The header carries the operator's social links and one theme toggle, keeps the site's name on a small screen, and links to the Archilyzer home in place of the sites menu.** Every site's header and the hub's end with the social icons (the site's `socialLinks`, else `settings.json`'s) followed by the theme toggle, all 36 px keys (44 px on a touch screen) with a focus ring. From 520 px wide the header shows every link, up to four (with more, the ones marked **Keep in header on small screens** first, then the last of the rest); below 520 px it shows only the marked ones (none marked → none) and keeps the site's name beside them. The switch is 32.5rem, so at a larger text size it comes later. With one marked link, every current site's name shows in full from 360 px wide on a touch screen. The footer keeps every link, in the same keys (its icons were 20 px and turned the accent on hover; they now turn the text colour), and wraps them rather than widen the page. The **Sites** dropdown and the **Hub** link are gone from the header and the slide-out menu: in their place a link, **Archilyzer**, goes to the Archilyzer home's Official Instances, in the same tab (not on the hub, which lists them itself). **Changelog** moved from the header and the menu to the footer, after Use with AI. The nav and the Archilyzer link are inline from 1024 px wide; below that they are in the slide-out menu, which now holds only them. Only as last resorts, for a very long name on a phone, does the name drop (its mark stays; the same before and after the page's font has loaded, and never with its last letter cut off) and do the icons scroll sideways in their own box. A site's `hubUrl` still loads and is no longer shown. Needs a rebuild and deploy of each site.
+- **An unlisted site is not in the hub or in another site's footer.** The hub's members (`hub-sites.json`), and so its federated search, `corpus.json` and `llms.txt`, leave out a site whose `listed` is `false`; the hub's instance figures count none of the channels only it carries; and no other site's footer links it, even from a featured group. The unlisted site's own pages are unchanged.
- **A clear screen until the first Search.** A plain visit to a site's search page, and to the hub's, shows the search bar, the page's intro and the footer: no count, no listing and no results controls, and the line under the bar, "Press Enter or click Search to apply", says what to do. Search with the box empty lists every video, as before. A link that carries a query or a filter (`qt=`, `q=`, `tg=`, a share link, the older filter keys) still shows its results on load. Within one visit the results stay: going to Ask AI or another page and coming back keeps them. A reload starts over, and shows results at once only when the address carries a query or a filter. A query restored from the last visit waits in the box until Search, over the clear screen or under a filter link's results, and it still waits after another page and Back. Needs a rebuild and deploy of each site and the hub.
## [0.10.0] - 2026-09-28
diff --git a/homepage/CHANGELOG.md b/homepage/CHANGELOG.md
@@ -1,6 +1,7 @@
# Homepage Changelog
## [Unreleased]
+- **An unlisted site is not on the homepage.** A site whose settings turn off **List on the Archilyzer homepage and hub** (`listed: false`) has no Official Instances card, chart series, `/stats` entry or recent item, is not in `channel-sites.json` or `stats/`, and the channels only it carries count in none of the numbers, the headline totals included. The summary's version is 6. The e2e fixture has a seventh, unlisted site that no page names.
- **`/#instances` goes straight to Official Instances.** The section carries `id="instances"`, clear of the sticky header, and every archive's header now links there (`INSTANCES_URL` in `common/lib/project.ts`). With no sites the link lands on the top of the page.
- **The social links are in the header beside the theme toggle, and on a small screen the header keeps the name.** The operator's social icons (`homepage.json`'s, else `settings.json`'s) sit in the header's bar as well as in the footer's Elsewhere column, followed by the theme toggle, all spaced alike. From 768 px wide the bar is wordmark, nav, icons, toggle; below 768 px it is wordmark, icons, toggle, and the nav has the rule below to itself, where its four links fit. From 520 px wide the header shows every link, up to four (with more, the ones marked **Keep in header on small screens** first, then the last of the rest); below 520 px it shows only the marked ones (none marked → none; the switch is 32.5rem, so at a larger text size it comes later), and the wordmark keeps its full name: "Archilyzer" shows from 292 px wide on a touch screen with one marked link. The footer always shows every link. Only as last resorts, on a screen narrower still or at a much larger text size, does the wordmark's text drop (its mark stays) and do the icons scroll sideways in their own box, the last one in view first. Each is an icon named by its label, with no text beside it.
diff --git a/homepage/e2e/fixture-summary.ts b/homepage/e2e/fixture-summary.ts
@@ -26,7 +26,14 @@ import type { VideoStat } from "../../common/lib/stats";
// symlog axis reaches 10K while the Recent weeks stay linear and small;
// • every site transcribes every day of the last 200, so every line is full
// and every series has a non-zero start for Indexed;
-// • uploads spread over 2019–2026, for the growth chart's months.
+// • uploads spread over 2019–2026, for the growth chart's months;
+// • a seventh site, UNLISTED (site.json `listed: false`, release 14 slice
+// HS): its own two channels transcribe every day like the rest, and it
+// also exposes the first site's first channel. The summary names it nowhere
+// and counts its own channels in no total, so `buildFixtureSummary()` is
+// exactly `buildFixtureSummary(FIXTURE_SITES, [])` — every number the specs
+// read is the six listed sites'. Every site in the second list is unlisted:
+// `buildFixtureInputs` sets `listed: false` on it, whatever it carries.
export const FIXTURE_SUMMARY_NAME = ".e2e-summary.json";
@@ -42,7 +49,8 @@ export const FIXTURE_PALE_HEX = "#f4c2d7";
// site.json one (a named accent's site, the custom hex's and a site with no
// accent have one — the last ends MID-WORD, "Fix" + "ture Three", as
// "Jer" + "alyzer" does; the rest show their title plain); `daily` how many
-// recordings each channel transcribes a day.
+// recordings each channel transcribes a day. Whether a site is listed is the
+// list it is passed in (buildFixtureInputs), not a field.
export type FixtureSite = {
siteId: string;
siteTitle: string;
@@ -61,6 +69,16 @@ export const FIXTURE_SITES: readonly FixtureSite[] = [
{ siteId: "fixture-six", siteTitle: "Fixture Six", accent: "vermilion", channels: 2, daily: 3 },
];
+// The unlisted site (an invented id, as every fixture's). Not in FIXTURE_SITES,
+// which is the six the pages show; buildFixtureInputs' second list, which
+// unlists it.
+export const FIXTURE_UNLISTED_SITE: FixtureSite = {
+ siteId: "fixture-unlisted",
+ siteTitle: "Fixture Unlisted",
+ channels: 2,
+ daily: 6,
+};
+
const DAY = 86_400_000;
const SPIKE = { site: 0, start: Date.UTC(2026, 4, 11), days: 7, count: 12_000 };
const HISTORY_DAYS = 200;
@@ -76,7 +94,13 @@ function uploadDate(n: number): string {
return ymd(d.getTime());
}
-export function buildFixtureSummary(fixtureSites: readonly FixtureSite[] = FIXTURE_SITES) {
+// The builder's inputs: the recordings, the channel → sites map and the sites.
+// `fixtureSites` are listed; every site in `unlistedSites` is written with
+// `listed: false`, and each also exposes the first listed site's first channel.
+export function buildFixtureInputs(
+ fixtureSites: readonly FixtureSite[] = FIXTURE_SITES,
+ unlistedSites: readonly FixtureSite[] = [FIXTURE_UNLISTED_SITE],
+): { stats: VideoStat[]; channelSites: Record<string, string[]>; sites: Site[] } {
const stats: VideoStat[] = [];
const channelSites: Record<string, string[]> = {};
const sites: Site[] = [];
@@ -112,17 +136,24 @@ export function buildFixtureSummary(fixtureSites: readonly FixtureSite[] = FIXTU
});
};
const today = Date.UTC(2026, 8, 15);
- fixtureSites.forEach((s, i) => {
+ // A site's own channels, `<siteId>-ch<N>`, each transcribing every day of the
+ // last HISTORY_DAYS; `shared` are other sites' channels it also exposes.
+ const addSite = (
+ s: FixtureSite,
+ i: number,
+ { listed = true, shared = [] }: { listed?: boolean; shared?: string[] } = {},
+ ) => {
const slugs = Array.from({ length: s.channels }, (_, c) => `${s.siteId}-ch${c + 1}`);
- for (const slug of slugs) channelSites[slug] = [s.siteId];
+ for (const slug of [...slugs, ...shared]) (channelSites[slug] ??= []).push(s.siteId);
sites.push({
siteId: s.siteId,
siteTitle: s.siteTitle,
siteDescription: `${s.siteTitle}, an e2e fixture archive.`,
siteUrl: `https://${s.siteId}.example`,
- channels: slugs.map((slug) => ({ slug })),
+ channels: [...slugs, ...shared].map((slug) => ({ slug })),
...(s.accent ? { accent: s.accent } : {}),
...(s.wordmarkLead ? { wordmarkLead: s.wordmarkLead } : {}),
+ ...(listed ? {} : { listed: false }),
} as unknown as Site);
slugs.forEach((slug, c) => {
const name = `${s.siteTitle} Channel ${c + 1}`;
@@ -133,13 +164,31 @@ export function buildFixtureSummary(fixtureSites: readonly FixtureSite[] = FIXTU
for (let k = 0; k < count; k++) record(i, slug, name, day);
}
});
- });
+ };
+ fixtureSites.forEach((s, i) => addSite(s, i));
// The megaspike: one week of bulk transcription on the first site's first
// channel.
const spikeSite = fixtureSites[SPIKE.site];
for (let k = 0; k < SPIKE.count; k++) {
record(SPIKE.site, `${spikeSite.siteId}-ch1`, `${spikeSite.siteTitle} Channel 1`, SPIKE.start + (k % SPIKE.days) * DAY);
}
+ // The unlisted sites LAST, so every listed site's recordings (their numbers,
+ // dates and states) are the same with them or without.
+ unlistedSites.forEach((s, j) =>
+ addSite(s, fixtureSites.length + j, {
+ listed: false,
+ shared: [`${spikeSite.siteId}-ch1`],
+ }),
+ );
+ return { stats, channelSites, sites };
+}
+
+// The summary the e2e dev server reads: the real builder over those inputs.
+export function buildFixtureSummary(
+ fixtureSites: readonly FixtureSite[] = FIXTURE_SITES,
+ unlistedSites: readonly FixtureSite[] = [FIXTURE_UNLISTED_SITE],
+) {
+ const { stats, channelSites, sites } = buildFixtureInputs(fixtureSites, unlistedSites);
return buildHomepageSummary(stats, channelSites, sites, FIXTURE_NOW);
}
diff --git a/homepage/e2e/unlisted-site.spec.ts b/homepage/e2e/unlisted-site.spec.ts
@@ -0,0 +1,61 @@
+import fs from "node:fs";
+import path from "node:path";
+import { test, expect } from "@playwright/test";
+import {
+ FIXTURE_SITES,
+ FIXTURE_SUMMARY_NAME,
+ FIXTURE_UNLISTED_SITE,
+ buildFixtureInputs,
+ buildFixtureSummary,
+} from "./fixture-summary";
+
+// Release 14 slice HS: a site with site.json `listed: false` builds and
+// deploys, and the homepage lists it nowhere — not a card, not a chart series,
+// not a recent item — and counts the channels only it exposes in no total.
+//
+// The fixture summary (fixture-summary.ts) is built over six listed sites and
+// one unlisted one, which has two channels of its own and shares the first
+// site's first channel.
+
+const NEEDLES = [FIXTURE_UNLISTED_SITE.siteId, FIXTURE_UNLISTED_SITE.siteTitle];
+
+test("the summary the server reads is the one built without the unlisted site", () => {
+ // The site is in the builder's input: unlisted, with recordings of its own
+ // and a channel shared with the first listed site.
+ const { stats, channelSites, sites } = buildFixtureInputs();
+ const unlisted = sites.find((s) => s.siteId === FIXTURE_UNLISTED_SITE.siteId);
+ expect(unlisted?.listed).toBe(false);
+ const own = stats.filter((s) => s.channelSlug.startsWith(`${FIXTURE_UNLISTED_SITE.siteId}-ch`));
+ expect(own.length).toBeGreaterThan(0);
+ expect(channelSites[`${FIXTURE_SITES[0].siteId}-ch1`]).toContain(FIXTURE_UNLISTED_SITE.siteId);
+ // Listed, the same site would change the summary: a card, and its recordings.
+ const withoutIt = buildFixtureSummary(FIXTURE_SITES, []);
+ const asListed = buildFixtureSummary([...FIXTURE_SITES, FIXTURE_UNLISTED_SITE], []);
+ expect(asListed.sites.map((s) => s.siteId)).toContain(FIXTURE_UNLISTED_SITE.siteId);
+ expect(asListed.totals.transcripts).toBe(withoutIt.totals.transcripts + own.length);
+
+ const onDisk = JSON.parse(
+ fs.readFileSync(path.resolve(process.cwd(), "e2e", FIXTURE_SUMMARY_NAME), "utf8"),
+ );
+ // Every array and every total: what the six listed sites alone produce.
+ expect(onDisk).toEqual(JSON.parse(JSON.stringify(withoutIt)));
+ const text = JSON.stringify(onDisk);
+ for (const needle of NEEDLES) expect(text).not.toContain(needle);
+ expect(onDisk.sites.map((s: { siteId: string }) => s.siteId)).toEqual(
+ FIXTURE_SITES.map((s) => s.siteId),
+ );
+});
+
+for (const route of ["/", "/stats/"]) {
+ test(`${route} names no unlisted site, in its HTML or on the page`, async ({ page }) => {
+ const res = await page.goto(route);
+ expect(res?.ok()).toBe(true);
+ const served = (await res?.text()) ?? "";
+ // The listed sites are there, so the check below is not vacuous.
+ for (const s of FIXTURE_SITES) expect(served).toContain(s.siteId);
+ for (const needle of NEEDLES) {
+ expect(served).not.toContain(needle);
+ expect(await page.content()).not.toContain(needle);
+ }
+ });
+}
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -153,6 +153,16 @@ THROWS at declaration — module load — on a name `SUB_FILE_RE` matches;
Never name the curated field `tags`. Never assume a `tags.json` is the keyword list:
`transcripts/tags.json` and `sites/<id>/tags.json` are curated tags.
+**`listed` / `unlisted` mean three unrelated things** (release 14, slice HS):
+
+| Which | Where | What it is |
+| --- | --- | --- |
+| A video's visibility | `common/lib/availability.ts` (the `"unlisted"` state, `isUnlisted`), `common/lib/transcripts{,-server}.ts`, `common/components/shareUrl.ts`, `common/controller/buildIndex.ts` | The platform's own "unlisted" (reachable by link, not listed on the channel). |
+| The hub's list has loaded | `export/app/components/hub/useHubSites.ts` — `listed` | `/hub-sites.json` has been answered and `/hub-summary.json` has settled. |
+| **A site the family lists** | `site.json` `listed` (`common/lib/siteSchema.ts` — `isListedSite`, `channelsOnlyOnUnlistedSites`) | Absent = listed; `false` keeps the site off the homepage, the hub and the other sites' footers, and out of the public totals. |
+
+A grep for either word finds all three; read the file before assuming which.
+
**Never publish a path segment named `.git`.** wrangler's Pages upload drops `**/.git` (and
`**/node_modules`) SILENTLY, and Cloudflare's managed rules block `/.git/` requests. The source
mirror is `/source/archilyzer.git/` for that reason (release 12, below). A mirror named `.git` would
@@ -7413,8 +7423,12 @@ on anchors elsewhere in this file:
index it:
- no `upload_date` (`buildIndex.ts:701`);
- a processing failure;
- - or the channel's media was unreachable during that build, since buildIndex has no drive
- guard.
+ - or its channel was held during that build (its media unreachable). Since release 15 (slice
+ IG) an incremental build keeps a held channel's records, so this is a video that reached the
+ drive after the last index build that could read it. Once the drive is back, a stats-only run
+ (Build stats dataset, or a pool composer) before the next index build counts it here, and
+ that index build heals it. Under `ARCHILYZER_INDEX_ALLOW_HELD`, a full rebuild drops all of
+ the held channel's records the same way (see "The index build's hold").
It stays until fixed, and is logged as such rather than as pending (`:500`).
- **A transcript always has a date, and a caption video takes its captions' arrival.**
@@ -7446,10 +7460,9 @@ on anchors elsewhere in this file:
- This is the build's own guard. The job registry's `needsMedia` check
(`streamCommand.ts refuseForUnreachableMedia`) is per channel and needs a `channelSlug`, so it
never covered this pool-wide build, from the editor or from the CLI.
- - **buildIndex has no such guard.** An index build with a drive unmounted drops those channels'
- index records, and the site pages built from it lose them.
- - Its `Diff: … -R removed` line (`buildIndex.ts:641`) shows it.
- - A proper hold there is a follow-up slice (`stats-cache-key.md`, "Left").
+ - **buildIndex had no such guard until release 15.** An index build with a drive unmounted
+ dropped those channels' index records, and the site pages built from it lost them. It holds
+ them now: see "The index build's hold" below.
- **One stats build at a time: an operator rule, not a lock.**
- Two concurrent runs are harmless unless one clears the cache (a schema change) after the other
has scanned. The other then collects a partly refilled `statsByPath` and publishes truncated
@@ -7484,3 +7497,67 @@ on anchors elsewhere in this file:
finishes the rest.
- Run in the editor, it stalls the editor's event loop for the length of the pass. Prefer the CLI
with the editor idle.
+
+## The index build's hold (verified 2026-09-29, branch `r15/index-hold`)
+
+The record is [`release-15.md`](release-15.md), "Slice IG, as shipped". Anchors are at the branch
+tip. The branch added lines to `buildIndex.ts` from `:10` on, so every `buildIndex.ts` anchor above
+this section is stale, by +16 near the top and +203 at the end; they are not rewritten in place.
+
+- **An unmounted drive is not an empty channel, for the index either.**
+ - `scanSource` (`common/controller/buildIndex.ts:309`) calls `inspectChannelMedia({ channelsDir },
+ slug, cfg)` for every channel that is not excluded and not social (`:348`), before it reads
+ `data/`. A status other than `ok` or `in-place` (`isMediaHeld`,
+ `common/lib/channelMediaHold.ts:26`) puts the channel in `held` with a reason and no path, and
+ it is not scanned.
+ - It calls it again after the walk (`:465`), so a drive that goes away mid-walk holds the
+ channel instead of dropping the videos after that point (without it, `buildIndex.test.ts`
+ case (i) removes 3 of 4).
+ - A failed `readdir(data/)` holds too (`its data directory could not be read (<code>)`), except
+ ENOENT on an `in-place` channel, which is a channel with no downloads and is logged as such
+ (`:362`). A per-video metadata `stat` failing with anything but ENOENT or ENOTDIR holds the
+ channel too, logged as `a video in its data directory could not be read (<code>)` (`:387`).
+- **What a held channel keeps, on an incremental build:**
+ - its `mtimes` records: the removal pass skips its keys and counts them (`:715`), so `sums`,
+ `cues`, `subs`, `digests` and `byChannel` keep them too;
+ - its shared transcript, subs and digest trees: not rewritten, not pruned, not removed
+ (`:1204`, `:1393`, `:1772`, and the top-level cleanups `:1513`, `:1879`);
+ - its subs and digest stats, carried from `channelStatsDb` / `channelDigestStatsDb`, so the
+ per-site subs and digest manifests still list it;
+ - its availability states: the maybe-missing overlay skips it (`:1324`), and the last build's
+ `videoState` entries for it are carried over (`:1356`). Its `availability.json` files are on
+ the missing drive, and a missing one reads as `maybe_missing`.
+ - The per-site summaries come from LMDB, so the site built next still lists its videos.
+- **A curated-tag change while a channel is held:** the re-apply pass re-derives its records in
+ LMDB (no disk read), but its pages are not written, so `curatedPagesPending` is NOT cleared while
+ a channel is held and the pass had pages pending (`:1551`). The first build with the drive back
+ rewrites them.
+- **A full rebuild with a channel held refuses** (`:649`). A full rebuild is a schema change or a
+ first build (no `meta.schema`); it clears every sub-DB, and a held channel cannot be re-read.
+ - The scan now runs BEFORE the clear (`:639`), so the refusal leaves the index untouched.
+ `scanStartedAt` is still taken at the scan's start.
+ - The message names each channel with its location's label, the ways out (`HELD_WAYS_OUT`,
+ shared with the stats build's refusal), and `ARCHILYZER_INDEX_ALLOW_HELD` (`:512`, declared in
+ `envVars.ts`) with where it is set: the command's own environment for a CLI run; the editor's
+ own environment, and so a restart, for the editor's Build index job or a site build started
+ from the editor (their children inherit `process.env`). The CLI exits 1 on it, so a site
+ build's data phase fails with it.
+ - With the variable set, the build proceeds: the held channel's records go with the clear, its
+ shared trees are left on disk, and it is out of the index (the site lists it with 0 videos)
+ until its media is back and an index build runs.
+- **Who sees it:** `BuildIndexResult.heldChannels` (`:502`); the log's per-channel line, the
+ `Diff:` line's ` Held: N channel(s), K video(s) kept.` suffix (`:780`), and the `Done in` line's
+ ` Held, their media not readable: <slugs>.` suffix. The CLI (`archilyzer index`, the export's
+ and homepage's `build:index`, so every site build's data phase) prints the log; the editor's
+ **Build index** job (`buildIndexAction`) streams it into the job log, which `pnpm ops build-index
+ --wait` follows. Nothing reads the result's field outside the tests.
+- **The words are shared with the stats build** (`lib/channelMediaHold.ts`): `HELD_REASON` is a
+ `Record<ChannelMediaStatus, string>`, so a new status added to `inspectChannelMedia` fails tsc
+ until it has a reason; `isMediaHeld` treats any status but `ok` and `in-place` as held.
+- **Not covered:** a drive that drops during the PROCESSING phase (after the scan), for a changed
+ video whose metadata was read before the drop. A transcript read that fails after it is caught
+ as "no cues" (`cueList = undefined`, `:838`); a sub-track read that fails is skipped, and with
+ none left the video's subs are removed (`subs.remove`, `:895`); a digest load that fails leaves
+ no digest, which is removed (`digests.remove`, `:962`/`:965`). `mtimes` is then written with the
+ video's current mtimes, so the loss lasts until any of its tracked mtimes (metadata, transcript,
+ subs, availability, digest) moves.
diff --git a/plans/STATE.md b/plans/STATE.md
@@ -20,13 +20,13 @@ holds the record, the review and the rollout. FACTS has "The stats cache key".
step 0: it runs the source publish.
- **Merge note:** `homepage/social-visible` merged `main` (`10cefd15`) at `4d11542c`; the one
conflict, `homepage/CHANGELOG.md`'s `[Unreleased]`, kept both sides.
-- **FOLLOW-UP, its own slice: the index build still treats an unmounted drive as an empty
- channel.** It drops that channel's index records, and the next site build publishes the channel
- as gone.
- - The fix: give `buildIndex` the stats build's hold, or at least a refusal with an override.
- - Schedule it before routine builds resume after this rollout.
- - Until then, the rollout's step 3 `Diff:` check is the safeguard: thousands removed means a
- drive was missing.
+- **CLOSED by release 15 slice IG (branch `r15/index-hold`, [`release-15.md`](release-15.md)):
+ the index build no longer treats an unmounted drive as an empty channel.** It holds the channel:
+ not rescanned, its index records and shared pages kept, and the `Diff:` line says
+ ` Held: N channel(s), K video(s) kept.` A full rebuild with a channel held refuses unless
+ `ARCHILYZER_INDEX_ALLOW_HELD=1`. FACTS has "The index build's hold". Until that branch is merged
+ and rolled out, the rollout's step 3 `Diff:` check stays the safeguard: thousands removed means
+ a drive was missing.
**Now (2026-09-28, evening): release 12 — the source mirror — is merged to `main` and NOT rolled
out.** [`release-12.md`](release-12.md) holds Q's and R's records, their reviews, "Merged" and
diff --git a/plans/release-14.md b/plans/release-14.md
@@ -20,10 +20,11 @@ slice HP added to it on the operator's ruling of the same day. Rules:
|---|---|---|---|
| HP | `homepage/social-visible` | The homepage's social row and one theme toggle in the header at every width, the wordmark's text dropped first on a very small screen and the row scrolling only as the last resort, one shared `SocialLinks` component, larger keys with a focus ring; Changelog in the footer only; the instance cards' names as the site's wordmark; the social icon checked by an allowlist on save and at render; a sized SVG with no viewBox gets one; `featured` ("Show in header") on a social link | `common/components/{SocialLinks,SocialScroll,ThemeRadios,ThemeToggle,ThemeScript,ThemeProvider,Wordmark}.tsx` + `themeConfig.ts`, `common/bin/doctor.ts`, the growth chart, `common/lib/{socialSvg,socialLinks}.ts` + tests, `common/lib/settingsSchema.ts` (the social-link type, parser, docs; the normalizer moved to `socialSvg.ts`), `common/lib/normalizeSocialSvg.test.ts`, `common/lib/{settings,site,homepage}.ts` (the save errors), `common/lib/{homepageSummary,siteColor}.ts`, `homepage/app/components/{Header,Footer,ArchiveCards}.tsx`, `homepage/app/lib/{nav,summary}.ts`, `homepage/app/not-found.tsx`, `homepage/e2e/**` (the fixtures, `helpers.ts`, the new and the rewritten specs), `homepage/playwright.config.ts`, `export/app/components/{MobileMenu,Footer}.tsx` (the `ThemeRadios` swap; the footer's read path), `editor/app/components/SocialLinksField.tsx` + `socialLinksJson{,.test}.ts`, `editor/app/{settings,sites}/actions.ts` (the save errors), `editor/e2e/settings.spec.ts`, `SETTINGS.md`, `SITE.md` |
| Lows, chart gap, T1, H1 (H3 folded in), H2 | `r14/two-grounds-headers` | The final review's Lows; the charts' surface gap; two grounds and each site in its own accent; the export and hub headers carry the social row and the toggle as one group, with an Archilyzer link to the homepage's `#instances` in place of the sites dropdown and the hub link; Changelog to the footer; after its review, the narrow header keeps the name and shows only the marked links (every header) | `common/components/{ThemeProvider,ThemeScript,ThemeToggle,SocialScroll}.tsx` + `themeConfig.ts` (and the deleted `ThemeMenu`, `ThemeRadios`), `common/components/charts/{ChartView,CrossSiteChart,surfaceGap}`, `common/styles/tokens.css`, `common/lib/{brand,accent,siteColor,paths,project,socialSvg,siteSchema,settingsSchema}.ts` + tests, `scripts/next-build-trace.test.mjs`, `export/app/components/{Header,MobileMenu,Footer}.tsx` (and the deleted `SiblingSwitcher`), `export/app/{layout.tsx,globals.css,changelog/page.tsx,lib/brand.ts}`, `export/e2e{,-hub}/**` (the theme, header and branding specs), `export/playwright.config.ts`, `editor/app/{layout.tsx,globals.css,sites/components/SiteForm.tsx}`, `editor/e2e/theme.spec.ts`, `homepage/app/{page.tsx,layout.tsx,globals.css,lib/*,changelog/page.tsx,components/{Header,ArchiveGrowthChart,ArchiveCards}.tsx}`, `homepage/e2e/**`, `homepage/content/docs/operate.md`, `SETTINGS.md`, `SITE.md` |
+| HS | `r14/hidden-sites` | Hidden sites: `site.json` `listed` (absent = listed). An unlisted site builds and deploys as before, and is left off the homepage (cards, chart, `/stats`), the hub (members, federated search, `corpus.json`, `llms.txt`), every other site's footer and the published id lists (`channel-sites.json`, the pooled `stats/`); the channels only it exposes count in no public total. One checkbox in the site form | `common/lib/{siteSchema,site,homepageSummary}.ts` + tests, `common/controller/{poolSummary,buildStats}.ts` + tests, `common/bin/{compose-homepage,compose-hub}.ts` + `compose-hub.test.ts`, `editor/app/sites/{actions.ts,components/SiteForm.tsx}`, `editor/e2e/{helpers.ts,sites-crud.spec.ts}`, `homepage/e2e/{fixture-summary.ts,unlisted-site.spec.ts}`, `SITE.md`, `plans/FACTS.md` (Naming hazards) |
| S1 | `r14/first-search` | A clear screen until the first Search, on every site's search page and the hub's: no results area until the visitor asks (a Search, a profile load, or a link that carries a query or a filter), with the bar's "Press Enter or click Search to apply" line meanwhile; two page-life flags, the hold of release 8 unchanged and the gate new | `common/components/{SearchSessionContext,SearchResults,SearchBar}.tsx`, `export/e2e/first-search.spec.ts` (new), `export/e2e/{browse-all,workspace-shell,charts,restore-no-refire,responsive,tag-chips}.spec.ts`, `export/e2e/helpers.ts` (`showAll`), `export/e2e-hub/federated-search.spec.ts`, `export/CHANGELOG.md`, `plans/export-header-first-search.md` |
-**Order:** HP → `r14/two-grounds-headers` → S1. The shared files are the three changelogs'
-`[Unreleased]` sections and this record.
+**Order:** HP → `r14/two-grounds-headers` → HS (`r14/hidden-sites`) and S1 (`r14/first-search`),
+siblings. The shared files are the three changelogs' `[Unreleased]` sections and this record.
## Record
@@ -1004,6 +1005,223 @@ There the bar overflows and the wide row scrolls while the name shows: measured
1100 px at 150 % and from 779 to 1300 px at 200 %. With the browser's own text size `md` moves too
and nothing overflows. The `md` layout is slice HP's and unchanged here.
+### Slice HS, as shipped — hidden sites: a site that builds and deploys, and is listed nowhere (2026-09-29)
+
+Branch `r14/hidden-sites` off `main` `99d4d76a`, worktree `~/Projects/homepage-social-visible`
+(block #3: editor test 3311, export 3310, homepage e2e 3340, hub e2e 3341), one Opus implementer.
+Scratch files `hs-*` in the job's `tmp`. The rulings (2026-09-29, not re-opened):
+1. A hidden site is absent from the homepage (cards, chart, `/stats`), the hub (members, federated
+ search, `corpus.json`, `llms.txt`), every other site's related-sites footer, and the published
+ id lists (`channel-sites.json`, the pooled `stats/`).
+2. Its hours and transcripts are in no public total.
+3. The editor has one checkbox in each site's settings, and no homepage settings page.
+4. The hidden site itself builds and deploys exactly as before.
+
+| sha | what |
+|---|---|
+| `122b8879` | `common:` `site.json` `listed` (schema, docs, parser, writer; `SITE.md`); `isListedSite` and `channelsOnlyOnUnlistedSites`; the footer drops an unlisted sibling |
+| `77aa1456` | `common:` the homepage summary (v6), `channel-sites.json` and the whole-pool stats leave out an unlisted site and the channels only it exposes |
+| `cbd12a8b` | `common:` `hub-sites.json` (and through it the hub's `corpus.json` and `llms.txt`) lists listed sites only |
+| `5836d6ec` | `sites:` the checkbox, and `saveSiteAction` carries the key; `sites-crud` e2e |
+| `64ae9b47` | `homepage:` the e2e fixture's seventh, unlisted site; `unlisted-site.spec.ts` |
+| `30b3f486` | `plans:` this record; the three changelogs |
+| `e3ee2eb1` | `homepage:` the fixture's second list unlists every site in it; the spec proves the site is in the input (review L1, L2) |
+| `65675373` | `common:` a shared channel stays the listed site's when the unlisted id sorts first (review L3) |
+| `20ff9ecf` | `plans:` FACTS' naming hazards (L4); this release's slices table, Order and Rollout carry HS (L6) |
+| `7ea35dfb` | merge of `main` `ccf90892` (release 15 IG); `buildStats.ts` and `editor/CHANGELOG.md` merged clean, the HS bullet under `[Unreleased]` |
+| _this_ | `plans:` the commit table, the review and the post-merge gates |
+
+The six commits up to `30b3f486` were rewritten after the review for the release's commit trailer
+(`git filter-branch --msg-filter`, trees unchanged); their first shas were `5918ba27`, `10378f7f`,
+`1491a7ab`, `ca22f0e6`, `e12515c2`, `4eeb80ee`.
+
+- **The key:** `listed?: boolean` in `site.json`, after `siteUrl`, in the type, `SITE_FIELD_DOCS`,
+ `siteFieldsSchema` and `siteToDisk`. Absent or anything but `false` reads `true`; only `false` is
+ written (the `archives` idiom). `SITE.md` regenerated.
+- **One predicate.** `isListedSite(site)` (`site.listed !== false`) and
+ `channelsOnlyOnUnlistedSites(sites)` — the channels at least one site exposes and no listed site
+ does. Both live beside the key in `common/lib/siteSchema.ts` and are exported from `lib/site`
+ (its `export *`), like `isValidSiteId`: the summary builder is pure and imports them without
+ `lib/site`'s file I/O. Every filter below calls them.
+- **What each public output does now:**
+
+ | Output | Code | An unlisted site | A channel only unlisted sites expose |
+ |---|---|---|---|
+ | Homepage summary: `sites`, `official`, `monthly`, `series`, `recent`, `channels` | `buildHomepageSummary`, the public-site filter | absent | absent |
+ | Homepage summary: `totals`, `availability` | the same, over the in-scope stats | — | not counted |
+ | `channel-sites.json` | `channelSitesOf` (`controller/poolSummary.ts`) | absent from every list | absent |
+ | The homepage's `stats/` (whole-pool bundle) | `buildStats`, the whole-pool block | — | its records and its manifest entry absent |
+ | `hub-summary.json` | `toHubSummary` of the same summary | absent | not counted |
+ | `hub-sites.json`, the hub's `corpus.json`, `llms.txt` | `compose-hub.ts`, the built-in pool | absent | — |
+ | Every other site's footer | `resolveRelatedSites` | not linked, even from a featured group | — |
+
+ - A channel a listed site also exposes is credited to the listed site: `primarySiteOf` runs over
+ listed public sites only.
+ - The summary's version is 6. No field was added or removed; the number marks the scope change.
+- **What stays, and why:**
+ - The unlisted site's own build and deploy: `buildAll` and the deploy paths
+ (`common/publish/build.ts`), its per-site stats and summaries (`buildStats`, `buildIndex`), its
+ archives and chart templates, its own `/site.json`, `corpus.json`, `llms.txt`, sitemap. Its own
+ footer still lists its listed siblings.
+ - The editor's pages (`/sites`, the channel page's memberships, the nav's site switcher, the
+ priority focus), which list every site.
+ - `siteChannelIndex`, `renameChannel`, `migrateToSites`, `listSiteIds` for the CLI's
+ site argument, the export's default site: none is a public list.
+ - The per-site staging in `buildIndex.ts` (another slice's file, and per site).
+- **Editor.** The site form has **List on the Archilyzer homepage and hub** after Public URL, on
+ by default, with a hint. `saveSiteAction` rebuilds the `Site` from the form; it now carries
+ `listed: false`, so a save of any other field keeps it. The other writers (`writeSite` from the
+ channel page, `renameChannel`, `migrateToSites`) spread the stored site or create a new one.
+- **Tests:**
+ - `siteSchema.test.ts`: absent and `true` read listed, only `false` unlists and only `false` is
+ written, through `writeSite`/`getSite`; the "everything" round-trip fixture carries
+ `listed: false`; `channelsOnlyOnUnlistedSites` keeps a shared channel with the listed site; the
+ footer drops an unlisted sibling named in a featured group, and an unlisted site's footer
+ still lists the rest.
+ - `homepageSummary.test.ts`: an unlisted site with its own channel and one shared with a listed
+ site gives a summary deep-equal to the one without it (every array, every total), and its id,
+ title and channel appear nowhere in the JSON; an unlisted site with no `siteUrl`, and one whose
+ channels are all shared, change nothing either; `listed: true` equals no key.
+ - `buildStats.test.ts` (k): the whole-pool bundle leaves out the unlisted-only channel (records,
+ manifest, count, log line) and keeps a pool-only one; the unlisted site's own bundle keeps all
+ three of its videos.
+ - `poolSummary.test.ts` (new): `channelSitesOf` names listed sites only.
+ - `compose-hub.test.ts`: `hub-sites.json`, `corpus.json` and `llms.txt` name the listed site and
+ not the unlisted one.
+ - Editor `sites-crud`: a `listed: false` file opens unticked; a save that changes only the title
+ keeps `false`; ticking removes the key; unticking writes it again.
+ - Homepage `unlisted-site.spec.ts`: the builder's input holds the unlisted site (`listed:
+ false`, its own records, the shared channel), and listed it would add a card and its records;
+ the summary the dev server reads equals the one built without it; `/` and `/stats/` (served
+ HTML and DOM) name all six listed sites and not the unlisted one.
+ - `homepageSummary.test.ts`, after the review: an unlisted id that sorts BEFORE the listed one
+ (`aaa-hidden` < `beta`) still leaves the shared channel with the listed site.
+- **The homepage e2e fixture** (`homepage/e2e/fixture-summary.ts`), for the slices that build on
+ it:
+ - `FIXTURE_SITES`: the six listed sites, unchanged.
+ - `FIXTURE_UNLISTED_SITE`: `fixture-unlisted`, "Fixture Unlisted", two channels of its own
+ (`fixture-unlisted-ch1/2`, six a day, like the rest); it also exposes `fixture-one-ch1`.
+ - `buildFixtureInputs(fixtureSites = FIXTURE_SITES, unlistedSites = [FIXTURE_UNLISTED_SITE])`
+ returns the builder's inputs (`stats`, `channelSites`, `sites`). The first list is listed;
+ every site in the second is written with `listed: false`, whatever it carries (`FixtureSite`
+ has no `listed` field), and shares the first listed site's first channel.
+ - `buildFixtureSummary(…)` is the real builder over those inputs. The unlisted sites' records are
+ generated last, after the megaspike, so every listed record is the same with them or without,
+ and `buildFixtureSummary()` equals `buildFixtureSummary(FIXTURE_SITES, [])`. Passed in the
+ FIRST list, the same site is listed: 7 sites and 39,599 transcripts instead of 6 and 37,199.
+
+#### Proof: a hidden fixture site through the real builds
+
+A throwaway corpus in the job's scratch dir (`$T/hs-proof/`, `make-corpus.mjs`): three channels
+of three captioned videos each, and two sites with public URLs — `fixture-listed` (the listed
+channel and the shared one) and `fixture-unlisted` (`listed: false`; the shared channel and one of
+its own). Every path the builds write was pinned there (`TRANSCRIPTS_DIR`, `SETTINGS_FILE`,
+`EXPORT_INDEX_DIR`, …) except the two apps' own `public/` and `out/`. The worktree's own gitignored
+`homepage/public` data and `export/out` were set aside first and put back after; the
+`export/public` links were dropped (never their targets) and re-seeded from the primary after. The
+primary's `export/public` was untouched (no entry newer than the slice's start). Each build was
+capped at 5 GB with no swap.
+
+| Build | Result | Time | Max RSS |
+|---|---|---|---|
+| `archilyzer index` | 0 | 4 s | — |
+| `archilyzer build homepage --no-source` | 0 | 20 s | 776 MB |
+| `archilyzer build hub` | 0 | 34 s | 1,005 MB |
+| `archilyzer build site fixture-unlisted --skip-archives` | 0 | 41 s | 962 MB |
+| `archilyzer build site fixture-listed --skip-archives` | 0 | 84 s | 970 MB |
+
+Counts (files holding the string / occurrences, `grep -rF`):
+
+| Tree | `fixture-unlisted` | its title | its own channel's slug | its own channel's name | `fixture-listed` |
+|---|---|---|---|---|---|
+| `homepage/public` (summary, `channel-sites.json`, `stats/`) | 0 / 0 | 0 / 0 | 0 / 0 | 0 / 0 | 2 / 25 |
+| `homepage/out` | 0 / 0 | 0 / 0 | 0 / 0 | 0 / 0 | 10 / 145 |
+| `export/public` (hub compose) | 0 / 0 | 0 / 0 | 0 / 0 | 0 / 0 | 4 / 10 |
+| `export/out` (hub) | 0 / 0 | 0 / 0 | 0 / 0 | 0 / 0 | 4 / 10 |
+| `export/out` (the listed site) | 0 / 0 | 0 / 0 | 0 / 0 | — | 9 / 30 (its URL) |
+
+- The compose lines: `compose-homepage: 2 channel(s) mapped across 1 listed site(s) (1 unlisted
+ left out); summary covers 6 transcription(s) / 6 download(s) across 1 public site(s)` and
+ `compose-hub: 1 built-in pool site(s) …; hub-summary.json covers 1 official instance(s)`.
+- The summary: version 6; `totals` 6 transcripts, 6 downloads, 1 site, 2 channels, 6 hours;
+ `official` the same; `availability.counted` 6. With the unlisted site counted they would have
+ been 9 transcripts and 9 hours.
+- The pooled `stats/` manifest: 6 records, channels `proof-listed-channel` and
+ `proof-shared-channel`. `channel-sites.json` maps both to `fixture-listed` alone.
+- **The unlisted site still builds:** its own stats bundle holds all six of its videos (both its
+ channels); its build ships its own pages (its id in 17 files), and its footer links
+ `https://fixture-listed.example`. The listed site's footer links nothing: its only sibling is
+ unlisted.
+
+#### Gates (at `64ae9b47`, the tree of the first `e12515c2`; logs `$T/hs-*.log`)
+
+- **tsc** was clean before every commit (69 s, 33 s, 44 s — the last over the tip's code).
+- **Unit:**
+
+ | Suite | Result |
+ |---|---|
+ | common | 2,218/2,218 (8 new) |
+ | editor unit | 87/87 |
+ | homepage unit | 12/12 |
+ | `test:scripts` | 191 passed, 1 skipped (192) |
+ | mcp | 271/271 |
+
+- **Docs:** `docs files --check`, `settings example --check` and `docs env --check` all exit **0**.
+- **e2e**, each detached and queued:
+
+ | Suite | Passed | Failed | Time |
+ |---|---|---|---|
+ | homepage, full (the new `unlisted-site.spec.ts` 3) | 97 | 0 | 3.6 min |
+ | hub, full | 36 | 0 | 1.4 min (after 3.5 min in the queue) |
+ | editor: `sites-crud` (the new listed round-trip 1) | 15 | 0 | 0.9 min (after 11 min in the queue) |
+
+- **Builds:** the five above, in the proof. The editor's and umtool's `next build` were not run (no
+ route or bundled path changed; tsc covers the form and the action).
+- **Numbers tool:** none.
+- **After the review and the merge of `main` (at `7ea35dfb`; `$T/hs-gates3.log`):** tsc clean
+ (183 s, the machine under load); common 2,229/2,229 (release 15 IG's 2,220, this slice's 8 and
+ the review's 1); homepage unit 12/12; homepage e2e `unlisted-site.spec.ts` 3 passed, 0 failed
+ (19 s). The fixture's second list unlisting a site that carries no key, and the listed variant's
+ 7 sites / 39,599 transcripts, were checked by a script (`$T/hs-fixture-check2.ts`).
+
+#### Review (verdict SHIP; `$T/hs-review.md`)
+
+| Finding | Fix |
+|---|---|
+| L1: the fixture's second list relied on each site's own `listed: false` | `e3ee2eb1`: `buildFixtureInputs` writes `listed: false` on every site in it; `FixtureSite` has no `listed` |
+| L2: the spec's first case would pass if the builder ignored the unlisted site | `e3ee2eb1`: it asserts the site is in the input (unlisted, its records, the shared channel) and that, listed, it adds a card and exactly its own records |
+| L3: every unlisted id sorted after the listed one | `65675373`: `aaa-hidden` sorts before `beta`; the summary is still the one without it |
+| L4: `unlisted` / `listed` already mean other things | `20ff9ecf`: a FACTS "Naming hazards" table of the three |
+| L5: the listed site's out not searched for the own channel's name; the export site suite and `e2e:2origin` not run | left: both changes are no-ops for a site without the key, and the unit test and the real build's footer cover them |
+| L6: this record's slices table, Order and Rollout did not name HS; the hub check's N | `20ff9ecf`: HS row and Order; the Rollout names HS, counts public LISTED sites and checks the new box |
+
+#### Found and left
+
+- **The unlisted site's own `/site.json` and `/corpus.json` still carry its `hubUrl`** (ruling 4:
+ it deploys exactly as before). A visitor who adds its origin to the hub by hand gets it as any
+ added origin, and the hub can read that `hubUrl` as a family member's.
+- **`listed` / `unlisted` mean three things:** a video's visibility, `useHubSites`' `listed`
+ flag (the hub's list has loaded), and a site the family lists. FACTS' "Naming hazards" has the
+ table since the review.
+- **The Rollout's hub check** ("`hub-summary.json covers N official instance(s)`") counts public
+ LISTED sites; the Rollout says so since the review.
+- **Unlisting takes effect at the next builds.** The homepage, the hub and every other site are
+ static: until each is rebuilt and deployed, it still lists the site.
+- **The editor's `/sites` list** shows no marker for an unlisted site; the form's checkbox is the
+ one place (ruling 3).
+
+#### Decisions the operator could overturn
+
+| What I assumed | The alternative |
+|---|---|
+| `totals` and `availability` keep counting pool-only channels and channels of sites with no public URL, as before; only a channel exposed by unlisted sites alone leaves them | `totals` count the listed public sites only, the same scope as `official` |
+| A channel an unlisted site shares with a listed site with NO public URL is not "only on unlisted sites", so `totals` count it | count a channel only when a listed PUBLIC site exposes it |
+| The pooled `stats/` keep pool-only channels, as before | the pooled `stats/` hold only channels a listed public site exposes |
+| `isListedSite` lives beside the key in `siteSchema.ts`, exported from `lib/site` | define it in `lib/site.ts` itself, and have the summary builder import `lib/site`'s file I/O |
+| The summary's version is 6 | stay at 5 (no field changed) |
+| An unlisted site's own footer still links its listed siblings | an unlisted site shows no related-sites footer |
+| The checkbox sits after Public URL, with a hint naming what it removes | at the end of the form, or without a hint |
+
### Slice S1, as shipped — a clear screen until the first Search (2026-09-29)
Branch `r14/first-search` off `main` `99d4d76a` (`r14/two-grounds-headers` merged), worktree
@@ -1194,10 +1412,10 @@ The fix changes only which flag each place sets, so the full suites were not re-
## Rollout
-Release 14 is slice HP (merged, `bfa1ff3c`) and `r14/two-grounds-headers` (after the parent's
-merge). Every command below is typed **from the primary checkout's root**. There is no
-`archilyzer` on PATH, so it is `pnpm archilyzer …`. The command forms are the ones verified in
-`plans/stats-cache-key.md`'s rollout.
+Release 14 is slice HP (merged, `bfa1ff3c`), `r14/two-grounds-headers` and slice HS
+(`r14/hidden-sites`), each after the parent's merge. Every command below is typed **from the
+primary checkout's root**. There is no `archilyzer` on PATH, so it is `pnpm archilyzer …`. The
+command forms are the ones verified in `plans/stats-cache-key.md`'s rollout.
**Preconditions.**
1. `main` carries `r14/two-grounds-headers`.
@@ -1226,8 +1444,9 @@ hub goes before the sites, because in basic mode they share `export/out`. The si
- `pnpm ops build-hub --wait`, then `pnpm ops deploy-hub --wait`.
**Between the two,** the build's `compose-hub: …` line must end with `hub-summary.json covers
- N official instance(s)`, where N is the number of public sites. If it says `hub-summary.json
- skipped: …`, stop and fix what it names.
+ N official instance(s)`, where N is the number of public LISTED sites (a site whose
+ **List on the Archilyzer homepage and hub** is unticked is not counted). If it says
+ `hub-summary.json skipped: …`, stop and fix what it names.
4. **The six sites:** `pnpm ops build-deploy --json '{"all":true}' --wait`.
**Live checks.**
@@ -1245,6 +1464,9 @@ hub goes before the sites, because in basic mode they share `export/out`. The si
and `localStorage.getItem("ytdlp-tb:base")` now reads `"light"`.
- The homepage's growth chart has no slash in the page colour through any band. `/stats` in
Area mode has coloured top lines.
+- Every site's settings in the editor show **List on the Archilyzer homepage and hub**, ticked. With
+ none unticked, the homepage, the hub and every footer list the same sites as before, and
+ `https://archilyzer.pages.dev/homepage-summary.json` reads `"version":6`.
- **S1, a clear screen until the first Search**, is in every site's build and the hub's, so steps 3
and 4 above (every site and the hub rebuilt and deployed) carry it. On one site:
- a plain visit to `/` shows the search bar, the transcript count and the whole footer, with no
diff --git a/plans/release-15.md b/plans/release-15.md
@@ -0,0 +1,219 @@
+# Release 15 — storage and build hardening
+
+`main` at `99d4d76a` (release 14's T1, H1 and H2 merged). No plan file of its own: each slice's
+prompt carries its ruling, and this record carries what was built. Rules:
+`plans/tools/implementer-rules.md`, with the commit trailer this release's prompts give.
+
+**The standing choices** (not re-opened):
+- **An unmounted drive is not an empty channel,** for any build that walks the pool. The stats
+ build already holds such a channel ([`stats-cache-key.md`](stats-cache-key.md)); the index
+ build gets the same hold here.
+- **A slice that needs another slice's file stops and says so**; it does not edit it.
+- **Nothing is edited in the primary checkout**; each slice has its own worktree, and the parent
+ merges with `git merge --no-ff` only on a clean tree.
+
+## The slices
+
+| Slice | Branch | What | Owns |
+|---|---|---|---|
+| IG | `r15/index-hold` | The index build holds an unreachable channel instead of emptying it | `common/controller/buildIndex.ts` + new `buildIndex.test.ts`, `common/controller/buildStats.ts` (the hold's words move to a shared module), new `common/lib/channelMediaHold.ts`, `common/lib/envVars.ts`, `ENVIRONMENT.md`; records: `plans/{STATE,FACTS,stats-cache-key}.md` |
+| DS | `r15/drive-stall` | A stalled drive does not stop the editor answering | per its prompt |
+| UT | `r15/umtool-trace` | umtool's build stops tracing the whole `umtool/` folder | per its prompt |
+
+**Order:** IG → DS. DS adds a health gate inside `inspectChannelMedia`, which IG's hold calls
+through its public signature. UT is independent. The shared files are `editor/CHANGELOG.md`'s
+`[Unreleased]` and this record.
+
+## Record
+
+### Slice IG, as shipped — the index build holds an unreachable channel (2026-09-29)
+
+Branch `r15/index-hold` off `main` `99d4d76a`, worktree `~/Projects/r12-paths-fix` (block #12:
+editor 4201, test 4211, export 4210), one Opus implementer. Scratch files `ig-*` in the job's
+`tmp`. The ruling: an index build meets a channel it cannot read the way the stats build already
+does. It holds the channel instead of reading it as empty, and a full rebuild with one held refuses
+unless `ARCHILYZER_INDEX_ALLOW_HELD=1`.
+
+**What was wrong.** `scanSource` read `data/` with a bare `catch { continue }`. A relocated
+channel whose drive was unmounted (a dangling `data/` link) therefore contributed no videos. The
+removal pass then dropped every record the channel had, the page writers rewrote its shared
+transcript tree empty and removed its subs tree, and the next site build published the channel as
+gone. Only the `Diff: … -R removed` line showed it.
+
+- **The hold** (`common/controller/buildIndex.ts`):
+ - `scanSource` asks `inspectChannelMedia({ channelsDir }, slug, cfg)` for every channel that is
+ not excluded and not social, before it reads `data/`.
+ - Any status but `ok` or `in-place` holds the channel, with a reason and its storage
+ location's label, and no path.
+ - It asks again after the walk, so a drive that goes away mid-walk holds the channel instead of
+ dropping the videos after that point.
+ - A `readdir(data/)` that fails holds the channel too: `its data directory could not be read
+ (<code>)`.
+ - The exception is ENOENT on an `in-place` channel: a channel with nothing downloaded (or its
+ media deleted), which is emptied as before. The log now says it: `Channel <slug>: no data/
+ directory; indexed as a channel with no videos.`
+ - A per-video metadata `stat` failing with anything but ENOENT or ENOTDIR holds the channel,
+ logged as `a video in its data directory could not be read (<code>)`.
+ - **A held channel keeps everything:**
+ - its `mtimes` records, since the removal pass skips its keys, and so its `sums`, `cues`,
+ `subs`, `digests` and `byChannel` entries;
+ - its shared transcript, subs and digest trees: not rewritten, not pruned, and not removed by
+ the top-level cleanups;
+ - its subs and digest stats, carried from their sub-DBs, so the per-site manifests still list
+ it;
+ - its availability states. The maybe-missing overlay skips it, since its `availability.json`
+ files are on the missing drive and a missing one reads as "maybe missing". The last build's
+ `videoState` entries for it are carried over.
+ - The per-site summaries are built from LMDB, so the sites built next still list its videos.
+ - **A curated-tag change while held:** the re-apply pass re-derives the held channel's records
+ in LMDB as usual, but its pages are not written. The pages-pending flag is therefore kept (and
+ logged) while any channel is held, and the first build with the drive back writes them.
+- **A full rebuild refuses** (a schema change, or a first build with no index).
+ - The scan now runs BEFORE the clear, so a refusal leaves the index untouched. `scanStartedAt`
+ is still taken at the scan's start.
+ - The message names each channel with its location's label and says why a clear would publish
+ it as gone. It gives the ways out, mounting first (the same words as the stats build's
+ refusal), then the override by name and where it is set: the command's own environment for a
+ CLI run, and the editor's own environment (which takes a restart) for its Build index job or a
+ site build started from it. Their children inherit the editor's `process.env`.
+ - With `ARCHILYZER_INDEX_ALLOW_HELD=1` (1/true/yes/on, declared in `envVars.ts`, `ENVIRONMENT.md`
+ regenerated), the build proceeds. The held channel's records go with the clear; they cannot be
+ carried across a format change. Its shared trees are left on disk, and it is out of the index
+ until its media is back and an index build runs.
+- **The words are shared:** new `common/lib/channelMediaHold.ts` (`isMediaHeld`, `HELD_REASON`,
+ `heldReason`, `describeHeld`, `HELD_WAYS_OUT`). `buildStats.ts` uses it, and its messages are
+ byte-identical (its case (i) passes unchanged).
+- **The result and the log:** `BuildIndexResult.heldChannels`. The log carries one line per held
+ channel (`Channel <slug>: <why>; its N indexed video(s) are kept as they are, not rescanned, and
+ its transcript, subtitle and digest pages are left as they are.`). The `Diff:` line ends in
+ ` Held: N channel(s), K video(s) kept.` and the `Done in` line in ` Held, their media not
+ readable: <slugs>.` Both prefixes are unchanged: the e2e helper waits for `Done`.
+
+**Who reports it** (every caller of `buildIndex`):
+
+| Caller | What it shows |
+|---|---|
+| `pnpm archilyzer index` (also the root, export and homepage `build:index` scripts) | The log on stdout, with the three lines above. It exits 0 on a hold. On the refusal it exits 1 with the message on stderr (`runIfEntryPoint`). `common/bin/build-index.ts` is unchanged: it prints the log and discards the result. |
+| A site build's data phase (`build:data`, from `buildSite`'s steps and `buildAll`'s Phase A in `common/publish/build.ts`) | The same lines, in the site build's job log. A refusal stops the steps at the data phase (`Build failed (exit 1)` / `Data phase failed (exit 1)`), so nothing is composed or deployed from an emptied index. The hub build does not run the index. |
+| The editor's **Build index** job (`buildIndexAction`, `editor/app/sites/lib/buildAction.ts`) | The lines in the job's log on `/sites` and `/jobs`. A refusal fails the job with the message. The action discards the result, and it was left unchanged: the log already carries every held channel, and `editor/app/sites/**` belongs to another slice this release. |
+| `pnpm ops build-index --wait` | Follows the job's log, so it prints the same lines. |
+| The tests | `heldChannels`. |
+
+**Commits**
+
+| Commit | What |
+|---|---|
+| `c6ae51b0` | `plans:` this record: the header, the slices, and empty Record and Rollout sections. |
+| `51328098` | `common:` the hold's words move to `lib/channelMediaHold.ts`; the stats build uses them, with its messages unchanged. |
+| `ffb01d8e` | `common:` the index build's hold, the refusal and its override, `heldChannels`, the log lines; `envVars.ts` + `ENVIRONMENT.md`; new `buildIndex.test.ts` (9 cases). |
+| `702cd0da` | `plans:` this section; FACTS "The index build's hold" (and the two stale statements in "The stats cache key" corrected); the STATE follow-up closed; `stats-cache-key.md` "Left" marked closed; the editor changelog. |
+| `9ec48351` | `common:` review L3 + L5: one unreadable video directory is logged apart from an unreadable `data/`; the refusal says where the override is set. |
+| `5af516b7` | `common(test):` review M1: case (i), the drive lost mid-walk. |
+| this commit | `plans:` the review's findings to their commits; FACTS L1, L2 and the anchors; the changelog's override sentence (L5). |
+
+**Tests** (`common/controller/buildIndex.test.ts`, the real `buildIndex` over a temp corpus). The
+drive channel is seeded with `buildStats.test.ts`'s `seedDriveChannel` shape and unmounted by
+renaming its media root away. Every path is pinned under a temp root, and case (z) spies on
+node:fs writes.
+
+| Case | What it pins | On the pre-change `buildIndex.ts` |
+|---|---|---|
+| (a) | Unmounted, incremental build with another change: records kept, the drive's transcript and subs trees byte-identical (manifest included), the site still lists both videos and its subs count; the log lines, with the label and no path | removed 2, not 0 |
+| (b) | Availability carried: `maybe_missing` stays and a post-scan confirmation stays `available`; `videoState` unchanged | d1 no longer published |
+| (c) | A full rebuild refuses (message, ways out, the variable, no path); the index and pages untouched; the CLI exits non-zero; with the override it holds (records cleared, pages kept); the drive back re-adds both | no refusal |
+| (d) | A first build (no index) with a channel held refuses too | no refusal |
+| (e) | The drive back: held set empty, a video added to the drive meanwhile indexed, 0 changed | removed 2 while away |
+| (f) | A really empty in-place channel (an empty `data/`, and no `data/`) is emptied, not held; the missing `data/` is logged | the new log line only (the emptying matched, as it should) |
+| (g) | An unreadable `data/`, and an unreadable video dir mid-walk (mode 000), hold their channels with `EACCES` | removed 3 |
+| (h) | A tag rule added while held: the held pages untouched, the flag kept; the drive back writes the tag to them and clears the flag | the drive's pages emptied |
+| (i) | The drive lost MID-WALK: `node:fs/promises` `stat` unmounts it right after the walk's first drive metadata stat, so every later stat in the channel is ENOENT. The second look holds the channel: 0 removed, records and pages unchanged, the log line | removed 3 of 4; the same with only the second look deleted from the new code |
+| (z) | No write outside the temp root | passes on both |
+
+The pre-change column was run with the old `buildIndex.ts` swapped in once, with the
+`heldChannels` assertions removed so each case reached its first substantive assertion.
+
+#### Gates (at `ffb01d8e`, and after the review at `5af516b7`; logs `$T/ig-*.log`)
+
+- **tsc** was clean before every commit: 78 s at the branch point, 48 s at `ffb01d8e`, 50 s at
+ `5af516b7`.
+- **Unit:**
+
+ | Suite | Result |
+ |---|---|
+ | common | 2,219/2,219 at `ffb01d8e` (the branch point's 2,210 plus the 9 new cases), 77 s; **2,220/2,220** at `5af516b7` (case (i) added), 78 s |
+ | editor unit | 87/87 |
+ | `test:scripts` | 191 passed, 1 skipped (192) |
+ | mcp | 271/271 |
+
+- **Docs:** `docs env --check`, `docs files --check` and `settings example --check` all exit **0**,
+ at both points.
+- **Build:** the editor's `next build`, with the primary's `transcripts/` linked in and capped at
+ 5 GB with no swap: 64 s, max RSS 1,642 MB. The link was removed after the build, and nothing
+ ran through it.
+- **e2e** (editor, detached and queued; the spec list is every spec that runs Build index:
+ `availability`, `build`, `channel-build-toggle`, `chat-only`, `duplicate-shorts`, `jobs`,
+ `regional-vtt-fallback`, `tags`, passed as `e2e/<name>.spec.ts` so `availability` does not also
+ match `pre-clean-availability`): **35 passed, 0 failed, 3.6 min**, after 1 min 45 s in the
+ queue. No e2e fixture has an unreachable channel, so these confirm the hold changes nothing for a
+ readable corpus. Not rerun after the review: its two log-wording changes touch no spec (`git grep
+ "could not be read" editor/e2e` finds only the curated-tag preview and the title filter).
+- **Numbers tool:** none.
+
+#### Found and left
+
+- **A drive that drops during the processing phase** (after the scan) is not covered, for a changed
+ video whose metadata was read before the drop. Its transcript read is caught as "no cues", so its
+ cues are removed. A failed sub-track read is skipped, and with none left its subs are removed. A
+ failed digest load leaves no digest, which is removed. `mtimes` is then written with the video's
+ current mtimes, so the loss lasts until any of its tracked mtimes (metadata, transcript, subs,
+ availability, digest) moves. These are per-video reads inside the worker, left to the slice that
+ handles drive stalls.
+- **A drive that drops and comes back inside one walk** is not covered either: the second look
+ finds it, and the videos skipped in between are removed.
+- **`inspectChannelMedia` is asked twice per channel** (before and after the walk), three syscalls
+ each today. Slice DS puts a health gate inside it, and that gate runs twice per channel per
+ index build.
+- **Under the override,** a held channel is listed on its sites with 0 videos, and its subs and
+ digest counts are left out of the site manifests. Its old shared page trees stay on disk, and a
+ compose copies them.
+- **While a channel is held after a tag change,** the pages-pending flag stays set. Every build
+ until the drive is back is then a full page walk; it skips unchanged pages by hash, so it rewrites
+ none.
+- **The digest tree's retention has no test.** The code path mirrors the subs tree's, and no
+ fixture carries a digest sidecar.
+- **The editor shows a hold only in the job log.** A `/sites` badge would be in `editor/app/sites/**`.
+
+#### Decisions the operator could overturn
+
+| What I assumed | The alternative |
+|---|---|
+| A held channel's shared pages are not written at all. **Ruled at review: keep the skip.** | Rewrite them from the kept records: byte-identical on an incremental build, and a tag change would reach them at once. After an override rebuild the records are gone, and a rewrite would publish the channel empty. |
+| Under the override, the held channel's records go with the clear. | Carry them across the clear. A schema change means the stored format moved, so the old records cannot be trusted. |
+| An unreadable `data/`, or a per-video `stat` failing with anything but ENOENT, holds the channel. | Hold only for the statuses `inspectChannelMedia` reports, and keep treating other read errors as "no videos", as the old code did. |
+| A channel with no `data/` is logged, one line per build. | Stay silent, as before; the ruling asked for a log. |
+| The CLI exits 0 on a hold, since the hold is the safe outcome; the refusal exits 1. **Ruled at review: both stay.** | Exit non-zero so scripts notice; a site build's data phase would then fail whenever a drive is out. |
+| The words live in a new `lib/channelMediaHold.ts` shared with the stats build. | Duplicate them in `buildIndex.ts` and leave `buildStats.ts` untouched. |
+
+#### Review
+
+**Verdict: SHIP AFTER FIXES** (`ig-review.md` in the job's scratch). No High. Every write path a
+held channel could reach was traced, and the refusal fires before anything is deleted.
+
+| Finding | Where |
+|---|---|
+| M1: the second look after the walk had no test | `5af516b7`: case (i). It fails with 3 of 4 removed when that look is deleted. |
+| L1: FACTS said an unreachable drive reaches `notIndexable` only through the override | this commit: a video that reached the drive after the last index build that could read it is counted there by a stats-only run between the drive's return and the next index build, which heals it. |
+| L2: the processing-phase gap also loses subs and digests, until any tracked mtime moves | this commit, in "Found and left" and FACTS. |
+| L3: one unreadable video dir was logged as the whole data directory | `9ec48351`: `a video in its data directory could not be read (<code>)`; case (g) expects it. |
+| L4 (optional): a narrower `HELD_REASON` type | **Left**, as the review allowed. The current contract already fails tsc on a new status. |
+| L5: the refusal did not say where the override is set | `9ec48351` (message, case (c)) and this commit (changelog). |
+| L6: check the held set before routine builds resume | A rollout note; the parent records it. |
+| L7: stale trees under the override | Already in "Found and left". |
+| Q2: the commit trailer | Ruled correct. |
+
+**What runs which code, for the rollout.** Every CLI command and every spawned data phase runs the
+checkout's code, so they hold from the moment `main` has this branch. The editor's in-process
+**Build index** button runs its built bundle, so it holds only after the editor is rebuilt and
+restarted.
+
+## Rollout
diff --git a/plans/stats-cache-key.md b/plans/stats-cache-key.md
@@ -273,6 +273,8 @@ PATH, so it is `pnpm archilyzer …`.
`mtimes`, cues and pages), or at least a refusal with an override.
- When: schedule it before routine builds resume after this rollout.
- Until then, the rollout's step 3 `Diff:` check is the safeguard.
+ - **Closed by release 15 slice IG** ([`release-15.md`](release-15.md), "Slice IG, as shipped"):
+ the hold, and a refusal with an override for a full rebuild.
- **A cross-process lock for builds** (see "Concurrent stats builds" above). The rule stands in
for it.
- **O5:** a manual English caption (no inline timing tags) indexes as 0 cues (FACTS).