commit 6adeee9256ad8ca3b0c0539904b76e8005349e90
parent 5af43e79c98a2245bb1f2ea28ce6299ac46fb984
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 2 Oct 2026 00:11:22 -0400
Merge branch 'main' into deck/finale-dip
# Conflicts:
# umtool/lib/report/onscreen.mjs
Diffstat:
86 files changed, 5517 insertions(+), 246 deletions(-)
diff --git a/.gitignore b/.gitignore
@@ -169,6 +169,8 @@ yarn-error.log*
# run alongside a dev server someone is judging clips in (Next refuses two for
# one project).
umtool/.e2e-song/
+# ...and the storage spec's media root, a sibling of it (UMTOOL_MEDIA_DIR).
+umtool/.e2e-song-media/
umtool/.next/
# Any alternate dist dir, not just the e2e one.
#
diff --git a/SETTINGS.md b/SETTINGS.md
@@ -160,7 +160,7 @@ Default: `"when-required"`
## `social`
-Per-platform settings of the social-post fetchers. Today one key: where the X fetchers' login comes from (`social.x.cookieSource`, chosen in the X account session section of /settings). See common/social/xCookieSource.ts.
+Per-platform settings of the social posts. Today two keys, both X's, both chosen in the X account session section of /settings: where the X fetchers' login comes from (`social.x.cookieSource`) and where X posts may appear (`social.x.visibility`). See common/social/xCookieSource.ts.
#### `social`
@@ -173,6 +173,7 @@ Per-platform settings of the social-post fetchers. Today one key: where the X fe
| Key | Default | Description |
|---|---|---|
| `cookieSource` | absent | Where the X fetchers' login comes from. `"browser"`: the operator's everyday browser, named by `cookiesFromBrowser` — gallery-dl is handed `--cookies-from-browser <spec>` and reads it on every run, and the Playwright fallback reads the same store (Firefox only; common/social/xBrowserLogin.ts), so the login lasts as long as the browser's. `"profile"`: the session broker's persistent profile ("Connect X account" on /settings) and the cookie jar it exports. ABSENT (the default) is resolved at read time, never stored: `"browser"` when `cookiesFromBrowser` is set and no profile is connected (no exported jar carrying an auth_token), else `"profile"`. The source, not `cookieMode`, governs the X fetchers. |
+| `visibility` | absent | Where X posts may appear. `"public"` (the default; absent): an X channel's posts are built into every site that has the channel. `"private"`: every X channel's posts (a channel with `sourceKind: "social"` and `platform: "twitter"`) are left out of every PUBLIC site build — the channel with them, since posts are all an X channel holds — and built only into PRIVATE sites (`site.json` `audience`). Nothing on disk changes and fetching does not; a site already published changes on its next build and deploy, and flipping back is a rebuild. Chosen in the X account session section of /settings; the rule is common/lib/postsVisibility.ts. |
Default:
diff --git a/SITE.md b/SITE.md
@@ -24,6 +24,7 @@ Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/f
| [`accent`](#accent) | absent |
| [`siteUrl`](#siteurl) | absent |
| [`listed`](#listed) | `true` |
+| [`audience`](#audience) | absent |
| [`relatedSites`](#relatedsites) | `[]` |
| [`pwa`](#pwa) | `false` |
| [`archives`](#archives) | `true` |
@@ -164,6 +165,12 @@ Whether the family lists this site. Opt-OUT: absent/true = listed, only an expli
Default: `true`
+## `audience`
+
+Who this site is built for. `"public"` (the default; absent) or `"private"`: the operator's own reading copy, built on this machine and never deployed — every deploy path (Build & deploy, Deploy, `archilyzer deploy site`, Build & deploy all, docker/publish-site.sh) refuses it before any upload, while a build without a deploy still works. A private site is never listed (as `listed: false`, whatever `listed` says), publishes no `hubUrl`, and its `/corpus.json` says `"audience": "private"`. Content kept from the public — X posts while `social.x.visibility` is `"private"` — is built only into private sites. Only `"private"` is written.
+
+Default: absent
+
## `relatedSites`
Pulls specific siblings to the front of the footer's cross-site list, in named groups. Siblings not named here fall into a trailing "Other sites" group. Absent/empty = one flat list of every sibling.
diff --git a/common/bin/compose-hub.test.ts b/common/bin/compose-hub.test.ts
@@ -16,6 +16,7 @@ import { tmpdir } from "node:os";
import path from "node:path";
import { getPaths, type Paths } from "../lib/paths";
import { main } from "./compose-hub";
+import { readGlobalAliases } from "../lib/aliasesStore";
// Run with:
// pnpm --filter yt-dlp-transcript-common test
@@ -170,3 +171,54 @@ test("an unlisted site is in none of the hub's files; a listed one is in each",
rmSync(root, { recursive: true, force: true });
}
});
+
+// Release 17 slice XP (the review's HIGH 1): a site's compose leaves its data
+// in public/ — a private site's X posts included — and the hub builds from
+// public/ next. compose-hub removes every per-site entry, through a link only
+// the link, and ships the global alias dictionary as its own.
+test("compose-hub removes a site's data from public/, a linked entry by its link only", async () => {
+ const root = mkdtempSync(path.join(tmpdir(), "compose-hub-"));
+ const log = console.log;
+ try {
+ const paths = fixturePaths(root);
+ const pub = paths.exportPublicDir;
+ // What a private site's compose leaves.
+ for (const tree of ["summaries", "transcripts", "subs", "digests", "stats", "archives"]) {
+ mkdirSync(path.join(pub, tree, "x"), { recursive: true });
+ writeFileSync(path.join(pub, tree, "x", "page-0000.json"), "[]");
+ }
+ for (const f of ["site.json", "tags.json", "duplicates.json", "chart-templates.json", "sitemap.xml"]) {
+ writeFileSync(path.join(pub, f), "{}");
+ }
+ writeFileSync(path.join(pub, "search-aliases.json"), JSON.stringify({ aliases: [{ site: 1 }] }));
+ // posts/ as a worktree has it: a link into the primary checkout.
+ const primaryPosts = path.join(root, "primary-public", "posts");
+ mkdirSync(path.join(primaryPosts, "jer-x"), { recursive: true });
+ writeFileSync(path.join(primaryPosts, "manifest.json"), '{"channels":[{"slug":"jer-x"}]}');
+ symlinkSync(primaryPosts, path.join(pub, "posts"));
+ // A checked-in static asset stays.
+ writeFileSync(path.join(pub, "globe.svg"), "<svg/>");
+ // No global dictionary file: the seeded defaults are the global one.
+ const hubPaths = { ...paths, globalAliasesFile: path.join(root, "no-aliases.json") };
+
+ console.log = () => {};
+ await main({ paths: hubPaths });
+ console.log = log;
+
+ for (const gone of [
+ "summaries", "transcripts", "subs", "posts", "digests", "stats", "archives",
+ "site.json", "tags.json", "duplicates.json", "chart-templates.json", "sitemap.xml",
+ ]) {
+ assert.ok(!existsSync(path.join(pub, gone)), `${gone} was removed`);
+ }
+ assert.ok(existsSync(path.join(primaryPosts, "manifest.json")), "the link's target is untouched");
+ assert.ok(existsSync(path.join(pub, "globe.svg")));
+ assert.ok(existsSync(path.join(pub, "hub-sites.json")));
+ const aliases = JSON.parse(readFileSync(path.join(pub, "search-aliases.json"), "utf8"));
+ assert.deepEqual(aliases, { aliases: readGlobalAliases(hubPaths).aliases });
+ assert.ok(!JSON.stringify(aliases).includes('"site"'), "not the site's aliases");
+ } finally {
+ console.log = log;
+ rmSync(root, { recursive: true, force: true });
+ }
+});
diff --git a/common/bin/compose-hub.ts b/common/bin/compose-hub.ts
@@ -11,6 +11,10 @@
// there is no index to walk
// public/_headers <- CORS for the hub's own served JSON
// public/sw.js <- the hub service worker (the hub always ships a PWA)
+// public/search-aliases.json <- the global alias dictionary
+//
+// and REMOVES every per-site entry a site's compose left in public/
+// (SITE_ONLY_PUBLIC_ENTRIES below): the hub holds no site's data.
//
// The hub's branding ("Archilyzer") is resolved at build/render time from the
// HomepageConfig (see export/app/lib/site.ts hubSite()), not composed here.
@@ -32,6 +36,7 @@ import {
import { HUB_CORS_PATHS, renderHeadersFile } from "../lib/archive/headers";
import { buildPoolSummary } from "../controller/poolSummary";
import { HUB_SUMMARY_FILE, toHubSummary } from "../lib/hubSummary";
+import { readGlobalAliases } from "../lib/aliasesStore";
import { runIfEntryPoint } from "./_cli";
import { writePublicFile } from "./_publicFile";
@@ -76,10 +81,45 @@ async function composeHubSummary(
}
}
+// THE HUB CARRIES NO SITE'S DATA (release 17 slice XP, the review's HIGH 1).
+// public/ is the one directory every site composes into in turn, and the hub
+// builds from it next: whatever the last site's compose left there — its data
+// trees and its per-site files — `next build` copied into the hub's out/, and
+// the hub deploy shipped it. The live hub served jeralyzer's posts manifest;
+// after a PRIVATE site's build it would have served every X post. So the hub's
+// compose removes every per-site entry first. A worktree's public/ entries are
+// links into the primary checkout: rm removes the link, never its target.
+export const SITE_ONLY_PUBLIC_ENTRIES: readonly string[] = [
+ "summaries",
+ "transcripts",
+ "subs",
+ "posts",
+ "digests",
+ "stats",
+ "archives",
+ "site.json",
+ "tags.json",
+ "duplicates.json",
+ "search-aliases.json",
+ "chart-templates.json",
+ "sitemap.xml",
+];
+
export async function main(opts: { paths?: Paths } = {}): Promise<void> {
const paths = opts.paths ?? getPaths();
const publicDir = paths.exportPublicDir;
+ for (const entry of SITE_ONLY_PUBLIC_ENTRIES) {
+ await rm(path.join(publicDir, entry), { recursive: true, force: true });
+ }
+ // The hub's own alias dictionary is the global one (no site's overrides):
+ // what a hub reader loads for hub-wide search (lib/archive/reader-hub.ts),
+ // where it used to get whichever site had composed last.
+ await writePublicFile(
+ path.join(publicDir, "search-aliases.json"),
+ JSON.stringify({ aliases: readGlobalAliases(paths).aliases }),
+ );
+
// Built-in pool: every configured site that publishes a public URL and is
// listed. An unlisted site (`listed: false`) still builds and deploys, but the
// hub does not list it: not a member, not in federated search, not in the
diff --git a/common/bin/compose-site.postsVisibility.test.ts b/common/bin/compose-site.postsVisibility.test.ts
@@ -0,0 +1,331 @@
+// Integration: X posts are private (release 17 slice XP), through the REAL
+// index build and the REAL site compose, over a temp corpus.
+//
+// One video channel, one X channel and one Bluesky channel, on two sites that
+// both have all three: `pub` (public) and `priv` (`audience: "private"`). With
+// `social.x.visibility` "private", the public site's build carries no X channel
+// at all — no posts manifest entry, no posts tree, no channel in its channel
+// list, site.json or corpus.json — while the private one carries everything
+// and says `"audience": "private"` with no hubUrl. Flipping the setting back
+// is a rebuild. The rule itself is lib/postsVisibility.ts (its own tests).
+//
+// The export e2e cannot show this: its data is route-mocked, never built by
+// buildIndex and compose. export/e2e/x-posts-private.spec.ts serves the two
+// posts manifests this file pins and checks what a visitor sees.
+//
+// Run with: node_modules/.bin/tsx --test common/bin/compose-site.postsVisibility.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+// Every path getPaths() can resolve to a place this file's code may write is
+// pinned under ROOT before anything calls it (buildIndex.test.ts's list).
+const ROOT = mkdtempSync(path.join(tmpdir(), "posts-visibility-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("../controller/buildIndex");
+const { writePosts } = await import("../lib/posts-server");
+const { main: composeSite } = await import("./compose-site");
+const { builtAudienceProblem, deployAudienceProblem } = await import("../lib/builtExport");
+
+const paths = getPaths();
+const VIDEOS = "vids";
+const X = "jer-x";
+const SKY = "jer-sky";
+const HUB = "https://hub.example.test";
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const readJson = <T>(file: string): T => JSON.parse(readFileSync(file, "utf8")) as T;
+
+// YouTube's rolling-caption shape (parseVtt keeps lines with inline timing).
+const VTT =
+ "WEBVTT\nKind: captions\nLanguage: en\n\n" +
+ "00:00:00.000 --> 00:00:05.000 align:start position:0%\n" +
+ "First<00:00:01.000><c> caption</c><00:00:02.000><c> line.</c>\n";
+
+function post(slug: string, id: string, platform: "twitter" | "bluesky", text: string) {
+ return {
+ id,
+ slug: `${slug}/${id}`,
+ channelSlug: slug,
+ author: slug,
+ createdAt: "2026-09-01T12:00:00.000Z",
+ uploadDate: "20260901",
+ text,
+ url: `https://example.test/${slug}/${id}`,
+ platform,
+ isReply: false,
+ isRepost: false,
+ links: [],
+ };
+}
+
+function writeSettings(visibility?: "public" | "private") {
+ writeJson(paths.settingsFile, {
+ homepageUrl: HUB,
+ ...(visibility ? { social: { x: { visibility } } } : {}),
+ });
+}
+
+async function seedCorpus() {
+ writeJson(path.join(paths.channelsDir, VIDEOS, "config.json"), {
+ handling: "youtube",
+ name: "Videos",
+ url: "https://www.youtube.com/@vids/videos",
+ });
+ const dir = path.join(paths.channelsDir, VIDEOS, "data", "v1");
+ writeJson(path.join(dir, "metadata.info.json"), {
+ id: "v1",
+ title: "Video one",
+ channel: VIDEOS,
+ upload_date: "20260601",
+ duration: 120,
+ webpage_url: "https://www.youtube.com/watch?v=v1",
+ extractor_key: "Youtube",
+ });
+ writeFileSync(path.join(dir, "transcript.en.vtt"), VTT);
+
+ writeJson(path.join(paths.channelsDir, X, "config.json"), {
+ handling: "youtube",
+ name: "Jer on X",
+ url: "https://x.com/jer",
+ sourceKind: "social",
+ platform: "twitter",
+ socialHandle: "jer",
+ });
+ await writePosts(path.join(paths.channelsDir, X), [
+ post(X, "1001", "twitter", "an x post"),
+ post(X, "1002", "twitter", "another x post"),
+ ]);
+ writeJson(path.join(paths.channelsDir, SKY, "config.json"), {
+ handling: "youtube",
+ name: "Jer on Bluesky",
+ url: "https://bsky.app/profile/jer.example",
+ sourceKind: "social",
+ platform: "bluesky",
+ socialHandle: "jer.example",
+ });
+ await writePosts(path.join(paths.channelsDir, SKY), [
+ post(SKY, "3kabc", "bluesky", "a bluesky post"),
+ ]);
+
+ const site = (siteId: string, extra: Record<string, unknown> = {}) =>
+ writeJson(path.join(paths.sitesDir, siteId, "site.json"), {
+ siteId,
+ siteTitle: siteId,
+ siteDescription: "fixture",
+ headerTitle: siteId,
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [VIDEOS, X, SKY].map((slug) => ({ slug, groupId: "default" })),
+ siteUrl: `https://${siteId}.example.test`,
+ archives: false,
+ ...extra,
+ });
+ site("pub");
+ site("priv", { audience: "private" });
+}
+
+const quiet = () => {};
+async function index() {
+ await buildIndex({ paths, onLog: quiet });
+}
+
+// What one site's compose put in public/.
+type Composed = {
+ postsManifest: { channels: { slug: string; platform: string; postCount: number }[]; totalCount: number };
+ postTrees: string[];
+ transcriptTrees: string[];
+ siteJson: { channels: { slug: string }[]; hubUrl?: string };
+ corpus: {
+ site: { id: string; hubUrl?: string; audience?: string };
+ channels: { slug: string; postCount?: number; manifests: { posts?: string } }[];
+ postScheme?: unknown;
+ totals: { channels: number };
+ };
+};
+async function compose(siteId: string): Promise<Composed> {
+ const log = console.log;
+ console.log = quiet;
+ try {
+ await composeSite({ siteId, paths });
+ } finally {
+ console.log = log;
+ }
+ const pub = paths.exportPublicDir;
+ const trees = (dir: string) =>
+ [VIDEOS, X, SKY].filter((slug) => existsSync(path.join(dir, slug)));
+ return {
+ postsManifest: readJson(path.join(pub, "posts", "manifest.json")),
+ postTrees: trees(path.join(pub, "posts")),
+ transcriptTrees: trees(path.join(pub, "transcripts")),
+ siteJson: readJson(path.join(pub, "site.json")),
+ corpus: readJson(path.join(pub, "corpus.json")),
+ };
+}
+
+const slugs = (xs: { slug: string }[]) => xs.map((c) => c.slug).sort();
+
+test("X private: a public site carries no X channel; a private site carries all of it and says so", async () => {
+ await seedCorpus();
+ writeSettings("private");
+ await index();
+
+ // The index build's per-site manifests, before any compose.
+ const sitePosts = (id: string) =>
+ readJson<Composed["postsManifest"]>(
+ path.join(paths.exportSitesIndexDir, id, "posts", "manifest.json"),
+ );
+ assert.deepEqual(slugs(sitePosts("pub").channels), [SKY]);
+ assert.deepEqual(slugs(sitePosts("priv").channels), [SKY, X]);
+ // The shared posts tree is corpus-wide and keeps X: the private site and
+ // the MCP over its build read it.
+ assert.ok(existsSync(path.join(paths.exportSharedPostsDir, X, "manifest.json")));
+
+ const pub = await compose("pub");
+ assert.deepEqual(slugs(pub.postsManifest.channels), [SKY]);
+ assert.equal(pub.postsManifest.totalCount, 1);
+ assert.deepEqual(pub.postTrees, [SKY]);
+ assert.deepEqual(pub.transcriptTrees, [VIDEOS, SKY]);
+ assert.deepEqual(slugs(pub.siteJson.channels), [SKY, VIDEOS]);
+ assert.deepEqual(slugs(pub.corpus.channels), [SKY, VIDEOS]);
+ assert.equal(pub.corpus.totals.channels, 2);
+ assert.equal(pub.corpus.site.audience, undefined);
+ assert.equal(pub.corpus.site.hubUrl, HUB);
+ assert.equal(pub.siteJson.hubUrl, HUB);
+ assert.equal(builtAudienceProblem(paths.exportPublicDir), null);
+ assert.equal(deployAudienceProblem({ siteId: "pub" }, paths.exportPublicDir), null);
+
+ const priv = await compose("priv");
+ assert.deepEqual(slugs(priv.postsManifest.channels), [SKY, X]);
+ assert.equal(priv.postsManifest.totalCount, 3);
+ assert.deepEqual(priv.postTrees, [X, SKY]);
+ assert.deepEqual(slugs(priv.corpus.channels), [SKY, X, VIDEOS]);
+ assert.equal(priv.corpus.channels.find((c) => c.slug === X)?.postCount, 2);
+ assert.ok(priv.corpus.channels.find((c) => c.slug === X)?.manifests.posts);
+ assert.equal(priv.corpus.site.audience, "private");
+ // A private site belongs under no hub.
+ assert.equal(priv.corpus.site.hubUrl, undefined);
+ assert.equal(priv.siteJson.hubUrl, undefined);
+ // And its bundle refuses to deploy — under its own id, and under another
+ // site's (a public site's identity on a private build is still refused).
+ assert.match(builtAudienceProblem(paths.exportPublicDir) ?? "", /private build of "priv"/);
+ assert.match(
+ deployAudienceProblem({ siteId: "priv", audience: "private" }, paths.exportPublicDir) ?? "",
+ /^Site "priv" is private \(audience: private\)/,
+ );
+
+ // Compose the public site again over the private one's public/, with
+ // nothing changed since its last compose: the X channel the private site
+ // carried is pruned from every tree, and the summaries are the public
+ // site's again — its compose cache is not trusted over another site's
+ // compose (the summaries used to be skipped as "unchanged" and shipped the
+ // private site's channel list).
+ const again = await compose("pub");
+ assert.deepEqual(again.postTrees, [SKY]);
+ assert.deepEqual(again.transcriptTrees, [VIDEOS, SKY]);
+ assert.deepEqual(slugs(again.siteJson.channels), [SKY, VIDEOS]);
+ assert.deepEqual(slugs(again.corpus.channels), [SKY, VIDEOS]);
+ assert.equal(again.corpus.site.audience, undefined);
+ // And over its own last compose the cache is trusted as before.
+ const third = await compose("pub");
+ assert.deepEqual(slugs(third.corpus.channels), [SKY, VIDEOS]);
+});
+
+test("a config compose cannot read does not ship the posts tree the index build withheld", async () => {
+ // The index build (X private) withheld X from pub's posts manifest; compose
+ // then fails to read X's config, so its own rule reads X as visible. The
+ // site posts manifest is the index build's word: no X posts tree ships.
+ writeSettings("private");
+ await index();
+ const cfg = path.join(paths.channelsDir, X, "config.json");
+ const saved = readFileSync(cfg, "utf8");
+ rmSync(cfg);
+ try {
+ const pub = await compose("pub");
+ assert.deepEqual(pub.postTrees, [SKY]);
+ assert.deepEqual(slugs(pub.postsManifest.channels), [SKY]);
+ assert.deepEqual(slugs(pub.corpus.channels), [SKY, VIDEOS]);
+ } finally {
+ writeFileSync(cfg, saved);
+ }
+});
+
+test("a compose over an index built before the setting flipped lists no X channel anywhere", async () => {
+ // Index with X public, then flip to private and compose WITHOUT indexing.
+ writeSettings("public");
+ await index();
+ writeSettings("private");
+ const pub = await compose("pub");
+ assert.deepEqual(pub.postTrees, [SKY]);
+ assert.deepEqual(slugs(pub.postsManifest.channels), [SKY]);
+ assert.equal(pub.postsManifest.totalCount, 1);
+ assert.equal(pub.corpus.channels.find((c) => c.slug === X)?.postCount, undefined);
+ assert.equal(pub.corpus.postScheme !== undefined, true, "the Bluesky posts are still advertised");
+ // The channel list too: no X channel in site.json or corpus.json.
+ assert.deepEqual(slugs(pub.siteJson.channels), [SKY, VIDEOS]);
+ assert.deepEqual(slugs(pub.corpus.channels), [SKY, VIDEOS]);
+});
+
+test("X public again: the next build puts X back on the public site", async () => {
+ writeSettings("public");
+ await index();
+ const pub = await compose("pub");
+ assert.deepEqual(slugs(pub.postsManifest.channels), [SKY, X]);
+ assert.deepEqual(pub.postTrees, [X, SKY]);
+ assert.deepEqual(slugs(pub.corpus.channels), [SKY, X, VIDEOS]);
+
+ // No setting at all is public too.
+ writeSettings();
+ await index();
+ assert.deepEqual(slugs((await compose("pub")).postsManifest.channels), [SKY, X]);
+});
+
+test("a public site whose only posts were X posts ships an empty posts manifest and no post scheme", async () => {
+ writeSettings("private");
+ writeJson(path.join(paths.sitesDir, "xonly", "site.json"), {
+ siteId: "xonly",
+ siteTitle: "xonly",
+ siteDescription: "fixture",
+ headerTitle: "xonly",
+ homeTagline: "",
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [VIDEOS, X].map((slug) => ({ slug, groupId: "default" })),
+ archives: false,
+ });
+ await index();
+ const xonly = await compose("xonly");
+ // SearchSessionContext's hasPostsCorpus is `channels.length > 0`: no Posts
+ // toggle on this site, with nothing special-cased.
+ assert.deepEqual(xonly.postsManifest.channels, []);
+ assert.deepEqual(xonly.postTrees, []);
+ assert.equal(xonly.corpus.postScheme, undefined);
+ assert.deepEqual(slugs(xonly.corpus.channels), [VIDEOS]);
+});
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -60,6 +60,10 @@ import {
type ArchiveManifest,
type ArchiveManifestEntry,
} from "../lib/archiveOptions";
+import { readChannelConfig } from "../controller/channels";
+import { builtSiteIdIn } from "../lib/builtExport";
+import { publishedMemberSlugs } from "../lib/postsVisibility";
+import { isPrivateSite } from "../lib/siteSchema";
import { runIfEntryPoint } from "./_cli";
import { copyPublicFile, ownDir, writePublicFile } from "./_publicFile";
@@ -98,7 +102,10 @@ async function emitFederationFiles(
// navigate the already-served paginated shards; they never enumerate per-video
// files, so the count is constant regardless of corpus size. Runs after
// site.json and the archives are composed (both feed into these files).
-async function emitAiFiles(paths: ReturnType<typeof getPaths>): Promise<void> {
+async function emitAiFiles(
+ site: Site,
+ paths: ReturnType<typeof getPaths>,
+): Promise<void> {
const sitePath = path.join(paths.exportPublicDir, "site.json");
if (!(await exists(sitePath))) return; // no composed data → nothing to describe
const descriptor = JSON.parse(
@@ -161,6 +168,9 @@ async function emitAiFiles(paths: ReturnType<typeof getPaths>): Promise<void> {
postCounts,
digestCounts,
hasTags,
+ // A private site says so in its own corpus.json — the bundle's word that
+ // the deploy guard (lib/builtExport.ts builtAudienceProblem) reads.
+ private: isPrivateSite(site),
});
await writePublicFile(
path.join(paths.exportPublicDir, "corpus.json"),
@@ -485,6 +495,34 @@ async function replaceDir(src: string, dest: string): Promise<void> {
}
}
+// Rewrite a served manifest whose `channels` name a slug outside `members`,
+// keeping the rest of it. A missing or unreadable manifest is left alone.
+// Answers whether it rewrote the file.
+async function narrowManifestChannels(file: string, members: Set<string>): Promise<boolean> {
+ let m: { channels?: { slug?: string }[] };
+ try {
+ m = JSON.parse(await readFile(file, "utf8"));
+ } catch {
+ return false;
+ }
+ const channels = m.channels ?? [];
+ const kept = channels.filter((c) => typeof c.slug !== "string" || members.has(c.slug));
+ if (kept.length === channels.length) return false;
+ await writePublicFile(file, JSON.stringify({ ...m, channels: kept }));
+ return true;
+}
+
+// The channel slugs a site posts manifest lists, or null when there is no
+// readable manifest.
+async function readPostsManifestSlugs(file: string): Promise<Set<string> | null> {
+ try {
+ const pm = JSON.parse(await readFile(file, "utf8")) as PostsManifest;
+ return new Set((pm.channels ?? []).map((c) => c.slug));
+ } catch {
+ return null;
+ }
+}
+
// --- Incremental compose cache ------------------------------------------------
// Per-site record of what we last materialized into public/, keyed by a cheap
// content signature of each source. When the signature is unchanged and the
@@ -665,11 +703,58 @@ export async function main(
}
const paths = opts.paths ?? getPaths();
const site = getSite(siteId, paths);
- const memberSlugs = site.channels.map((c) => c.slug);
+ // The members this site may publish (lib/postsVisibility.ts): every member,
+ // less an X channel while `social.x.visibility` is "private" and the site is
+ // public. The index build's per-site manifests are narrowed by the same rule;
+ // narrowing the trees here prunes an X channel a public site shipped before.
+ const configs = new Map(
+ await Promise.all(
+ site.channels.map(
+ async (c) => [c.slug, await readChannelConfig(paths, c.slug).catch(() => null)] as const,
+ ),
+ ),
+ );
+ const memberSlugs = publishedMemberSlugs(
+ site,
+ (slug) => configs.get(slug),
+ getSettings(),
+ );
+ const withheld = site.channels.length - memberSlugs.length;
+ if (withheld > 0) {
+ console.log(
+ `[compose] ${withheld} X channel(s) left out: X posts are private (social.x.visibility) and this site is public.`,
+ );
+ }
// Incremental compose: skip stages whose source is unchanged since last build.
+ //
+ // ONLY OVER THIS SITE'S OWN LAST COMPOSE. The cache is per site but public/
+ // is one directory every site composes into in turn (the basic build), so
+ // after another site's compose a skipped stage would ship THAT site's files —
+ // its summaries, its whole channel list — under this site's name: composing
+ // a private site and then a public one shipped the private site's summaries
+ // as the public site's. public/site.json names the site composed into it
+ // last (it is written below, every compose); another name, or none, and the
+ // site's own stages (summaries, stats, duplicates) are composed afresh.
+ //
+ // The per-channel tree signatures stay trusted: those trees are copies of
+ // the SHARED trees, the same bytes whichever site copied them, and a
+ // channel another site pruned is re-copied because its directory is gone.
const cachePath = composeCachePath(paths, siteId);
- const cache = await readComposeCache(cachePath);
+ const lastComposed = builtSiteIdIn(paths.exportPublicDir);
+ const cached = await readComposeCache(cachePath);
+ const cache: ComposeCache =
+ lastComposed === siteId
+ ? cached
+ : { ...cached, summaries: undefined, stats: undefined, duplicates: undefined };
+ if (lastComposed !== siteId && lastComposed !== null) {
+ console.log(
+ `[compose] public/ was last composed for "${lastComposed}": composing ${siteId}'s summaries, stats and duplicates afresh.`,
+ );
+ }
+ // Until this compose writes its own, public/ names no site: a compose cut
+ // short part-way leaves the next one nothing to trust.
+ await rm(path.join(paths.exportPublicDir, "site.json"), { force: true });
// --- per-site aggregates (whole-dir swaps), gated on the source signature ---
const summariesSrc = path.join(paths.exportSitesIndexDir, siteId, "summaries");
@@ -683,6 +768,19 @@ export async function main(
} else {
console.log("[compose] summaries: unchanged.");
}
+ // The served summaries manifest lists only the members this compose
+ // publishes (lib/postsVisibility.ts), even over an index built before
+ // `social.x.visibility` flipped: site.json and corpus.json are built from it.
+ // A narrowed copy is no longer the source's copy: the next compose copies
+ // the summaries again rather than trusting it.
+ if (
+ await narrowManifestChannels(
+ path.join(paths.exportSummariesDir, "manifest.json"),
+ new Set(memberSlugs),
+ )
+ ) {
+ cache.summaries = undefined;
+ }
const statsSrc = path.join(paths.exportSitesIndexDir, siteId, "stats");
const statsSig = await dirSignature(statsSrc);
if (cache.stats !== statsSig || !(await exists(paths.exportStatsDir))) {
@@ -713,11 +811,20 @@ export async function main(
// The social-post corpus: same shared-tree shape, same incremental reconcile.
// Only social member channels have a source dir; reconcileChannelTree treats a
// missing one as "nothing to copy", so passing every member slug is correct.
+ //
+ // NEVER A TREE THE SITE'S POSTS MANIFEST DOES NOT LIST. The index build wrote
+ // that manifest by the same visibility rule as memberSlugs above, so the two
+ // agree — unless a channel's config could not be read here (it then reads as
+ // visible): the manifest is the index build's word, and a tree it withheld is
+ // not shipped on a failed read. A site with no manifest yet keeps the old rule.
+ const postsListed = await readPostsManifestSlugs(
+ path.join(paths.exportSitesIndexDir, siteId, "posts", "manifest.json"),
+ );
cache.posts = await reconcileChannelTree(
"posts",
paths.exportSharedPostsDir,
paths.exportPostsDir,
- memberSlugs,
+ postsListed ? memberSlugs.filter((slug) => postsListed.has(slug)) : memberSlugs,
cache.posts ?? {},
console.log,
);
@@ -752,7 +859,21 @@ export async function main(
);
if (await exists(postsManifestSrc)) {
await ownDir(paths.exportPostsDir);
- await copyPublicFile(postsManifestSrc, path.join(paths.exportPostsDir, "manifest.json"));
+ // Narrowed to the members this compose publishes, so a compose run over an
+ // index built before `social.x.visibility` flipped (`archilyzer compose
+ // site` alone, `build site --nodata`) does not list a withheld X channel's
+ // name and count beside the pruned tree.
+ const pm = JSON.parse(await readFile(postsManifestSrc, "utf8")) as PostsManifest;
+ const members = new Set(memberSlugs);
+ const channels = (pm.channels ?? []).filter((c) => members.has(c.slug));
+ await writePublicFile(
+ path.join(paths.exportPostsDir, "manifest.json"),
+ JSON.stringify({
+ ...pm,
+ channels,
+ totalCount: channels.reduce((n, c) => n + (c.postCount ?? 0), 0),
+ }),
+ );
}
// Same for the per-site digests manifest (which channels carry digests).
const digestsManifestSrc = path.join(
@@ -934,7 +1055,7 @@ export async function main(
// --- AI discovery: llms.txt / corpus.json / robots.txt / sitemap.xml ---
// (after site.json + archives — both feed into these fixed-count files)
- await emitAiFiles(paths);
+ await emitAiFiles(site, paths);
// Persist the incremental-compose signatures for the next build.
await writeComposeCache(cachePath, cache);
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -141,6 +141,7 @@ import {
postsAvailabilityPath,
} from "../lib/posts-server";
import { isSocialChannel } from "../lib/channelConfig";
+import { publishedMemberSlugs } from "../lib/postsVisibility";
import {
DIGESTS_MANIFEST_VERSION,
SITE_DIGESTS_MANIFEST_VERSION,
@@ -1940,8 +1941,21 @@ export async function buildIndex({
let aggregateSummaries = 0;
let representativeChannelCount = 0;
+ // Which members a site may publish: an X channel is built only into private
+ // sites while `social.x.visibility` is "private" (lib/postsVisibility.ts —
+ // compose-site narrows its trees by the same rule).
+ const visibilitySettings = getSettings();
for (const site of sites) {
- const memberSlugs = site.channels.map((c) => c.slug);
+ const memberSlugs = publishedMemberSlugs(
+ site,
+ (slug) => channelConfigs.get(slug),
+ visibilitySettings,
+ );
+ // Members the rule leaves out of THIS build, named in the fingerprint below
+ // so flipping the setting (or the site's audience) rebuilds the site.
+ const withheld = site.channels
+ .map((c) => c.slug)
+ .filter((slug) => !memberSlugs.includes(slug));
const slugSet = new Set(memberSlugs);
const slugGroup = new Map<string, string>();
for (const m of site.channels) {
@@ -1972,6 +1986,9 @@ export async function buildIndex({
// yesterday's counts.
curatedRules: curated.rulesHash,
curatedAssign: curated.assignHash,
+ // Only when the visibility rule withholds a member, so a site it does
+ // not touch keeps the fingerprint it had.
+ ...(withheld.length > 0 ? { withheld } : {}),
});
const fpKey = `siteFp:${site.siteId}`;
if (
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -674,6 +674,41 @@ export { SNAPSHOT_FILENAME, snapshotPath, readChannelSnapshot };
const SNAPSHOT_VIDEO_CONCURRENCY = 16;
+// THE WALK YIELDS TO THE EVENT LOOP every SNAPSHOT_YIELD_EVERY videos.
+//
+// It runs in the editor's own process, on the main thread, and a video's unit
+// parses its cues.json and (in the reconcile pass and for a non-YouTube
+// archive) its metadata.info.json — ~0.6 MB each on a long VOD, a few ms of
+// JSON.parse apiece. Sixteen units in flight keep every turn of the loop busy
+// with that work; on 2026-10-01 two regenerations of 2,000–3,000-video
+// channels ran for over an hour while `/` and `/jobs` did not answer. So the
+// ids are walked in chunks, each fanned out under the same limit, and between
+// chunks the walk waits one `setImmediate`: the loop gets a turn with no
+// snapshot work queued in it, and every request whose I/O completed meanwhile
+// runs before the next chunk starts.
+//
+// Two full waves of the limit (32 at 16 wide): a chunk is ~0.1–0.2 s of
+// parsing on such a channel, and a multiple of the width keeps every wave full
+// — 25 left the second wave nine wide, a fifth of the walk's throughput.
+// Measured in the slice D0 record (plans/release-17.md).
+const SNAPSHOT_YIELD_EVERY = 2 * SNAPSHOT_VIDEO_CONCURRENCY;
+
+// `fn` over `items` in chunks of `size`, in order, one setImmediate between
+// chunks. The concurrency inside a chunk is whatever `fn` imposes. Exported for
+// snapshotYield.test.ts, which pins the yield.
+export async function mapInYieldingChunks<T, R>(
+ items: readonly T[],
+ size: number,
+ fn: (item: T) => Promise<R>,
+): Promise<R[]> {
+ const out: R[] = [];
+ for (let i = 0; i < items.length; i += size) {
+ if (i > 0) await new Promise<void>((resolve) => setImmediate(resolve));
+ out.push(...(await Promise.all(items.slice(i, i + size).map(fn))));
+ }
+ return out;
+}
+
// ONE LEVEL of a video dir's `clips/` — the files in it, and nothing deeper.
// Deliberately not recursive: `clipWindow-server.ts` writes `<from>-<to>.<ext>`
// and `<from>-<to>.json` flat into it and nothing else does, so a recursion
@@ -839,8 +874,10 @@ export async function generateChannelSnapshot(
}
const limit = pLimit(SNAPSHOT_VIDEO_CONCURRENCY);
- const perVideo = await Promise.all(
- videoDirNames.map((id) =>
+ const perVideo = await mapInYieldingChunks(
+ videoDirNames,
+ SNAPSHOT_YIELD_EVERY,
+ (id) =>
// One video directory's reads are one unit through the watchdog.
limit(() => through(async () => {
const dir = path.join(dataDir, id);
@@ -974,7 +1011,6 @@ export async function generateChannelSnapshot(
digest,
};
})),
- ),
);
const filesById = new Map<string, VideoFiles>();
diff --git a/common/controller/channelWriters.test.ts b/common/controller/channelWriters.test.ts
@@ -1,5 +1,10 @@
import { test } from "node:test";
import assert from "node:assert/strict";
+import { mkdtemp, mkdir, readFile, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import { settleRunningJobMetas } from "../jobs/bootQueuedJobs";
import { getRegistry, newJobId, type JobRecord } from "../jobs/registry";
import type { AutoQueueKind } from "../lib/autoQueueTypes";
import type { AutoRunnerInFlight } from "./autoRunner";
@@ -214,3 +219,43 @@ test("through the live registry: cancel leaves it stopping, the job's end stamps
assert.equal(typeof record.endedAt, "number");
assert.deepEqual(channelWriters(slug), []);
});
+
+// A GHOST NEVER HOLDS A MOVE (release 17 slice D0): a `running` meta a dead
+// process left on disk is no writer — before the boot pass closes it, and
+// after.
+test("a running meta a dead process left on disk is never a writer", async () => {
+ const root = await mkdtemp(path.join(tmpdir(), "channel-writers-ghost-"));
+ try {
+ const jobsDir = path.join(root, ".jobs");
+ await mkdir(jobsDir, { recursive: true });
+ const id = newJobId();
+ const slug = `ghost-${id}`;
+ await writeFile(
+ path.join(jobsDir, `${id}.meta.json`),
+ JSON.stringify({
+ id,
+ kind: "refresh-report",
+ queueKey: "",
+ channelSlug: slug,
+ status: "running",
+ queuedAt: Date.now() - 60_000,
+ startedAt: Date.now() - 60_000,
+ pid: 2 ** 22 + 1, // past pid_max: no such process
+ }),
+ );
+ assert.deepEqual(channelWriters(slug, { includeQueued: true }), []);
+ const res = await settleRunningJobMetas({
+ paths: { jobsDir } as Paths,
+ bootedAt: Date.now(),
+ isLive: (j) => getRegistry().get(j) !== undefined,
+ });
+ assert.deepEqual(res.interrupted.map((j) => j.id), [id]);
+ const meta = JSON.parse(
+ await readFile(path.join(jobsDir, `${id}.meta.json`), "utf8"),
+ ) as { status: string };
+ assert.equal(meta.status, "cancelled");
+ assert.deepEqual(channelWriters(slug, { includeQueued: true }), []);
+ } finally {
+ await rm(root, { recursive: true, force: true });
+ }
+});
diff --git a/common/controller/channelWriters.ts b/common/controller/channelWriters.ts
@@ -21,6 +21,14 @@ import { getAutoRunnerStatus, type AutoRunnerInFlight } from "./autoRunner";
// enqueued it, and the refusal names the writer so the operator knows what to
// wait for or cancel.
//
+// A GHOST NEVER HOLDS A MOVE (release 17 slice D0). Only THIS process's
+// registry is read, never a `<id>.meta.json`: a meta a dead process left
+// `running` (three `refresh-report`s on 2026-10-01) is not a record here and
+// cannot name a writer. The boot pass closes such metas as interrupted
+// (jobs/bootQueuedJobs.ts `settleRunningJobMetas`) so /jobs stops showing them;
+// nothing here needs to know. Keep it that way: a reader of metas here would
+// have to ask whether the writer is alive (`writerIsGone`) first.
+//
// TWO HALVES, for the reason editor/app/channels/lib/mediaBusy.ts gives: the
// job registry is half the truth. The auto-queue lanes run their per-video
// units in-process and make no job record (the omnimirror incident,
diff --git a/common/controller/poolSummary.test.ts b/common/controller/poolSummary.test.ts
@@ -26,3 +26,26 @@ test("channel-sites.json names listed sites only; a channel only an unlisted sit
});
assert.ok(!JSON.stringify(channelSitesOf(sites)).includes("fixture-unlisted"));
});
+
+// Release 17 slice XP: a PRIVATE site is never listed, so it is in neither; and
+// with X posts private an X channel a public site leaves out of its build is
+// not mapped to that site (buildPoolSummary passes the narrowing).
+test("channel-sites.json leaves out a private site, and an X channel its public site withholds", async () => {
+ const { publishedMemberSlugs } = await import("../lib/postsVisibility");
+ const sites = [
+ parseSite("fixture-a", { channels: [{ slug: "vids" }, { slug: "jer-x" }] }),
+ parseSite("fixture-private", {
+ audience: "private",
+ channels: [{ slug: "vids" }, { slug: "jer-x" }, { slug: "own" }],
+ }),
+ ];
+ const configs: Record<string, { sourceKind?: "social"; platform?: "twitter" }> = {
+ "jer-x": { sourceKind: "social", platform: "twitter" },
+ };
+ const privateX = { social: { x: { visibility: "private" } } };
+ assert.deepEqual(
+ channelSitesOf(sites, (site) => publishedMemberSlugs(site, (slug) => configs[slug], privateX)),
+ { vids: ["fixture-a"] },
+ );
+ assert.deepEqual(channelSitesOf(sites), { vids: ["fixture-a"], "jer-x": ["fixture-a"] });
+});
diff --git a/common/controller/poolSummary.ts b/common/controller/poolSummary.ts
@@ -12,6 +12,9 @@ import { mkdir, readFile } from "node:fs/promises";
import type { Paths } from "../lib/paths";
import { buildStats } from "./buildStats";
import { isListedSite, listSites, type Site } from "../lib/site";
+import { getSettings } from "../lib/settings";
+import { publishedMemberSlugs } from "../lib/postsVisibility";
+import { readChannelConfig } from "./channels";
import {
statsPageFileName,
type StatsManifest,
@@ -49,12 +52,21 @@ export async function readStatsPages(statsDir: string): Promise<VideoStat[]> {
// published `channel-sites.json`. A channel on multiple sites maps to all of
// them; a pool-only channel is simply absent, and so is an unlisted site
// (site.json `listed: false`) and a channel only unlisted sites expose.
-export function channelSitesOf(sites: readonly Site[]): ChannelSitesMap {
+//
+// `membersOf` answers the members a site's build publishes; absent, every
+// member. buildPoolSummary passes lib/postsVisibility.ts's narrowing, so an X
+// channel a public site leaves out while X posts are private is not mapped to
+// that site here either.
+export function channelSitesOf(
+ sites: readonly Site[],
+ membersOf: (site: Site) => readonly string[] = (site) =>
+ site.channels.map((c) => c.slug),
+): ChannelSitesMap {
const channelSites: ChannelSitesMap = {};
for (const site of sites) {
if (!isListedSite(site)) continue;
- for (const c of site.channels) {
- (channelSites[c.slug] ??= []).push(site.siteId);
+ for (const slug of membersOf(site)) {
+ (channelSites[slug] ??= []).push(site.siteId);
}
}
return channelSites;
@@ -79,7 +91,15 @@ export async function buildPoolSummary(opts: {
// side effect, which is harmless.
await buildStats({ paths, wholePoolStatsDir: statsDir });
const sites = listSites(paths);
- const channelSites = channelSitesOf(sites);
+ // The members each site's build publishes (lib/postsVisibility.ts).
+ const configs = new Map<string, Awaited<ReturnType<typeof readChannelConfig>>>();
+ for (const slug of new Set(sites.flatMap((s) => s.channels.map((c) => c.slug)))) {
+ configs.set(slug, await readChannelConfig(paths, slug).catch(() => null));
+ }
+ const settings = getSettings();
+ const channelSites = channelSitesOf(sites, (site) =>
+ publishedMemberSlugs(site, (slug) => configs.get(slug), settings),
+ );
const stats = await readStatsPages(statsDir);
const summary = buildHomepageSummary(
stats,
diff --git a/common/controller/snapshotYield.test.ts b/common/controller/snapshotYield.test.ts
@@ -0,0 +1,39 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mapInYieldingChunks } from "./channelSnapshot";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/snapshotYield.test.ts
+//
+// THE WALK YIELDS BETWEEN CHUNKS (release 17 slice D0). A macrotask queued
+// while the first chunk runs — a request's I/O callback, here a setImmediate
+// probe — runs before the second chunk's first unit starts. Without the yield
+// every unit below would finish in microtasks before the probe ever ran.
+
+test("a macrotask queued during chunk 1 runs before chunk 2 starts", async () => {
+ const events: string[] = [];
+ const items = Array.from({ length: 64 }, (_, i) => i);
+ const out = await mapInYieldingChunks(items, 32, async (i) => {
+ if (i === 0) setImmediate(() => events.push("probe"));
+ events.push(`unit ${i}`);
+ await Promise.resolve();
+ return i * 2;
+ });
+ assert.deepEqual(out, items.map((i) => i * 2), "results in order");
+ const probe = events.indexOf("probe");
+ assert.ok(probe > events.indexOf("unit 31"), "after the first chunk");
+ assert.ok(probe < events.indexOf("unit 32"), "before the second chunk");
+});
+
+test("one chunk, no yield; an empty list, no call", async () => {
+ let calls = 0;
+ assert.deepEqual(
+ await mapInYieldingChunks([1, 2, 3], 32, async (i) => {
+ calls++;
+ return i;
+ }),
+ [1, 2, 3],
+ );
+ assert.equal(calls, 3);
+ assert.deepEqual(await mapInYieldingChunks([], 32, async (i: number) => i), []);
+});
diff --git a/common/jobs/bootQueuedJobs.test.ts b/common/jobs/bootQueuedJobs.test.ts
@@ -1,16 +1,20 @@
import { test } from "node:test";
import assert from "node:assert/strict";
-import { mkdtemp, mkdir, readFile, rm, writeFile } from "node:fs/promises";
+import { mkdtemp, mkdir, readFile, rm, utimes, writeFile } from "node:fs/promises";
import { tmpdir } from "node:os";
import path from "node:path";
import type { Paths } from "../lib/paths";
import type { JobMeta } from "./jobMeta";
import type { JobSpec } from "./jobSpec";
import {
+ INTERRUPTED_REASON,
STORAGE_PASS_WAIT_MS,
+ processIsAlive,
settleAfterStoragePass,
settleQueuedJobMetas,
+ settleRunningJobMetas,
waitForStoragePass,
+ writerIsGone,
type RequeueFn,
} from "./bootQueuedJobs";
@@ -451,3 +455,103 @@ test("a storage pass that throws counts as finished: no timeout, the settle runs
test("the production bound is 60 s", () => {
assert.equal(STORAGE_PASS_WAIT_MS, 60_000);
});
+
+// THE RUNNING PASS (release 17 slice D0). A job running when its process died
+// keeps `running` on disk forever; the next boot closes it `cancelled` with
+// INTERRUPTED_REASON — unless the process that wrote it is still alive (an
+// `archilyzer run` beside the editor), in which case it is left alone.
+
+const ALIVE_OTHER = 4242; // a live process that is not this one
+const DEAD = 4343;
+const SELF = 4444;
+const check = {
+ selfPid: SELF,
+ isProcessAlive: (pid: number) => pid === ALIVE_OTHER || pid === SELF,
+};
+
+test("writerIsGone: no pid, this pid, a dead pid are gone; another live pid is not", () => {
+ const m = (pid?: number) =>
+ ({ id: "X", kind: "k", queueKey: "", status: "running", queuedAt: 0, ...(pid === undefined ? {} : { pid }) }) as JobMeta;
+ assert.equal(writerIsGone(m(), check), true, "a pre-release-17 meta");
+ assert.equal(writerIsGone(m(SELF), check), true, "a previous process with this pid");
+ assert.equal(writerIsGone(m(DEAD), check), true);
+ assert.equal(writerIsGone(m(ALIVE_OTHER), check), false);
+});
+
+test("processIsAlive: this process is; a pid past pid_max is not", () => {
+ assert.equal(processIsAlive(process.pid), true);
+ assert.equal(processIsAlive(2 ** 22 + 1), false);
+});
+
+test("a running meta from a dead process is closed as interrupted, ended at its log's last write", async () => {
+ const f = await fixture([
+ { id: "R1", kind: "refresh-report", status: "running", channelSlug: "the-quartering-rumble", pid: DEAD, startedAt: BOOT - 3 * HOUR },
+ { id: "R2", kind: "refresh-report", status: "running", channelSlug: "the-quartering" }, // no pid: pre-release-17
+ { id: "R3", kind: "whisper-all", status: "running", channelSlug: "x", pid: ALIVE_OTHER }, // an `archilyzer run`
+ { id: "R4", kind: "whisper-all", status: "running", queuedAt: BOOT + 1, pid: SELF }, // this process's own
+ { id: "R5", kind: "whisper-all", status: "running", pid: SELF }, // live in the registry
+ { id: "D1", kind: "whisper-all", status: "done", pid: DEAD },
+ ]);
+ try {
+ const logMtime = BOOT - 2 * HOUR;
+ const logFile = path.join(f.paths.jobsDir, "R1.log");
+ await writeFile(logFile, "Regenerating report for the-quartering-rumble…\n");
+ await utimes(logFile, logMtime / 1000, logMtime / 1000);
+ const lines: string[] = [];
+ const res = await settleRunningJobMetas({
+ paths: f.paths,
+ bootedAt: BOOT,
+ isLive: (id) => id === "R5",
+ log: (l) => lines.push(l),
+ ...check,
+ });
+ assert.deepEqual(
+ res.interrupted.map((j) => j.id),
+ ["R1", "R2"],
+ );
+ const r1 = await f.read("R1");
+ assert.equal(r1.status, "cancelled");
+ assert.equal(r1.cancelReason, INTERRUPTED_REASON);
+ assert.equal(r1.endedAt, logMtime);
+ assert.equal(r1.pid, DEAD, "the rest of the meta is kept");
+ assert.match(await f.log("R1"), /\[boot\] interrupted/);
+ const r2 = await f.read("R2");
+ assert.equal(r2.status, "cancelled");
+ assert.ok(typeof r2.endedAt === "number", "no log: ended at this boot");
+ for (const id of ["R3", "R4", "R5"]) assert.equal((await f.read(id)).status, "running", id);
+ assert.equal((await f.read("D1")).status, "done");
+ assert.equal(lines.length, 1);
+ assert.match(lines[0], /closed as interrupted: 2 \(refresh-report 2\)/);
+ // Idempotent: the next boot finds nothing.
+ const again = await settleRunningJobMetas({
+ paths: f.paths,
+ bootedAt: BOOT,
+ isLive: (id) => id === "R5",
+ ...check,
+ });
+ assert.deepEqual(again.interrupted, []);
+ } finally {
+ await rm(f.root, { recursive: true, force: true });
+ }
+});
+
+test("the queued pass leaves a queued meta whose writer is still alive", async () => {
+ const f = await fixture([
+ { id: "Q1", spec: SPEC, pid: ALIVE_OTHER },
+ { id: "Q2", spec: SPEC, pid: DEAD, kind: "sync" },
+ ]);
+ try {
+ const rq = recordingRequeue();
+ const res = await settleQueuedJobMetas({
+ paths: f.paths,
+ requeue: rq.fn,
+ bootedAt: BOOT,
+ ...check,
+ });
+ assert.deepEqual(rq.calls, []);
+ assert.deepEqual(res.cancelled.map((c) => c.id), ["Q2"]);
+ assert.equal((await f.read("Q1")).status, "queued");
+ } finally {
+ await rm(f.root, { recursive: true, force: true });
+ }
+});
diff --git a/common/jobs/bootQueuedJobs.ts b/common/jobs/bootQueuedJobs.ts
@@ -1,5 +1,5 @@
import path from "node:path";
-import { appendFile, readdir } from "node:fs/promises";
+import { appendFile, readdir, stat } from "node:fs/promises";
import { writeFileAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
import { metaPath, readJobMeta, type JobMeta } from "./jobMeta";
@@ -31,11 +31,14 @@ import type { JobSpec } from "./jobSpec";
// Every meta this pass does not re-queue is closed `cancelled` with a
// `cancelReason`; a re-queued one is closed naming its new id.
//
-// Only metas from BEFORE this boot, and not held by the live registry, are
-// touched: the storage boot pass can enqueue a job of its own while this runs.
-// `running` metas are left alone — whether a job that was mid-flight should be
-// re-run is not a decision a boot pass can make. A malformed meta (readJobMeta
-// → null) is skipped.
+// Only metas from BEFORE this boot, not held by the live registry, and whose
+// writer is gone (`writerIsGone`: another live process — `archilyzer run` —
+// writes into the same `.jobs/`) are touched: the storage boot pass can
+// enqueue a job of its own while this runs. `running` metas are not re-queued
+// — whether a job that was mid-flight should be re-run is not a decision a
+// boot pass can make — but they are CLOSED, by the second pass below
+// (`settleRunningJobMetas`, release 17 slice D0). A malformed meta
+// (readJobMeta → null) is skipped.
//
// Best-effort throughout: a meta that cannot be read or written is skipped,
// and the caller voids the promise so readiness never waits on it.
@@ -80,7 +83,48 @@ export function specKey(spec: JobSpec): string {
});
}
-export type SettleQueuedOpts = {
+// WHO WROTE THE META, AND ARE THEY STILL THERE.
+//
+// The registry is in memory, so after a restart nothing in this process knows a
+// job the last one was running or queuing — and the meta alone cannot say
+// whether its process died or is another process that is very much alive:
+// `archilyzer run` (bin/run-operation.ts) runs a job offline and writes its
+// meta into the same `.jobs/`. So a meta names its writer (`pid`, release 17)
+// and the writer is gone when:
+// - the meta names none: it predates release 17, so the only writer it can
+// have had is a process older than this one;
+// - it names THIS process's pid: a previous process with the same number —
+// a container's editor comes back as the same pid every restart, and the
+// caller already dropped every meta this process wrote (`bootedAt`,
+// `isLive`);
+// - no process has that pid (`kill(pid, 0)` → ESRCH).
+// A pid that answers is left alone, even if the number was reused by an
+// unrelated process: a job left `running` on /jobs is the cost, a live job
+// closed under its own feet would be the alternative.
+export type WriterCheck = {
+ // This process's pid; defaults to process.pid.
+ selfPid?: number;
+ // Defaults to processIsAlive. Injected by the tests.
+ isProcessAlive?: (pid: number) => boolean;
+};
+
+export function processIsAlive(pid: number): boolean {
+ try {
+ process.kill(pid, 0);
+ return true;
+ } catch (err) {
+ // EPERM: it exists, it is just not ours to signal.
+ return (err as NodeJS.ErrnoException).code === "EPERM";
+ }
+}
+
+export function writerIsGone(meta: JobMeta, check: WriterCheck = {}): boolean {
+ if (typeof meta.pid !== "number") return true;
+ if (meta.pid === (check.selfPid ?? process.pid)) return true;
+ return !(check.isProcessAlive ?? processIsAlive)(meta.pid);
+}
+
+export type SettleQueuedOpts = WriterCheck & {
paths: Paths;
// null: cancel, never re-queue (an idle boot, or the e2e test server).
requeue: RequeueFn | null;
@@ -206,6 +250,7 @@ export async function settleQueuedJobMetas(
continue;
}
if (opts.isLive?.(id)) continue;
+ if (!writerIsGone(meta, opts)) continue;
stale.push(meta);
}
@@ -308,11 +353,12 @@ async function closeMeta(
paths: Paths,
meta: JobMeta,
reason: string,
+ endedAt: number = Date.now(),
): Promise<boolean> {
const closed: JobMeta = {
...meta,
status: "cancelled",
- endedAt: Date.now(),
+ endedAt,
cancelReason: reason,
};
try {
@@ -333,3 +379,86 @@ async function closeMeta(
}
return true;
}
+
+// THE BOOT PASS OVER STALE `running` METAS (release 17 slice D0).
+//
+// A job running when its process died never got its terminal write either —
+// a SIGKILL, an OOM, a crash; and a SIGTERM too, whenever the job's function
+// is still unwinding when Next exits (shutdownCancel.ts cancels every live
+// job but does not wait for one to finish, so the `cancelled` write that
+// streamCommand makes when the function returns is usually lost with the
+// process). Its meta says `running` forever. On 2026-10-01 three
+// `refresh-report` metas from processes that had been gone for hours still
+// read `running`.
+//
+// Each is CLOSED as the queued pass closes one — `cancelled`, with
+// INTERRUPTED_REASON — never re-run (whether a half-finished job should run
+// again is the operator's call; Retry is one click). There is no separate
+// `interrupted` status: `cancelled` + `cancelReason` is the terminal state this
+// file already writes for "the server went down under it", and every reader of
+// a meta (listJobs, /jobs, Retry) already handles it. `endedAt` is the job
+// log's last write — the last moment the job is known to have been alive —
+// else this boot.
+//
+// Same filters as the queued pass: before this boot, not in the live registry,
+// writer gone. It does not wait for the storage pass: it re-queues nothing.
+export const INTERRUPTED_REASON =
+ "interrupted: the process running it stopped before it finished";
+
+export type SettleRunningOpts = WriterCheck & {
+ paths: Paths;
+ bootedAt: number;
+ isLive?: (id: string) => boolean;
+ log?: (line: string) => void;
+};
+
+export type BootRunningResult = {
+ interrupted: { id: string; kind: string; channelSlug?: string }[];
+};
+
+export async function settleRunningJobMetas(
+ opts: SettleRunningOpts,
+): Promise<BootRunningResult> {
+ const result: BootRunningResult = { interrupted: [] };
+ let names: string[];
+ try {
+ names = await readdir(opts.paths.jobsDir);
+ } catch {
+ return result;
+ }
+ const ids = names
+ .filter((n) => n.endsWith(".meta.json"))
+ .map((n) => n.slice(0, -".meta.json".length))
+ .sort();
+ for (const id of ids) {
+ const meta = await readJobMeta(opts.paths, id);
+ if (!meta || meta.status !== "running") continue;
+ if (typeof meta.queuedAt === "number" && meta.queuedAt >= opts.bootedAt) {
+ continue;
+ }
+ if (opts.isLive?.(id)) continue;
+ if (!writerIsGone(meta, opts)) continue;
+ let lastAlive = Date.now();
+ try {
+ lastAlive = (await stat(path.join(opts.paths.jobsDir, `${id}.log`))).mtimeMs;
+ } catch {
+ /* no log: this boot is the best bound there is */
+ }
+ if (await closeMeta(opts.paths, meta, INTERRUPTED_REASON, Math.round(lastAlive))) {
+ result.interrupted.push({
+ id,
+ kind: meta.kind,
+ ...(meta.channelSlug ? { channelSlug: meta.channelSlug } : {}),
+ });
+ }
+ }
+ if (result.interrupted.length > 0) {
+ const by = new Map<string, number>();
+ for (const j of result.interrupted) by.set(j.kind, (by.get(j.kind) ?? 0) + 1);
+ opts.log?.(
+ `[boot] jobs a previous process left running, closed as interrupted: ${result.interrupted.length}` +
+ ` (${[...by].map(([k, n]) => `${k} ${n}`).join(", ")})`,
+ );
+ }
+ return result;
+}
diff --git a/common/jobs/jobMeta.ts b/common/jobs/jobMeta.ts
@@ -27,8 +27,14 @@ export type JobMeta = {
spec?: JobSpec;
// Why a job ended `cancelled` without anyone pressing Cancel — written only
// by the boot pass (bootQueuedJobs.ts) for a job that was still `queued`
- // when the server went down. Absent on every other meta.
+ // when the server went down, or still `running` when the process that ran
+ // it died. Absent on every other meta.
cancelReason?: string;
+ // The process that wrote this meta (`process.pid`). The boot pass reads it
+ // to tell a job a dead process left `running` or `queued` from one another
+ // LIVE process owns right now — `archilyzer run` writes into the same
+ // `.jobs/`. Absent on metas written before release 17.
+ pid?: number;
};
export function metaPath(paths: Paths, id: string): string {
@@ -55,6 +61,7 @@ export async function writeJobMeta(
endedAt: record.endedAt,
exitCode: record.exitCode,
spec: record.spec,
+ pid: process.pid,
};
await writeFile(metaPath(paths, record.id), JSON.stringify(meta), "utf8");
} catch {
diff --git a/common/jobs/snapshotScheduler.test.ts b/common/jobs/snapshotScheduler.test.ts
@@ -0,0 +1,185 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test jobs/snapshotScheduler.test.ts
+//
+// ONE REGENERATION AT A TIME, ONCE PER CHANNEL (release 17 slice D0). Every
+// refresh-report used to run on the empty queue key — all at once — and two
+// passes a moment apart could both enqueue the same channel. Temp corpus only.
+
+const root = await mkdtemp(path.join(tmpdir(), "snapshot-scheduler-"));
+process.env.TRANSCRIPTS_DIR = path.join(root, "transcripts");
+process.env.SETTINGS_FILE = path.join(root, "settings.json");
+await writeFile(process.env.SETTINGS_FILE, "{}");
+for (const slug of ["alpha", "beta", "gamma"]) {
+ const dir = path.join(process.env.TRANSCRIPTS_DIR, "channels", slug);
+ await mkdir(path.join(dir, "data"), { recursive: true });
+ await writeFile(path.join(dir, "config.json"), JSON.stringify({ handling: "youtube" }));
+}
+// A channel relocated to a drive that is not there: its walk refuses (guard 3)
+// with a sentence naming the unreachable media.
+{
+ const dir = path.join(process.env.TRANSCRIPTS_DIR, "channels", "unmounted");
+ await mkdir(dir, { recursive: true });
+ await writeFile(
+ path.join(dir, "config.json"),
+ JSON.stringify({ handling: "youtube", dataDir: path.join(root, "no-such-drive", "unmounted", "data") }),
+ );
+}
+
+const { getPaths } = await import("../lib/paths");
+const { getRegistry, newJobId } = await import("./registry");
+const {
+ REFRESH_REPORT_ACTIVE,
+ REFRESH_REPORT_QUEUE,
+ isRefreshReportPending,
+ refreshReportWaitNotice,
+ requestRefreshReport,
+ startRefreshReport,
+ waitForRefreshReport,
+} = await import("./snapshotScheduler");
+
+// A refresh-report that is RUNNING until released: a record on the queue whose
+// start does nothing.
+function holdQueue(slug: string): () => void {
+ const registry = getRegistry();
+ const id = newJobId();
+ const holder = {
+ id,
+ kind: "refresh-report",
+ queueKey: REFRESH_REPORT_QUEUE,
+ channelSlug: slug,
+ status: "queued" as const,
+ queuedAt: Date.now(),
+ logPath: "/dev/null",
+ };
+ registry.register(holder);
+ registry.enqueue(holder, { start: () => {}, onCancel: () => {} });
+ return () => registry.finalize(id, "done");
+}
+
+test.after(async () => {
+ await rm(root, { recursive: true, force: true });
+});
+
+test("two requests for one channel at once enqueue one regeneration", async () => {
+ const paths = getPaths();
+ const [a, b] = await Promise.all([
+ startRefreshReport(paths, "alpha"),
+ startRefreshReport(paths, "alpha"),
+ ]);
+ const started = [a, b].filter((r) => r.ok);
+ const refused = [a, b].filter((r) => !r.ok);
+ assert.equal(started.length, 1, "exactly one starts");
+ assert.equal(refused.length, 1);
+ const r = refused[0];
+ assert.ok(!r.ok && r.info === true && r.error === REFRESH_REPORT_ACTIVE);
+ const s = started[0];
+ assert.ok(s.ok);
+ assert.equal((await s.done).status, "done");
+ assert.equal(isRefreshReportPending("alpha"), false);
+ // Once it has finished, a new request starts a new one.
+ const again = await startRefreshReport(paths, "alpha");
+ assert.ok(again.ok);
+ await again.done;
+});
+
+test("every regeneration runs on the one serial refresh-report queue", async () => {
+ const paths = getPaths();
+ const a = await startRefreshReport(paths, "alpha");
+ const b = await startRefreshReport(paths, "beta");
+ assert.ok(a.ok && b.ok);
+ for (const id of [a.jobId, b.jobId]) {
+ assert.equal(getRegistry().get(id)?.queueKey, REFRESH_REPORT_QUEUE);
+ }
+ // Serial: the second is behind the first unless the first already finished.
+ const first = getRegistry().get(a.jobId);
+ const second = getRegistry().get(b.jobId);
+ if (first?.status === "running") assert.equal(second?.status, "queued");
+ assert.equal((await a.done).status, "done");
+ assert.equal((await b.done).status, "done");
+});
+
+test("a running regeneration gets one queued successor, and no second", async () => {
+ const paths = getPaths();
+ // A regeneration of beta that is RUNNING and stays so until released: a
+ // record on the refresh-report queue whose start does nothing.
+ const registry = getRegistry();
+ const holderId = newJobId();
+ const holder = {
+ id: holderId,
+ kind: "refresh-report",
+ queueKey: REFRESH_REPORT_QUEUE,
+ channelSlug: "beta",
+ status: "queued" as const,
+ queuedAt: Date.now(),
+ logPath: "/dev/null",
+ };
+ registry.register(holder);
+ registry.enqueue(holder, { start: () => {}, onCancel: () => {} });
+ assert.equal(registry.get(holderId)?.status, "running");
+
+ // A change during the walk: it may already have passed it, so one successor
+ // is queued behind it…
+ const successor = await startRefreshReport(paths, "beta");
+ assert.ok(successor.ok, "queued behind the running one, not dropped");
+ assert.equal(registry.get(successor.jobId)?.status, "queued");
+ assert.equal(isRefreshReportPending("beta"), true);
+ // …and a further change finds it queued: it has not started reading yet.
+ const third = await startRefreshReport(paths, "beta");
+ assert.ok(!third.ok && third.info === true && third.error === REFRESH_REPORT_ACTIVE);
+
+ registry.finalize(holderId, "done");
+ assert.equal((await successor.done).status, "done");
+});
+
+// A PERSON'S REFRESH WAITS A BOUNDED TIME (re-review R1, R2).
+
+test("a refresh behind the queue answers with where it is, once the wait runs out", async () => {
+ const paths = getPaths();
+ const release = holdQueue("beta");
+ try {
+ const asked = await requestRefreshReport(paths, "gamma");
+ assert.ok(asked.ok && asked.started);
+ const wait = await waitForRefreshReport(paths, asked.jobId, { timeoutMs: 300, pollMs: 50 });
+ assert.equal(wait.state, "waiting");
+ assert.ok(wait.state === "waiting" && wait.ahead === 1, "one regeneration ahead of it");
+ assert.equal(
+ wait.state === "waiting" && refreshReportWaitNotice(wait),
+ `Queued behind 1 report regeneration — the report updates when it finishes (job ${asked.jobId}).`,
+ );
+ // A second person's click finds the same job, queued.
+ const again = await requestRefreshReport(paths, "gamma");
+ assert.deepEqual(again, { ok: true, jobId: asked.jobId, started: false });
+ release();
+ const finished = await waitForRefreshReport(paths, asked.jobId, { pollMs: 20 });
+ assert.deepEqual(finished, { state: "done", jobId: asked.jobId });
+ } finally {
+ release();
+ }
+});
+
+test("a failed walk is a failure with its sentence, for the click that found it queued too", async () => {
+ const paths = getPaths();
+ const release = holdQueue("beta");
+ let queuedId = "";
+ try {
+ const first = await requestRefreshReport(paths, "unmounted");
+ assert.ok(first.ok && first.started);
+ queuedId = first.jobId;
+ // Another click while it waits: it did not start this job, and still
+ // hears how it ended.
+ const second = await requestRefreshReport(paths, "unmounted");
+ assert.deepEqual(second, { ok: true, jobId: queuedId, started: false });
+ release();
+ const wait = await waitForRefreshReport(paths, queuedId, { pollMs: 20 });
+ assert.equal(wait.state, "failed");
+ assert.ok(wait.state === "failed" && /unmounted/.test(wait.error), "the walk's own sentence");
+ } finally {
+ release();
+ }
+});
diff --git a/common/jobs/snapshotScheduler.ts b/common/jobs/snapshotScheduler.ts
@@ -1,3 +1,5 @@
+import path from "node:path";
+import { readFile } from "node:fs/promises";
import type { Paths } from "../lib/paths";
import {
DEFAULT_REPORT_DEBOUNCE_PRESET,
@@ -10,6 +12,9 @@ import {
} from "../controller/channelSnapshot";
import { getRegistry } from "./registry";
import { drainStream } from "./drainStream";
+import type { StreamActionResult } from "./streamCommand";
+import { REFRESH_REPORT_QUEUE } from "../lib/queueKeys";
+import { readJobMeta } from "./jobMeta";
// Global, debounced "channel report" (snapshot) regeneration scheduler.
//
@@ -93,6 +98,13 @@ type SchedulerState = {
// counts (/channels, the dashboard) would sit stale until the next unrelated
// change. One integer closes that gap at zero cost.
generation: number;
+ // Slugs whose refresh-report is being enqueued right now. Between the
+ // registry check and the record's registration runManagedFunction awaits
+ // (the media guard, the jobs dir), so two passes a moment apart both found
+ // nothing queued and both enqueued — two regenerations of the same
+ // 3,000-video channel, seen live on 2026-10-01. Held until the record is in
+ // the registry, which answers from then on.
+ starting: Set<string>;
};
declare global {
@@ -107,6 +119,7 @@ function getState(): SchedulerState {
timer: null,
firstDirtyAt: null,
generation: 0,
+ starting: new Set(),
};
}
return globalThis.__yttSnapshotScheduler__;
@@ -146,11 +159,210 @@ export function requestChannelSnapshot(paths: Paths, slug: string): void {
state.timer.unref?.();
}
+// THE ONE QUEUE EVERY REFRESH-REPORT RUNS ON (lib/queueKeys.ts): one
+// regeneration at a time, corpus-wide.
+//
+// It was the empty queueKey — "a local filesystem scan, so it runs in
+// parallel" — and every dirty channel's walk ran at once in the editor's own
+// process. On 2026-10-01 two of them (2,000 and 3,260 videos) ran side by side
+// for over an hour and `/` and `/jobs` did not answer. A non-empty key is how
+// the registry serializes (registry.ts `enqueue`: concurrency 1 per key), so
+// nothing new was needed; a report that waits its turn is a report a minute
+// late, and the pages keep answering meanwhile.
+export { REFRESH_REPORT_QUEUE };
+
+// Why startRefreshReport started nothing for a slug that already has one.
+export const REFRESH_REPORT_ACTIVE = "already queued";
+
+// The QUEUED regeneration of `slug` (not yet started), if there is one.
+export function queuedRefreshReportId(slug: string): string | null {
+ const queued = getRegistry()
+ .list()
+ .find(
+ (j) =>
+ j.kind === "refresh-report" &&
+ j.channelSlug === slug &&
+ j.status === "queued",
+ );
+ return queued?.id ?? null;
+}
+
+// Is a regeneration of `slug` queued or being enqueued — one that has not
+// started reading yet, and so will see any change made before it does?
+//
+// A RUNNING one does not count. It may already have walked past the change
+// that asks for this regeneration, so one successor is queued behind it (the
+// queue is serial, so it starts when the running one ends). At most one: a
+// second request finds the successor queued.
+export function isRefreshReportPending(slug: string): boolean {
+ if (getState().starting.has(slug)) return true;
+ return queuedRefreshReportId(slug) !== null;
+}
+
+// Enqueue one channel's report regeneration on REFRESH_REPORT_QUEUE — unless
+// one is already queued or being enqueued for it, which is answered `info`
+// with REFRESH_REPORT_ACTIVE (it has not started reading, so it will see the
+// change). Every walk goes through here — the debounced pass, "Update all
+// reports" and a channel's own Refresh report — so none runs outside the
+// queue, and they dedup against each other. `onError` hears the walk's error
+// sentence (an unmounted drive's) when the job fails.
+export async function startRefreshReport(
+ paths: Paths,
+ slug: string,
+): Promise<StreamActionResult> {
+ if (isRefreshReportPending(slug)) {
+ return { ok: false, error: REFRESH_REPORT_ACTIVE, info: true };
+ }
+ const state = getState();
+ state.starting.add(slug);
+ try {
+ const { runManagedFunction } = await import("./streamCommand");
+ return await runManagedFunction({
+ kind: "refresh-report",
+ queueKey: REFRESH_REPORT_QUEUE,
+ paths,
+ channelSlug: slug,
+ fn: async (onLog) => {
+ onLog(`Regenerating report for ${slug}…`);
+ const snap = await generateChannelSnapshot(paths, slug);
+ getState().generation++;
+ const excluded = excludedDownloadIdSet(snap);
+ const awaitingTranscription = excluded.size
+ ? snap.buckets.downloadedNoTranscript.filter(
+ (id) => !excluded.has(id),
+ ).length
+ : snap.buckets.downloadedNoTranscript.length;
+ onLog(
+ `Done. ${snap.totals.videos} videos · ` +
+ `${snap.undownloadedIds.length} undownloaded · ` +
+ `${awaitingTranscription} awaiting transcription.`,
+ );
+ // No revalidatePath here — the caller revalidates once, after its
+ // streams drain.
+ },
+ });
+ } finally {
+ // The record is registered by now (or nothing was made): the registry
+ // answers for this slug from here on. A reset (resetSnapshotScheduler)
+ // may have swapped the state meanwhile; deleting from the old set is
+ // harmless.
+ state.starting.delete(slug);
+ }
+}
+
+// ONE CHANNEL'S REPORT, ASKED FOR BY A PERSON OR A SCRIPT: the job that will
+// regenerate it — started now, or the one already queued for the channel
+// (which has not started reading, so it is as fresh). A caller mid-enqueue for
+// the same slug has no id yet; it is waited for, briefly. The stream is
+// cancelled: nothing reads it (the log is on disk).
+export async function requestRefreshReport(
+ paths: Paths,
+ slug: string,
+): Promise<{ ok: true; jobId: string; started: boolean } | { ok: false; error: string }> {
+ const started = await startRefreshReport(paths, slug);
+ if (started.ok) {
+ void started.stream.cancel();
+ return { ok: true, jobId: started.jobId, started: true };
+ }
+ if (!started.info) return { ok: false, error: started.error };
+ for (let i = 0; i < 40; i++) {
+ const queued = queuedRefreshReportId(slug);
+ if (queued) return { ok: true, jobId: queued, started: false };
+ if (!getState().starting.has(slug)) break;
+ await new Promise((resolve) => setTimeout(resolve, 50));
+ }
+ // The one we would have waited for started in the meantime (or another
+ // caller's enqueue failed): ask once more — now nothing is queued, so this
+ // queues one, or says why it cannot.
+ const again = await startRefreshReport(paths, slug);
+ if (again.ok) {
+ void again.stream.cancel();
+ return { ok: true, jobId: again.jobId, started: true };
+ }
+ const queued = queuedRefreshReportId(slug);
+ return queued
+ ? { ok: true, jobId: queued, started: false }
+ : { ok: false, error: again.error };
+}
+
+// HOW LONG A PERSON'S "Refresh report" WAITS before it answers with where its
+// job is instead. The queue is serial: behind a 3,000-video walk, or during
+// Update all reports, the report may be minutes away, and a button that says
+// "Refreshing…" for minutes says nothing.
+export const REFRESH_REPORT_WAIT_MS = 15_000;
+
+export type RefreshReportWait =
+ | { state: "done"; jobId: string }
+ | { state: "failed"; jobId: string; error: string }
+ // Still queued or running when the wait ran out. `ahead`: the jobs before it
+ // on the refresh-report queue — 0 means it is the one regenerating now.
+ | { state: "waiting"; jobId: string; ahead: number };
+
+// The `[error]` line a failed job's log ends with (streamCommand writes the
+// thrown sentence there — an unmounted drive's, for a walk), else null.
+async function lastErrorLine(paths: Paths, jobId: string): Promise<string | null> {
+ try {
+ const raw = await readFile(path.join(paths.jobsDir, `${jobId}.log`), "utf8");
+ const lines = raw.split("\n").filter((l) => l.startsWith("[error] "));
+ const last = lines[lines.length - 1];
+ return last ? last.slice("[error] ".length) : null;
+ } catch {
+ return null;
+ }
+}
+
+// Wait up to `timeoutMs` for a refresh-report job to end, and say how it ended
+// — or where it is. Polls the registry (it hands out no completion promise for
+// a job someone else started); a job evicted from it is read from its meta.
+export async function waitForRefreshReport(
+ paths: Paths,
+ jobId: string,
+ opts: { timeoutMs?: number; pollMs?: number } = {},
+): Promise<RefreshReportWait> {
+ const timeoutMs = opts.timeoutMs ?? REFRESH_REPORT_WAIT_MS;
+ const pollMs = opts.pollMs ?? 250;
+ const until = Date.now() + timeoutMs;
+ for (;;) {
+ const job = getRegistry().get(jobId);
+ let status = job?.status;
+ if (!job) {
+ status = (await readJobMeta(paths, jobId))?.status ?? "failed";
+ }
+ if (status === "done") return { state: "done", jobId };
+ if (status !== "queued" && status !== "running") {
+ const error =
+ (await lastErrorLine(paths, jobId)) ??
+ `Refresh report ${status} (job ${jobId})`;
+ return { state: "failed", jobId, error };
+ }
+ if (Date.now() >= until) {
+ return {
+ state: "waiting",
+ jobId,
+ ahead: Math.max(0, getRegistry().positionInQueue(jobId)),
+ };
+ }
+ await new Promise((resolve) => setTimeout(resolve, pollMs));
+ }
+}
+
+// The operator's sentence for a wait that ran out.
+export function refreshReportWaitNotice(w: Extract<RefreshReportWait, { state: "waiting" }>): string {
+ const where =
+ w.ahead === 0
+ ? "Regenerating now"
+ : `Queued behind ${w.ahead} report regeneration${w.ahead === 1 ? "" : "s"}`;
+ return `${where} — the report updates when it finishes (job ${w.jobId}).`;
+}
+
async function fire(): Promise<void> {
const state = getState();
// Snapshot and clear before the async work: any action that fires during
// regeneration re-marks the slug dirty and schedules a fresh pass (trailing
- // edge), instead of being swallowed by this in-flight batch.
+ // edge), instead of being swallowed by this in-flight batch. That pass is
+ // not dropped when the slug's regeneration is RUNNING: startRefreshReport
+ // queues one successor behind it (it is skipped only when one is already
+ // queued, which has not started reading and so will see the change).
const batch = [...state.dirty.entries()];
state.dirty.clear();
state.timer = null;
@@ -158,61 +370,18 @@ async function fire(): Promise<void> {
if (batch.length === 0) return;
try {
- const { runManagedFunction } = await import("./streamCommand");
-
- // Skip channels that already have a refresh-report job queued/running —
- // a fresh snapshot is already on its way (mirrors
- // refreshAllChannelSnapshotsAction's dedup).
- const active = new Set(
- getRegistry()
- .list()
- .filter(
- (j) =>
- j.kind === "refresh-report" &&
- (j.status === "queued" || j.status === "running") &&
- j.channelSlug,
- )
- .map((j) => j.channelSlug as string),
- );
-
const regenerated: string[] = [];
const streams: ReadableStream<string>[] = [];
for (const [slug, paths] of batch) {
- if (active.has(slug)) continue;
- const result = await runManagedFunction({
- kind: "refresh-report",
- // Empty queueKey: bypass queue serialization. Snapshot regen is a local
- // filesystem scan, so it runs in parallel rather than waiting behind
- // sync/download work (see registry.ts enqueue).
- queueKey: "",
- paths,
- channelSlug: slug,
- fn: async (onLog) => {
- onLog(`Regenerating report for ${slug}…`);
- const snap = await generateChannelSnapshot(paths, slug);
- getState().generation++;
- const excluded = excludedDownloadIdSet(snap);
- const awaitingTranscription = excluded.size
- ? snap.buckets.downloadedNoTranscript.filter(
- (id) => !excluded.has(id),
- ).length
- : snap.buckets.downloadedNoTranscript.length;
- onLog(
- `Done. ${snap.totals.videos} videos · ` +
- `${snap.undownloadedIds.length} undownloaded · ` +
- `${awaitingTranscription} awaiting transcription.`,
- );
- // No revalidatePath here — it's done once after all drains below.
- },
- });
+ const result = await startRefreshReport(paths, slug);
if (!result.ok) continue;
regenerated.push(slug);
streams.push(result.stream);
}
// Wait for every snapshot to finish writing before revalidating so the
- // re-rendered pages read fresh counts. queueKey "" runs them in parallel,
- // so this is ~the slowest snapshot, not the sum.
+ // re-rendered pages read fresh counts. They run one at a time on
+ // REFRESH_REPORT_QUEUE, so this is the sum of them.
await Promise.all(streams.map(drainStream));
if (regenerated.length > 0) {
diff --git a/common/lib/builtExport.test.ts b/common/lib/builtExport.test.ts
@@ -116,6 +116,14 @@ test("builtHubProblem accepts only a hub bundle", () => {
builtHubProblem(none.dir),
"export/out holds no hub build — build the hub first",
);
+
+ // Release 17 XP: a hub bundle composed over a site's data is refused.
+ mkdirSync(path.join(hub.dir, "posts", "jer-x"), { recursive: true });
+ mkdirSync(path.join(hub.dir, "summaries"));
+ assert.equal(
+ builtHubProblem(hub.dir),
+ "export/out holds a hub build that still carries a site's data (summaries, posts) — build the hub again",
+ );
} finally {
hub.cleanup();
site.cleanup();
diff --git a/common/lib/builtExport.ts b/common/lib/builtExport.ts
@@ -98,6 +98,63 @@ export function builtBundleProblem(outDir: string, siteId: string): string | nul
return null;
}
+/**
+ * Why `site` may not be deployed because of who it is built for, as one
+ * sentence — or null when it may (release 17 slice XP).
+ *
+ * A PRIVATE site (`site.json` `audience: "private"`) is the operator's own
+ * reading copy: it may carry what the public may not (X posts while
+ * `social.x.visibility` is "private"), so no deploy path ships it. Every
+ * deploy path asks this BEFORE ANY UPLOAD — the R2 archive push included — in
+ * the place it asks builtBundleProblem: runDeployIntoLog, the container deploy
+ * phase, deploySite, and the editor's Build & deploy, Deploy and Build &
+ * deploy all. A build without a deploy is untouched.
+ */
+export function siteDeployProblem(site: {
+ siteId: string;
+ audience?: string;
+}): string | null {
+ if (site.audience !== "private") return null;
+ return (
+ `Site "${site.siteId}" is private (audience: private): it is built for reading ` +
+ `on this machine and is never deployed. Build it without deploying, or set its ` +
+ `audience to public on its Settings tab`
+ );
+}
+
+/**
+ * Why the bundle in `outDir` may not be deployed because it was built PRIVATE
+ * — its corpus.json says `"audience": "private"` (compose-site writes it for a
+ * private site) — as one sentence, or null. Asked beside siteDeployProblem, so
+ * a site switched to public after a private build still cannot ship that
+ * build: it is rebuilt first.
+ */
+export function builtAudienceProblem(outDir: string): string | null {
+ try {
+ const parsed: unknown = JSON.parse(readFileSync(path.join(outDir, "corpus.json"), "utf8"));
+ const site = (parsed as { site?: { audience?: unknown; id?: unknown } } | null)?.site;
+ if (site?.audience !== "private") return null;
+ const id = typeof site.id === "string" ? ` of "${site.id}"` : "";
+ return (
+ `${outDir} holds a private build${id} (its corpus.json says "audience": "private"), ` +
+ `which is never deployed`
+ );
+ } catch {
+ return null;
+ }
+}
+
+/**
+ * Both audience refusals, the site's first: the one sentence a deploy path
+ * logs or throws, or null.
+ */
+export function deployAudienceProblem(
+ site: { siteId: string; audience?: string },
+ outDir: string,
+): string | null {
+ return siteDeployProblem(site) ?? builtAudienceProblem(outDir);
+}
+
// corpus.json's `site.id`, or null when there is no readable one.
function corpusSiteIdIn(outDir: string): string | null {
try {
@@ -128,9 +185,23 @@ export function builtHubProblem(outDir: string): string | null {
if (!existsSync(path.join(outDir, "hub-sites.json"))) {
return "export/out holds no hub build — build the hub first";
}
+ // The hub holds no site's data (compose-hub removes it): a hub bundle that
+ // still carries a site's data trees was composed over one, and could ship
+ // that site's posts — a private site's included.
+ const carried = HUB_FORBIDDEN_TREES.filter((tree) => existsSync(path.join(outDir, tree)));
+ if (carried.length > 0) {
+ return (
+ `export/out holds a hub build that still carries a site's data (${carried.join(", ")}) — ` +
+ `build the hub again`
+ );
+ }
return null;
}
+// The per-site data trees a hub bundle must never carry (the trees of
+// compose-hub's SITE_ONLY_PUBLIC_ENTRIES).
+const HUB_FORBIDDEN_TREES = ["summaries", "transcripts", "subs", "posts", "digests", "stats", "archives"];
+
/**
* Why `outDir` — the homepage package's `homepage/out` — may not be deployed as
* the homepage, as one sentence, or null when it holds a build.
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -175,6 +175,9 @@ export type SiteCorpus = {
description: string;
url?: string;
hubUrl?: string;
+ // Present only on a PRIVATE site's build (site.json `audience`, release 17
+ // slice XP): the operator's own reading copy, which no deploy path ships.
+ audience?: "private";
};
totals: { channels: number; videos: number };
channels: CorpusChannel[];
@@ -229,6 +232,9 @@ export function buildSiteCorpus(
// visible tag has a non-zero count here). Absent/false leaves corpus.json
// shaped as before apart from the spec bump.
hasTags?: boolean;
+ // A private site's build (site.json `audience: "private"`): corpus.json's
+ // `site.audience` says so. Absent/false leaves corpus.json as before.
+ private?: boolean;
},
): SiteCorpus {
const base = descriptor.siteUrl;
@@ -268,6 +274,7 @@ export function buildSiteCorpus(
description: descriptor.siteDescription,
...(descriptor.siteUrl ? { url: descriptor.siteUrl } : {}),
...(descriptor.hubUrl ? { hubUrl: descriptor.hubUrl } : {}),
+ ...(opts.private ? { audience: "private" as const } : {}),
},
totals: { channels: channels.length, videos },
channels,
diff --git a/common/lib/postsVisibility.test.ts b/common/lib/postsVisibility.test.ts
@@ -0,0 +1,99 @@
+// The one rule for which sites an X channel's posts are built into (release 17
+// slice XP). The build that applies it is pinned by
+// bin/compose-site.postsVisibility.test.ts.
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ isXPostsChannel,
+ postsVisibleTo,
+ publishedMemberSlugs,
+ xPostsVisibility,
+} from "./postsVisibility";
+import { isListedSite, parseSite, siteToDisk } from "./siteSchema";
+import { sanitizeSocial } from "../social/xCookieSource";
+
+const X = { sourceKind: "social" as const, platform: "twitter" as const };
+const BLUESKY = { sourceKind: "social" as const, platform: "bluesky" as const };
+const VIDEOS = { platform: "youtube" as const };
+const PRIVATE_X = { social: { x: { visibility: "private" } } };
+const PUBLIC_X = { social: { x: { visibility: "public" } } };
+const publicSite = { audience: undefined };
+const privateSite = { audience: "private" as const };
+
+test("an X channel is a social channel on platform twitter, and nothing else is", () => {
+ assert.equal(isXPostsChannel(X), true);
+ assert.equal(isXPostsChannel(BLUESKY), false);
+ assert.equal(isXPostsChannel(VIDEOS), false);
+ // A video channel on X (were there one) is not a posts channel.
+ assert.equal(isXPostsChannel({ platform: "twitter" }), false);
+ assert.equal(isXPostsChannel(null), false);
+ assert.equal(isXPostsChannel(undefined), false);
+});
+
+test("social.x.visibility: absent and unknown read as public", () => {
+ assert.equal(xPostsVisibility({}), "public");
+ assert.equal(xPostsVisibility({ social: { x: {} } }), "public");
+ assert.equal(xPostsVisibility({ social: { x: { visibility: "hidden" } } }), "public");
+ assert.equal(xPostsVisibility(PUBLIC_X), "public");
+ assert.equal(xPostsVisibility(PRIVATE_X), "private");
+});
+
+test("public: X posts go wherever the channel is a member", () => {
+ for (const settings of [{}, PUBLIC_X]) {
+ assert.equal(postsVisibleTo(publicSite, X, settings), true);
+ assert.equal(postsVisibleTo(privateSite, X, settings), true);
+ }
+});
+
+test("private: X posts go to private sites only; every other channel is untouched", () => {
+ assert.equal(postsVisibleTo(publicSite, X, PRIVATE_X), false);
+ assert.equal(postsVisibleTo({}, X, PRIVATE_X), false);
+ assert.equal(postsVisibleTo(privateSite, X, PRIVATE_X), true);
+ for (const site of [publicSite, privateSite]) {
+ assert.equal(postsVisibleTo(site, BLUESKY, PRIVATE_X), true);
+ assert.equal(postsVisibleTo(site, VIDEOS, PRIVATE_X), true);
+ assert.equal(postsVisibleTo(site, null, PRIVATE_X), true);
+ }
+});
+
+test("publishedMemberSlugs narrows the membership, in its order", () => {
+ const configs: Record<string, object> = { vids: VIDEOS, x: X, sky: BLUESKY };
+ const channels = [{ slug: "x" }, { slug: "vids" }, { slug: "sky" }, { slug: "gone" }];
+ const configOf = (slug: string) => configs[slug] as typeof X | undefined;
+ assert.deepEqual(
+ publishedMemberSlugs({ channels }, configOf, PRIVATE_X),
+ ["vids", "sky", "gone"],
+ );
+ assert.deepEqual(
+ publishedMemberSlugs({ channels, audience: "private" }, configOf, PRIVATE_X),
+ ["x", "vids", "sky", "gone"],
+ );
+ assert.deepEqual(
+ publishedMemberSlugs({ channels }, configOf, {}),
+ ["x", "vids", "sky", "gone"],
+ );
+});
+
+test("site.json audience: only private is kept and written; a private site is never listed", () => {
+ assert.equal(parseSite("a", {}).audience, undefined);
+ assert.equal(parseSite("a", { audience: "public" }).audience, undefined);
+ assert.equal(parseSite("a", { audience: "secret" }).audience, undefined);
+ const priv = parseSite("a", { audience: "private" });
+ assert.equal(priv.audience, "private");
+ assert.equal(siteToDisk(priv).audience, "private");
+ assert.equal("audience" in siteToDisk(parseSite("a", {})), false);
+ assert.equal(isListedSite(priv), false);
+ assert.equal(isListedSite({ ...priv, listed: true }), false);
+ assert.equal(isListedSite(parseSite("a", {})), true);
+});
+
+test("sanitizeSocial keeps visibility beside cookieSource and drops an unknown one", () => {
+ assert.deepEqual(sanitizeSocial({ x: { visibility: "private" } }), {
+ x: { visibility: "private" },
+ });
+ assert.deepEqual(
+ sanitizeSocial({ x: { cookieSource: "browser", visibility: "public" } }),
+ { x: { cookieSource: "browser", visibility: "public" } },
+ );
+ assert.deepEqual(sanitizeSocial({ x: { visibility: "nobody" } }), { x: {} });
+});
diff --git a/common/lib/postsVisibility.ts b/common/lib/postsVisibility.ts
@@ -0,0 +1,75 @@
+// WHICH SITES A CHANNEL'S POSTS ARE BUILT INTO (release 17 slice XP).
+//
+// THE ONE RULE, asked by both places a site's posts are decided:
+// - the index build's per-site loop (controller/buildIndex.ts), which writes
+// each site's summaries, subs, posts and digests manifests from the site's
+// member channels;
+// - the site compose (bin/compose-site.ts), which copies the shared
+// per-channel trees (transcripts, subs, posts, digests) of those members
+// into the served public dir and writes /site.json and /corpus.json.
+// Both narrow the site's members through `publishedMemberSlugs`, so the
+// manifests and the trees can never disagree about a channel.
+//
+// The rule: with `social.x.visibility` "private", an X channel (a social
+// channel on platform "twitter") is built only into a PRIVATE site
+// (`site.json` `audience: "private"`). Posts are all a social channel holds — it
+// has no videos (buildIndex's scan skips it) — so a public site leaves the
+// channel out whole: no posts tree, no posts-manifest entry, no empty channel
+// in its channel list or its corpus.json. Every other channel, and every
+// channel on a private site, is unaffected. Nothing on disk changes and
+// fetching does not: the shared posts tree (exportSharedPostsDir) and the LMDB
+// posts sub-DB are corpus-wide and keep every channel, which is what a private
+// site and the MCP over a private build read.
+//
+// The Search in row's Posts toggle needs no rule of its own: it shows only
+// when the site's posts manifest lists a channel (SearchSessionContext
+// hasPostsCorpus), so a public site whose only posts were X posts ships an
+// empty posts manifest and no Posts toggle.
+//
+// Pure: no fs, no settings read. The callers pass the settings and the configs.
+
+import { isSocialChannel, type ChannelConfig } from "./channelConfig";
+import { isPrivateSite, type Site } from "./siteSchema";
+import type { XPostsVisibility } from "../social/xCookieSource";
+
+// An X channel: the posts of a social channel whose platform is X.
+export function isXPostsChannel(
+ config: Pick<ChannelConfig, "sourceKind" | "platform"> | null | undefined,
+): boolean {
+ return isSocialChannel(config) && config?.platform === "twitter";
+}
+
+// `social.x.visibility`, resolved: absent (or anything unknown) is "public".
+export function xPostsVisibility(settings: {
+ social?: { x?: { visibility?: unknown } };
+}): XPostsVisibility {
+ return settings.social?.x?.visibility === "private" ? "private" : "public";
+}
+
+// Whether `site` may carry the posts of the channel `config` describes. A
+// channel with no readable config is not an X channel as far as this rule
+// knows, and is left to whatever already decides its fate.
+export function postsVisibleTo(
+ site: Pick<Site, "audience">,
+ config: Pick<ChannelConfig, "sourceKind" | "platform"> | null | undefined,
+ settings: { social?: { x?: { visibility?: unknown } } },
+): boolean {
+ if (!isXPostsChannel(config)) return true;
+ if (xPostsVisibility(settings) === "public") return true;
+ return isPrivateSite(site);
+}
+
+// The site's member slugs, in membership order, less the channels whose posts
+// it may not carry (an X channel holds nothing else). `configOf` answers a
+// slug's channel config, or null/undefined when it has none.
+export function publishedMemberSlugs(
+ site: Pick<Site, "audience" | "channels">,
+ configOf: (
+ slug: string,
+ ) => Pick<ChannelConfig, "sourceKind" | "platform"> | null | undefined,
+ settings: { social?: { x?: { visibility?: unknown } } },
+): string[] {
+ return site.channels
+ .map((c) => c.slug)
+ .filter((slug) => postsVisibleTo(site, configOf(slug), settings));
+}
diff --git a/common/lib/queueKeys.ts b/common/lib/queueKeys.ts
@@ -30,6 +30,10 @@ export const DIGEST_REMOTE_QUEUE = "digest:remote";
// is a separate question, answered by backfillLimit() rather than by the queue.
export const BACKFILL_QUEUE = "backfill";
+// Every channel report regeneration (`refresh-report`), one at a time
+// corpus-wide — jobs/snapshotScheduler.ts says why (release 17 slice D0).
+export const REFRESH_REPORT_QUEUE = "refresh-report";
+
// Per-channel queue for channel-local bookkeeping jobs (clean/clear/verify).
export function channelQueueKey(slug: string): string {
return `channel:${slug}`;
diff --git a/common/lib/settingsSchema.ts b/common/lib/settingsSchema.ts
@@ -116,6 +116,15 @@ export const X_SOCIAL_SETTINGS_FIELD_DOCS: FieldDocs<XSocialSettings> = {
"stored: `\"browser\"` when `cookiesFromBrowser` is set and no profile is connected (no " +
"exported jar carrying an auth_token), else `\"profile\"`. The source, not `cookieMode`, " +
"governs the X fetchers.",
+ visibility:
+ "Where X posts may appear. `\"public\"` (the default; absent): an X channel's posts are " +
+ "built into every site that has the channel. `\"private\"`: every X channel's posts (a " +
+ "channel with `sourceKind: \"social\"` and `platform: \"twitter\"`) are left out of every " +
+ "PUBLIC site build — the channel with them, since posts are all an X channel holds — and " +
+ "built only into PRIVATE sites (`site.json` `audience`). Nothing on disk changes and " +
+ "fetching does not; a site already published changes on its next build and deploy, and " +
+ "flipping back is a rebuild. Chosen in the X account session section of /settings; the " +
+ "rule is common/lib/postsVisibility.ts.",
};
export type { AutoQueueSettings } from "./autoQueueTypes";
export type { ChannelPriority } from "./channelPriority";
@@ -1477,7 +1486,7 @@ export const siteSettingsSchema = z.object({
"How yt-dlp invocations use the configured cookies (see common/lib/cookiePolicy.ts): \"always\" passes them on every invocation, \"when-required\" (default; the historical behavior) only to retry an auth/age failure, \"defer\" never in normal runs — auth-gated videos are excluded from batches and collected into the per-channel \"Needs cookies\" bucket for a manual cookie run. Per-channel override available (ChannelConfig.cookieMode).",
),
social: settingsField((v): SocialSettings => sanitizeSocial(v)).describe(
- "Per-platform settings of the social-post fetchers. Today one key: where the X fetchers' login comes from (`social.x.cookieSource`, chosen in the X account session section of /settings). See common/social/xCookieSource.ts.",
+ "Per-platform settings of the social posts. Today two keys, both X's, both chosen in the X account session section of /settings: where the X fetchers' login comes from (`social.x.cookieSource`) and where X posts may appear (`social.x.visibility`). See common/social/xCookieSource.ts.",
),
sleepBetweenDownloadsSeconds: settingsField((v): number => clampSleepBetweenDownloadsSeconds(v)).describe(
"Pause (seconds) inserted between per-video yt-dlp invocations in managed batch downloads. yt-dlp's own `-t sleep` only paces requests within one invocation, so without this the managed loop hammers the source IP back-to-back. 0 disables. Per-channel override available.",
diff --git a/common/lib/site.ts b/common/lib/site.ts
@@ -14,6 +14,7 @@ import { socialLinksForSave } from "./socialLinks";
import { readJsonFileSync, writeJsonAtomic } from "./jsonFile-server";
import {
isListedSite,
+ isPrivateSite,
isValidSiteId,
parseSite,
parseSiteUrl,
@@ -87,11 +88,14 @@ export function siteStatsDir(paths: Paths, siteId: string): string {
}
// The hub URL this site points visitors toward: its own override, else the
-// family default (SiteSettings.homepageUrl). Undefined when neither is set.
+// family default (SiteSettings.homepageUrl). Undefined when neither is set —
+// and always for a PRIVATE site (`audience: "private"`), which belongs under no
+// hub: a hub tells its members by the hubUrl they publish.
export function resolveHubUrl(
site: Site,
settings: SiteSettings = getSettings(),
): string | undefined {
+ if (isPrivateSite(site)) return undefined;
return site.hubUrl ?? parseSiteUrl(settings.homepageUrl);
}
diff --git a/common/lib/siteSchema.ts b/common/lib/siteSchema.ts
@@ -68,6 +68,27 @@ export const RELATED_SITE_GROUP_FIELD_DOCS: FieldDocs<RelatedSiteGroup> = {
"Sibling site ids, in display order. Invalid and repeated ids are dropped, and a group left with none is dropped. Ids are resolved against the live pool at render time, so an id for a site that does not exist (yet) is harmless — it is skipped.",
};
+// Who a site is built for (release 17 slice XP). "public" (the default, never
+// written) is every site there has ever been. "private" is the operator's own
+// reading copy: never deployed (publish/build.ts asks siteDeployProblem in
+// lib/builtExport.ts before any upload), never listed (isListedSite below),
+// publishing no hubUrl (lib/site.ts resolveHubUrl), and the only kind of site
+// that content kept from the public (X posts while `social.x.visibility` is
+// "private", lib/postsVisibility.ts) is built into.
+export type SiteAudience = "public" | "private";
+
+export const SITE_AUDIENCES: readonly SiteAudience[] = ["public", "private"];
+
+export function isSiteAudience(v: unknown): v is SiteAudience {
+ return v === "public" || v === "private";
+}
+
+// THE ONE PREDICATE for a private site. Absent or anything but "private" reads
+// as public.
+export function isPrivateSite(site: Pick<Site, "audience">): boolean {
+ return site.audience === "private";
+}
+
// A Site is a selection + presentation layer over the single global channel
// pool. Each field is documented in SITE_FIELD_DOCS below.
export type Site = {
@@ -85,6 +106,7 @@ export type Site = {
accent?: string;
siteUrl?: string;
listed?: boolean;
+ audience?: SiteAudience;
relatedSites?: RelatedSiteGroup[];
pwa?: boolean;
archives?: boolean;
@@ -119,6 +141,8 @@ export const SITE_FIELD_DOCS: FieldDocs<Site> = {
"Absolute public URL of this site's deployment, e.g. `https://jeralyzer.pages.dev` (trimmed, trailing slashes removed; anything not absolute http(s) is dropped). Drives the cross-site footer: a site with no siteUrl is omitted from every other site's list.",
listed:
"Whether the family lists this site. Opt-OUT: absent/true = listed, only an explicit `false` is written. An unlisted site still builds and deploys as before, and its own pages are unchanged; it is left out of the homepage (cards, chart, `/stats`), the hub (members, federated search, `/corpus.json`, `/llms.txt`), every other site's footer, and the published `channel-sites.json` and pooled `stats/`. A channel only unlisted sites expose is in none of the family's public totals; a channel a listed site also exposes is credited to the listed one.",
+ audience:
+ 'Who this site is built for. `"public"` (the default; absent) or `"private"`: the operator\'s own reading copy, built on this machine and never deployed — every deploy path (Build & deploy, Deploy, `archilyzer deploy site`, Build & deploy all, docker/publish-site.sh) refuses it before any upload, while a build without a deploy still works. A private site is never listed (as `listed: false`, whatever `listed` says), publishes no `hubUrl`, and its `/corpus.json` says `"audience": "private"`. Content kept from the public — X posts while `social.x.visibility` is `"private"` — is built only into private sites. Only `"private"` is written.',
relatedSites:
"Pulls specific siblings to the front of the footer's cross-site list, in named groups. Siblings not named here fall into a trailing \"Other sites\" group. Absent/empty = one flat list of every sibling.",
pwa:
@@ -150,8 +174,11 @@ export function isValidSiteId(id: unknown): id is string {
// (lib/site.ts resolveRelatedSites). The editor's own pages list every site.
// Here, beside the key, and exported from lib/site like isValidSiteId, so the
// pure summary builder can use it without importing file I/O.
-export function isListedSite(site: Pick<Site, "listed">): boolean {
- return site.listed !== false;
+//
+// A PRIVATE site (`audience: "private"`) is never listed, whatever `listed`
+// says: it is never deployed, so there is nothing at its URL to list.
+export function isListedSite(site: Pick<Site, "listed" | "audience">): boolean {
+ return site.listed !== false && !isPrivateSite(site);
}
// The channels whose content belongs to unlisted sites alone: exposed by at
@@ -160,7 +187,7 @@ export function isListedSite(site: Pick<Site, "listed">): boolean {
// and a channel no site exposes (pool-only) is not here either — the family's
// instance-wide totals have always counted it.
export function channelsOnlyOnUnlistedSites(
- sites: readonly Pick<Site, "listed" | "channels">[],
+ sites: readonly Pick<Site, "listed" | "audience" | "channels">[],
): Set<string> {
const onListed = new Set<string>();
const onUnlisted = new Set<string>();
@@ -281,6 +308,10 @@ export const siteFieldsSchema = z.object({
siteUrl: settingsField(parseSiteUrl).describe(d.siteUrl),
// Opt-out: only an explicit false unlists. Absent/true stays listed.
listed: settingsField((v): boolean => v !== false).describe(d.listed),
+ // Only "private" is kept; absent (and anything else) is the public default.
+ audience: settingsField((v): SiteAudience | undefined =>
+ v === "private" ? "private" : undefined,
+ ).describe(d.audience),
relatedSites: settingsField(parseRelatedSites).describe(d.relatedSites),
pwa: settingsField((v): boolean => v === true).describe(d.pwa),
// Opt-out: only an explicit false disables. Absent/true stays on.
@@ -372,6 +403,8 @@ export function siteToDisk(site: Site): Site {
...(siteUrl ? { siteUrl } : {}),
// Listed is the default: only the opt-out is persisted.
...(site.listed === false ? { listed: false } : {}),
+ // Public is the default: only the private audience is persisted.
+ ...(isPrivateSite(site) ? { audience: "private" as const } : {}),
...(relatedSites.length > 0 ? { relatedSites } : {}),
...(site.pwa ? { pwa: true } : {}),
// Persist only the non-default: archives is on unless explicitly disabled.
diff --git a/common/publish/build.test.ts b/common/publish/build.test.ts
@@ -23,6 +23,7 @@ import {
homepageDeployArgs,
homepageOutDir,
deployHomepage,
+ deploySite,
dockerSiteOutDir,
dockerSiteStagingDir,
resolveOutDir,
@@ -393,3 +394,120 @@ test("runDeployIntoLog refuses a bundle that is not the site's own before wrangl
rmSync(root, { recursive: true, force: true });
}
});
+
+// A PRIVATE site (site.json `audience: "private"`, release 17 slice XP) is never
+// deployed, and neither is a bundle built private (its corpus.json says so):
+// refused at the same door as the wrong-site bundle, before wrangler, in words
+// naming the audience. The fake `pnpm` is the test above's.
+function writePrivateBundle(dir: string, siteId: string): void {
+ writeBundle(dir, siteId, siteId);
+ writeFileSync(
+ path.join(dir, "corpus.json"),
+ JSON.stringify({ site: { id: siteId, audience: "private" } }),
+ );
+}
+
+test("runDeployIntoLog refuses a private site and a private build before wrangler", async () => {
+ const root = mkdtempSync(path.join(os.tmpdir(), "deploy-private-"));
+ const bin = path.join(root, "bin");
+ const argvFile = path.join(root, "pnpm-argv");
+ mkdirSync(bin);
+ writeFileSync(path.join(bin, "pnpm"), `#!/bin/sh\nprintf '%s\\n' "$@" >> '${argvFile}'\n`);
+ chmodSync(path.join(bin, "pnpm"), 0o755);
+ const savedPath = process.env.PATH;
+ process.env.PATH = bin;
+ const signal = new AbortController().signal;
+ const testPaths = { ...paths, exportDir: root } as Paths;
+ try {
+ await runChildIntoLog(() => {}, signal, { command: "pnpm", args: ["--fake?"], cwd: root, env: { ...process.env } });
+ assert.equal(readFileSync(argvFile, "utf8"), "--fake?\n");
+ rmSync(argvFile);
+
+ // The site is private: its own, well-formed bundle is still refused.
+ const own = path.join(root, "own", "out");
+ writeBundle(own, "mine", "mine");
+ const priv = { siteId: "mine", cloudflareProject: "w3c-never-real", audience: "private" } as Site;
+ let log: string[] = [];
+ assert.equal(await runDeployIntoLog((l) => log.push(l), signal, priv, own, testPaths), 1);
+ assert.equal(log.length, 1, log.join(""));
+ assert.match(
+ log[0],
+ /^\[deploy\] REFUSED — Site "mine" is private \(audience: private\): it is built for reading on this machine and is never deployed\./,
+ );
+ assert.match(log[0], /Nothing was sent to Cloudflare Pages\.\n$/);
+
+ // The site is public now, but the bundle was built private.
+ const built = path.join(root, "built", "out");
+ writePrivateBundle(built, "mine");
+ log = [];
+ const pub = { siteId: "mine", cloudflareProject: "w3c-never-real" } as Site;
+ assert.equal(await runDeployIntoLog((l) => log.push(l), signal, pub, built, testPaths), 1);
+ assert.match(log[0], /holds a private build of "mine" \(its corpus\.json says "audience": "private"\)/);
+ assert.equal(existsSync(argvFile), false, "pnpm was spawned");
+ } finally {
+ process.env.PATH = savedPath;
+ rmSync(root, { recursive: true, force: true });
+ }
+});
+
+test("runDockerDeployAllPhase skips a private site before the upload, in the audience's words", async () => {
+ const root = mkdtempSync(path.join(os.tmpdir(), "deploy-all-private-"));
+ try {
+ const outFor = (id: string) => path.join(root, id, "out");
+ writeBundle(outFor("mine"), "mine", "mine");
+ writePrivateBundle(outFor("built"), "built");
+ const log: string[] = [];
+ // A Cloudflare project on each: without the audience check the run would
+ // reach the upload, which the log would show.
+ const sites = [
+ { siteId: "mine", audience: "private", cloudflareProject: "w3c-never-real" },
+ { siteId: "built", cloudflareProject: "w3c-never-real" },
+ ] as Site[];
+ const outcomes = await runDockerDeployAllPhase(
+ (l) => log.push(l),
+ new AbortController().signal,
+ sites,
+ new Set(["mine", "built"]),
+ { ...paths, exportBuildsDir: root } as Paths,
+ outFor,
+ );
+ assert.deepEqual(outcomes.map((o) => [o.siteId, o.status]), [["mine", "skipped"], ["built", "skipped"]]);
+ assert.match(outcomes[0].reason!, /^Site "mine" is private \(audience: private\)/);
+ assert.match(outcomes[1].reason!, /private build of "built"/);
+ assert.ok(log.some((l) => l.startsWith("[mine] deploy skipped — Site \"mine\" is private")), log.join("\n"));
+ assert.ok(!log.some((l) => l.startsWith("=== Deploy")), "nothing reached the deploy");
+ } finally {
+ rmSync(root, { recursive: true, force: true });
+ }
+});
+
+test("deploySite (archilyzer deploy site) refuses a private site before anything, and a private build before the upload", async () => {
+ const root = mkdtempSync(path.join(os.tmpdir(), "deploy-site-private-"));
+ try {
+ const sitesDir = path.join(root, "sites");
+ const site = (id: string, extra: Record<string, unknown> = {}) => {
+ mkdirSync(path.join(sitesDir, id), { recursive: true });
+ writeFileSync(
+ path.join(sitesDir, id, "site.json"),
+ JSON.stringify({ siteId: id, cloudflareProject: "w3c-never-real", ...extra }),
+ );
+ };
+ site("mine", { audience: "private" });
+ site("other");
+ const testPaths = { ...paths, exportDir: root, sitesDir } as Paths;
+ const log: string[] = [];
+ await assert.rejects(
+ deploySite("mine", { paths: testPaths, onLog: (l) => log.push(l) }),
+ /^Error: Site "mine" is private \(audience: private\): it is built for reading on this machine and is never deployed\. Build it without deploying, or set its audience to public on its Settings tab\.$/,
+ );
+ // Public, but export/out holds a private build of it.
+ writePrivateBundle(path.join(root, "out"), "other");
+ await assert.rejects(
+ deploySite("other", { paths: testPaths, onLog: (l) => log.push(l) }),
+ /private build of "other".*Build other again, then deploy\.$/,
+ );
+ assert.deepEqual(log, [], "nothing was logged: no upload, no deploy");
+ } finally {
+ rmSync(root, { recursive: true, force: true });
+ }
+});
diff --git a/common/publish/build.ts b/common/publish/build.ts
@@ -14,7 +14,14 @@ import { createReadStream, existsSync } from "node:fs";
import { S3Client, HeadObjectCommand } from "@aws-sdk/client-s3";
import { Upload } from "@aws-sdk/lib-storage";
import { runChildIntoLog } from "../jobs/runChild";
-import { builtBundleProblem, builtHubProblem, builtSiteProblem } from "../lib/builtExport";
+import {
+ builtAudienceProblem,
+ builtBundleProblem,
+ builtHubProblem,
+ builtSiteProblem,
+ deployAudienceProblem,
+ siteDeployProblem,
+} from "../lib/builtExport";
import { getHomepageConfig } from "../lib/homepage";
import {
deploymentUrlIn,
@@ -340,6 +347,16 @@ export async function runDeployIntoLog(
// job (a build of another site, the hub) can rewrite export/out. Nothing runs
// between this check and the spawn. The hub and the homepage deploy through
// runPagesDeployIntoLog and never come here.
+ //
+ // A PRIVATE site, or a bundle built private, is refused first: it is never
+ // deployed, whatever the bundle's identity (deployAudienceProblem).
+ const audienceProblem = deployAudienceProblem(site, outDir);
+ if (audienceProblem) {
+ onLog(
+ `[deploy] REFUSED — ${audienceProblem}. Nothing was sent to Cloudflare Pages.\n`,
+ );
+ return 1;
+ }
const bundleProblem = builtBundleProblem(outDir, site.siteId);
if (bundleProblem) {
onLog(
@@ -604,10 +621,21 @@ export async function runDockerDeployAllPhase(
outcomes.push({ siteId: site.siteId, status: "skipped", reason: "build failed" });
continue;
}
- // The bundle must be this site's own before anything else is asked of it —
- // the check build-site.sh makes before it hands the bundle back, made again
- // over whatever the per-site dir holds now. First, so that nothing past it
- // (the R2 upload, the Pages deploy) is ever reached with another site's data.
+ // A private site (or a private build) is not deployed at all: skipped, in
+ // its own words, so a family with one private site does not fail every run.
+ // Asked first; a per-site dir holding another site's bundle built private
+ // therefore reads "skipped" rather than "REFUSED" — never shipped either way.
+ //
+ // Then the bundle must be this site's own — the check build-site.sh makes
+ // before it hands the bundle back, made again over whatever the per-site dir
+ // holds now — before anything past it (the R2 upload, the Pages deploy) is
+ // reached with another site's data.
+ const audienceProblem = deployAudienceProblem(site, outDirFor(site.siteId));
+ if (audienceProblem) {
+ onLog(`[${site.siteId}] deploy skipped — ${audienceProblem}`);
+ outcomes.push({ siteId: site.siteId, status: "skipped", reason: audienceProblem });
+ continue;
+ }
const bundleProblem = builtBundleProblem(outDirFor(site.siteId), site.siteId);
if (bundleProblem) {
onLog(`[${site.siteId}] deploy REFUSED — ${bundleProblem}`);
@@ -753,6 +781,9 @@ export async function deploySite(
}
const branch = opts.previewBranch?.trim() || undefined;
const site = getSite(siteId.trim(), paths);
+ // Before anything else is asked of a private site: it is never deployed.
+ const privateProblem = siteDeployProblem(site);
+ if (privateProblem) throw new Error(`${privateProblem}.`);
if (!site.cloudflareProject) {
throw new Error(
`Site "${site.siteId}" has no Cloudflare Pages project configured.`,
@@ -761,6 +792,9 @@ export async function deploySite(
const outDir = resolveOutDir(site.siteId, paths);
const builtProblem = builtSiteProblem(outDir, site.siteId);
if (builtProblem) throw new Error(builtProblem);
+ // Before the R2 upload below: a bundle built private is never deployed.
+ const builtPrivate = builtAudienceProblem(outDir);
+ if (builtPrivate) throw new Error(`${builtPrivate}. Build ${site.siteId} again, then deploy.`);
// The production path logs no banner and gains none here: its log has
// always opened on wrangler's own first line.
if (branch) {
@@ -892,7 +926,9 @@ export async function composeHub(opts: PublishOpts = {}): Promise<number> {
/**
* Build the hub into export/out. Removes public/site.json first — a site's
* compose left it there, and a hub bundle carrying one would read as that
- * site's (builtExport.ts). Returns the exit code.
+ * site's (builtExport.ts). compose-hub then removes every other per-site entry
+ * a site's compose left in public/ (SITE_ONLY_PUBLIC_ENTRIES), so the hub
+ * never ships a site's data. Returns the exit code.
*/
export async function buildHub(opts: PublishOpts = {}): Promise<number> {
const { paths, onLog, signal } = resolved(opts);
diff --git a/common/social/xCookieSource.ts b/common/social/xCookieSource.ts
@@ -27,11 +27,28 @@ export function isXCookieSource(v: unknown): v is XCookieSource {
return v === "browser" || v === "profile";
}
-// The `social` block of settings.json. Only X has a login to choose today; the
-// block is per platform so a second one does not need a second top-level key.
+// WHERE X POSTS MAY APPEAR — `social.x.visibility` (release 17 slice XP).
+// "public" — the default (absent): an X channel's posts are built into every
+// site that has the channel, as every other channel's are.
+// "private" — an X channel's posts are left out of every PUBLIC site build and
+// built only into PRIVATE sites (`site.json` `audience`). Nothing
+// on disk changes, and fetching does not; flipping back is a
+// rebuild. The rule itself is lib/postsVisibility.ts.
+export type XPostsVisibility = "public" | "private";
+
+export const X_POSTS_VISIBILITIES: readonly XPostsVisibility[] = ["public", "private"];
+
+export function isXPostsVisibility(v: unknown): v is XPostsVisibility {
+ return v === "public" || v === "private";
+}
+
+// The `social` block of settings.json. Only X has settings today; the block is
+// per platform so a second one does not need a second top-level key.
export type XSocialSettings = {
// Absent = the read-time default above.
cookieSource?: XCookieSource;
+ // Absent = "public".
+ visibility?: XPostsVisibility;
};
export type SocialSettings = {
@@ -39,7 +56,8 @@ export type SocialSettings = {
};
// Total over `unknown`, as every settings coercion is: anything that is not a
-// known source reads as absent (the default), and unknown keys are dropped.
+// known source or visibility reads as absent (the default), and unknown keys
+// are dropped.
export function sanitizeSocial(value: unknown): SocialSettings {
const r = (value && typeof value === "object" && !Array.isArray(value)
? value
@@ -48,7 +66,10 @@ export function sanitizeSocial(value: unknown): SocialSettings {
? r.x
: {}) as Record<string, unknown>;
return {
- x: isXCookieSource(x.cookieSource) ? { cookieSource: x.cookieSource } : {},
+ x: {
+ ...(isXCookieSource(x.cookieSource) ? { cookieSource: x.cookieSource } : {}),
+ ...(isXPostsVisibility(x.visibility) ? { visibility: x.visibility } : {}),
+ },
};
}
diff --git a/common/views/autoQueueStatus.test.ts b/common/views/autoQueueStatus.test.ts
@@ -8,7 +8,9 @@ import type { AutoRunnerStatus, LeafPending } from "../controller/autoRunner";
import { buildAutoQueueLanes } from "./autoQueueLanes";
import type { PriorityView } from "./channelPriority";
import {
+ AUTO_QUEUE_STATUS_MEMO_MS,
buildAutoQueueStatusPayload,
+ singleFlightMemo,
type AutoQueueStatusInputs,
} from "./autoQueueStatus";
@@ -203,3 +205,101 @@ test("the pending fold is per lane and reaches the focus banner", () => {
// A lane with nothing pending is still an entry, with the same shape.
assert.deepEqual(payload.download.pendingByLeaf, {});
});
+
+// THE POLL'S MEMO (slice D0, release 17): N pollers of the 3 s poll cost one
+// fold of the snapshots, concurrent callers share the one in flight, and time
+// is the only thing that expires it.
+
+function deferred<T>() {
+ let resolve!: (v: T) => void;
+ let reject!: (e: unknown) => void;
+ const promise = new Promise<T>((res, rej) => {
+ resolve = res;
+ reject = rej;
+ });
+ return { promise, resolve, reject };
+}
+
+test("memo: concurrent callers share the computation in flight", async () => {
+ const memo = singleFlightMemo<number>({ now: () => 0 });
+ const d = deferred<number>();
+ let calls = 0;
+ const compute = () => {
+ calls++;
+ return d.promise;
+ };
+ const all = Promise.all([memo.get("k", compute), memo.get("k", compute), memo.get("k", compute)]);
+ d.resolve(42);
+ assert.deepEqual(await all, [42, 42, 42]);
+ assert.equal(calls, 1);
+});
+
+test("memo: a landed value is reused for the window, then recomputed", async () => {
+ let t = 1_000;
+ const memo = singleFlightMemo<number>({ now: () => t });
+ let calls = 0;
+ const compute = async () => ++calls;
+ assert.equal(await memo.get("k", compute), 1);
+ t += AUTO_QUEUE_STATUS_MEMO_MS - 1;
+ assert.equal(await memo.get("k", compute), 1, "inside the window: the memo");
+ t += 1;
+ assert.equal(await memo.get("k", compute), 2, "at the window's end: computed again");
+ assert.equal(calls, 2);
+});
+
+test("memo: the window runs from when the value LANDED, not when it started", async () => {
+ let t = 0;
+ const memo = singleFlightMemo<number>({ ttlMs: 3_000, now: () => t });
+ const d = deferred<number>();
+ const first = memo.get("k", () => d.promise);
+ t = 10_000; // a slow fold: ten seconds
+ d.resolve(7);
+ assert.equal(await first, 7);
+ t = 12_000;
+ assert.equal(await memo.get("k", async () => 8), 7);
+});
+
+test("memo: a rejection is shared by its waiters and never memoized", async () => {
+ const memo = singleFlightMemo<number>({ now: () => 0 });
+ const d = deferred<number>();
+ const a = memo.get("k", () => d.promise);
+ const b = memo.get("k", async () => 99);
+ d.reject(new Error("drive not answering"));
+ await assert.rejects(a, /drive not answering/);
+ await assert.rejects(b, /drive not answering/);
+ assert.equal(await memo.get("k", async () => 5), 5);
+});
+
+test("memo: clear() drops the value and detaches the computation in flight", async () => {
+ const memo = singleFlightMemo<string>({ now: () => 0 });
+ assert.equal(await memo.get("k", async () => "old fixture"), "old fixture");
+ memo.clear();
+ const slow = deferred<string>();
+ const before = memo.get("k", () => slow.promise);
+ memo.clear();
+ // A caller after the clear does not join the detached computation…
+ assert.equal(await memo.get("k", async () => "new fixture"), "new fixture");
+ // …and when it lands, it neither answers the memo nor evicts the new value.
+ slow.resolve("stale");
+ assert.equal(await before, "stale");
+ assert.equal(await memo.get("k", async () => "unused"), "new fixture");
+});
+
+test("memo: a different key misses the memo and the computation in flight", async () => {
+ const memo = singleFlightMemo<string>({ now: () => 0 });
+ assert.equal(await memo.get("tree A", async () => "counts under A"), "counts under A");
+ // A rule added, a focus set: the settings the fold reads changed.
+ assert.equal(await memo.get("tree B", async () => "counts under B"), "counts under B");
+ assert.equal(await memo.get("tree B", async () => "unused"), "counts under B");
+ // Asking under A again is a miss too — the stored value is B's.
+ assert.equal(await memo.get("tree A", async () => "A again"), "A again");
+
+ // In flight: a caller with another key does not join it, and the older
+ // computation landing late does not overwrite the newer key's value.
+ const slowA = deferred<string>();
+ const a = memo.get("tree A2", () => slowA.promise);
+ assert.equal(await memo.get("tree C", async () => "C"), "C");
+ slowA.resolve("late A2");
+ assert.equal(await a, "late A2");
+ assert.equal(await memo.get("tree C", async () => "unused"), "C");
+});
diff --git a/common/views/autoQueueStatus.ts b/common/views/autoQueueStatus.ts
@@ -208,3 +208,97 @@ export function buildAutoQueueStatusPayload(
lanes: inputs.lanes,
};
}
+
+// THE POLL'S EXPENSIVE HALF, SHARED: single-flight plus a short memo.
+//
+// The payload above is cheap; what it is built FROM is not. Each lane's
+// pending work is a fold over every channel's snapshot (the shell's
+// `computeLeafPending`, four lanes), and every surface that shows the lanes
+// polls it every 3 s (`useOperationsStatus`). Each poll used to compute its
+// own: on 2026-10-01, with the main thread busy regenerating a report, one
+// computation took 96 s, and every tab's poll started another behind it — a
+// queue that only grew. Now concurrent callers share the computation in flight,
+// and a result is reused for AUTO_QUEUE_STATUS_MEMO_MS after it lands, so N
+// pollers cost one fold per window.
+//
+// KEYED BY THE SETTINGS THE FOLD READS, AND OTHERWISE BY TIME. The fold reads
+// settings itself (`computeLeafPending`: each lane's policy and rule tree,
+// and `channelPriority` — paused channels, the focus, the compiled leaf ids),
+// so the shell passes a key built from those (`autoQueue` + `channelPriority`):
+// a rule added, a focus set or a channel paused changes the key and misses the
+// memo, so a count is never keyed by a tree that is no longer the one drawn.
+// What nothing tells the memo about — a snapshot rewritten, the runner picking
+// a video (its in-flight set is subtracted inside the fold) — is up to
+// AUTO_QUEUE_STATUS_MEMO_MS behind; the next poll is the correction. A lane's
+// hold, the runner's status and its picks are not behind it at all: the shell
+// reads them on every call (editor/app/operations/status.ts).
+export const AUTO_QUEUE_STATUS_MEMO_MS = 3_000;
+
+export type SingleFlightMemo<T> = {
+ // The memoized value while it is fresh AND was computed under `key`; else the
+ // computation in flight under `key`; else `compute()`, started now and shared
+ // with every caller asking with `key` until it settles. A rejection is not
+ // memoized: the next caller computes again.
+ get(key: string, compute: () => Promise<T>): Promise<T>;
+ // Forget the value AND detach the computation in flight, whose result is
+ // then dropped rather than stored (a test reset must not be answered with
+ // the previous fixture's numbers).
+ clear(): void;
+};
+
+export function singleFlightMemo<T>(
+ opts: { ttlMs?: number; now?: () => number } = {},
+): SingleFlightMemo<T> {
+ const ttlMs = opts.ttlMs ?? AUTO_QUEUE_STATUS_MEMO_MS;
+ const now = opts.now ?? Date.now;
+ let value: { key: string; v: T; at: number } | null = null;
+ let inFlight: { key: string; p: Promise<T>; seq: number } | null = null;
+ // Every computation started gets the next number; only the latest started
+ // may store its value or clear the in-flight slot, so a computation under an
+ // old key that lands late never overwrites a newer one. clear() bumps it too.
+ let seq = 0;
+ return {
+ get(key, compute) {
+ if (value && value.key === key && now() - value.at < ttlMs) {
+ return Promise.resolve(value.v);
+ }
+ if (inFlight && inFlight.key === key) return inFlight.p;
+ const mine = ++seq;
+ const p = compute().then(
+ (v) => {
+ if (seq === mine) {
+ value = { key, v, at: now() };
+ inFlight = null;
+ }
+ return v;
+ },
+ (err: unknown) => {
+ if (seq === mine) inFlight = null;
+ throw err;
+ },
+ );
+ inFlight = { key, p, seq: mine };
+ return p;
+ },
+ clear() {
+ seq++;
+ value = null;
+ inFlight = null;
+ },
+ };
+}
+
+// The editor's one instance, on globalThis like the registry: the polled route
+// and the operations pages are separate bundles, and the e2e reset route
+// (editor/app/api/test/invalidate-cache) drops it through the same global.
+declare global {
+ // eslint-disable-next-line no-var
+ var __yttAutoQueueStatusMemo__: SingleFlightMemo<unknown> | undefined;
+}
+
+export function autoQueueStatusMemo<T>(): SingleFlightMemo<T> {
+ if (!globalThis.__yttAutoQueueStatusMemo__) {
+ globalThis.__yttAutoQueueStatusMemo__ = singleFlightMemo<unknown>();
+ }
+ return globalThis.__yttAutoQueueStatusMemo__ as SingleFlightMemo<T>;
+}
diff --git a/docker/publish-site.sh b/docker/publish-site.sh
@@ -29,6 +29,20 @@ if [ -z "${SITE_ID}" ]; then
fi
export SITE_ID
+
+# A PRIVATE site (site.json `audience: "private"`) is never deployed, and the
+# volume this script fills is what the `site` service serves: refused before
+# the build, and again over the built corpus.json below (lib/builtExport.ts).
+# Building it without publishing is `archilyzer build site <id>`.
+SITE_JSON="${SITES_DIR:-${TRANSCRIPTS_DIR:-/data/transcripts}/sites}/${SITE_ID}/site.json"
+refuse_private() {
+ echo "[publish-site] REFUSED — site '${SITE_ID}' is private (audience: private): it is built for reading on this machine and is never deployed. Nothing was published to ${SITE_OUT}. Build it without publishing: pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts build site ${SITE_ID}" >&2
+ exit 1
+}
+if grep -Eq '"audience"[[:space:]]*:[[:space:]]*"private"' "${SITE_JSON}" 2>/dev/null; then
+ refuse_private
+fi
+
cd /repo
echo "[publish-site] building '${SITE_ID}'"
@@ -37,6 +51,9 @@ echo "[publish-site] building '${SITE_ID}'"
pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts build site "${SITE_ID}"
[ -d /repo/export/out ] || { echo "[publish-site] no export/out after build" >&2; exit 1; }
+if grep -Eq '"audience"[[:space:]]*:[[:space:]]*"private"' /repo/export/out/corpus.json 2>/dev/null; then
+ refuse_private
+fi
echo "[publish-site] publishing -> ${SITE_OUT}"
mkdir -p "${SITE_OUT}"
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,8 @@
# Changelog
## [Unreleased]
+- **umtool can keep each report's render folder on a media drive.** With `UMTOOL_MEDIA_DIR` set, in umtool's environment (restart umtool after setting it), to a directory inside that drive, a report project's `out/` (its fetched windows, segments and finished video) is a link to the same path under that directory: a project's first build makes it there, and `umtool storage move-out <project>` (or `--all`) moves an existing one, copying it, checking the copy and only then leaving the link; `--dry-run` says how much would move, and `umtool storage move-back` brings one home. The manifest, its revisions, notes and sources stay where they are, and nothing in umtool reads a project differently. When the drive is not mounted, a build or source check refuses and says so instead of starting a new folder on the main disk; umtool never creates the media directory itself. `umtool storage` lists where each project's `out/` is. With `UMTOOL_MEDIA_DIR` unset nothing changes.
+- **umtool's cache moves to `~/.cache/archilyzer/umtool`** (`$XDG_CACHE_HOME/archilyzer/umtool` when that is set, or `UMTOOL_CACHE_DIR`). It was inside the song project's data folder, so it followed that folder onto whatever drive it was on. Run `umtool index` once after updating to rebuild the project index in its new place; umtool works without it, only slower, and the rest of the cache is remade as it is needed. `umtool doctor` now also shows the reports, media and cache folders, and the old cache folder while it is still there; it can be deleted.
- **umtool's report videos keep every clip's sound on its picture.** In a crossfaded cut each clip's audio was placed by the audio's own length and its picture by the picture's, and an encoded clip's audio is routinely a few to twenty milliseconds shorter or longer than its video, so the sound drifted further ahead clip by clip: by the end of a seventeen-clip cut it was a third of a second early, and two seconds on one with title and sources cards. Each clip's sound is now padded or trimmed to exactly its picture's length before the crossfade. Every crossfaded report video changes when it is rebuilt, and is in sync; a hard-cut video was not affected.
- **umtool's report videos can wear an on-screen deck: one panel under the footage for the whole cut, with a pip timeline, a title per clip, its source and date, and its QR.** A report manifest whose `render` says `"chrome": { "engine": "hyperframes", "layout": "deck" }` scales the footage into a box above a 190 px panel (both sizes are settings) and draws, over the whole cut, one unlabelled pip per clip on a track that fills as the cut plays, the clip's own title from `onscreen.title`, a subtitle naming the recording and its date (the channel too when the cut spans more than one; `onscreen.subtitle` replaces it), and the clip's QR. At each clip change the marker travels to the next pip and the title, subtitle and QR hand over; over a card the panel slides away and comes back after. The citation header, the corner QR and the section footer are not drawn on such a cut, and chapters take the clip's on-screen title. Every setting (sizes, spacing, date format, what the subtitle names, whether cards keep the panel, the motion's timings) is in `render.chrome.deck` and checked when it is saved; an unknown or out-of-range one is refused with a sentence saying why. The panel is rendered once per cut by HyperFrames (pinned to 0.8.24; `HYPERFRAMES_PKG` or `HYPERFRAMES_BIN` override it) and reused until its text or settings change. `build-video.mjs --chrome-only` redraws it over the built segments without rebuilding or fetching anything, `--no-chrome` builds the framed cut without it, and `--chrome-preview <at> <dur>` renders a short window. In umtool, the report page has an **On-screen** section — a switch, the settings, a table of every entry's title and subtitle with the automatic subtitle as its placeholder and a character counter, a live preview with a scrubber, a true still, **Re-render on-screen** and the built video — and the clip bench has on-screen title and subtitle fields with the panel previewed over the clip. The deck changes nothing, byte for byte, in a cut whose manifest has no `render.chrome`.
- **A report cut that wears the on-screen deck can show posts — Bluesky or X statements — as cards over the footage.** A report manifest's `posts` list (each with its platform, handle, date, words and link) is drawn near the end of the clip each post belongs with: the clip whose recording most closely precedes it by date, unless the post names one with `attachTo`; `hide` leaves one out. A clip's posts appear four seconds apart and stack down a column at the frame's top right; as the first appears, the footage eases aside (to 86 % of its box, at the far side) to make room, and the clip's last frame is held, in silence, for 2.5 seconds so the last post can be read; then they all leave together in the change to the next clip, which comes in at the normal size. When the column is full the oldest slide up and out. Each card slides in from the edge of the frame and flares in the deck's accent as it lands; it has an accent rail down its edge and shows the post's date, a platform label ("Bluesky" or "X") beside `@handle`, its words in paragraphs up to seven lines with an ellipsis, and a QR of the post's link, in the deck's colours and faces. The hold and the move are made where the cut is joined, not in a clip, so `--chrome-only` changes them without rebuilding one; chapters and the deck's timing count the hold. The timing, the hold (`hold`, 0 turns it off), the move (`shift`: its scale and seconds, or `false`), the column's side, width and inset, the QR size and the line limit are settings under `render.chrome.deck.posts`, and a bad post or setting is refused with a sentence before a build fetches anything. Only the seconds the cards are up are rendered, one short sequence per clip, cached like the deck; `--chrome-only`, `--chrome-preview` and a hard-cut cut lay them as they lay the deck, and `--no-chrome` draws neither — though it still holds and moves the footage, which are part of the cut rather than the chrome. A first post that appears inside the hold still moves the footage, and a hold is a whole number of frames. `posts` changes nothing in a cut that has none, and without the deck it is not drawn at all.
@@ -22,6 +24,13 @@
- **A form whose save is refused keeps what you typed.** Every editor form put its plain fields back to the stored values when its save was refused — a site's ID rejected, a page size out of range, a slug already taken — so everything typed had to be typed again. A refused save now leaves every field as you left it, beside the reason: **Settings**; a site's form (new and existing); the hub's config on `/sites`; **Cut release**; a channel's form (new and **Configure**), **Rename** and **Delete**; a video's **Delete directory**; **Drive health timing** on `/storage`; the backup config on `/saved-videos`; the sync operation's controls; the **Digest**, **Diarization**, **Speaker attribution** and **Speaker work lane** settings; and the worker list on `/workers`. A save that succeeds behaves as before, with one difference you may notice: a drop-down, and a checkbox or choice that the page tracks as you change it (a cadence, a worker's **Enabled**, a social link's **Keep in header**, a site membership, a site's accent), now shows what was saved. A form's own drop-downs used to go back to what the page had loaded with until a reload, and a second save from the same page sent that old choice again; the others went back until the page next refreshed itself (every 5 seconds by default).
- **A media move no longer starts over a job that is writing into the channel, holds the channel's writers while it runs, and makes its copy match the source before it verifies — so a transcription or a download during a move cannot fail it.** A move that has waited its turn behind other moves now checks again when it starts: if a job is running on the channel, or an auto-queue lane is working on one of its videos, it stops at once and says which ("a transcription of abc123 is running (Transcribe all, job …) — wait for it or cancel it"), with nothing copied — and a job you have just cancelled counts until it has actually stopped ("is stopping … — wait for it to stop"); **Preview** says the same, and the Storage panel's blocked message now names the job too. While a move's marker stands, the channel is held: every lane skips it, and every job that reads or writes its media (single-video transcriptions, downloads and transcodes and the availability checks now included) refuses to start, including one that was already queued when the move began. The rack shows a **media held** chip in the channel's Tier cell and the Storage panel says "Held: its media is moving"; both go when the move finishes or its marker is cleared. The copy is now followed by a pass that makes the destination copy match the source — files the source no longer has are removed from the copy, never from the source — so a file written or deleted during the copy (a transcriber's scratch folder, say) no longer fails the check, and **Resume move** finishes a move whose copy holds such leftovers. Every file removed from a copy is listed in the move's log, and **Preview** says so when a copy from an earlier attempt is already there. If the source keeps changing, the move stops and lists what differs: extra on the destination, missing there, or changed. A new **Reconcile and resume** button beside **Resume move** lists those differences, makes the copy match and finishes the move, so no file has to be deleted by hand. The saved-video store's move does the same matching and the same check before it starts. Needs a rebuild and restart of the editor.
- **Connecting an X account opens your own browser, and the X fetchers can use your everyday browser's X login instead.** **Settings → X account session → Connect X account** used to open Playwright's bundled Chromium with its automation signals on (the "controlled by automated test software" bar, `navigator.webdriver`): Google's sign-in refused it and X's own login form stalled in it. It now opens your Chromium or Chrome when one is installed (`ARCHILYZER_X_BROWSER` names another; Playwright's bundled Chromium otherwise), without those signals. Google's sign-in may still refuse an embedded browser; X's password login is the reliable path. A new **Login source** choice (`social.x.cookieSource` in `settings.json`) says where the X fetchers' login comes from: **Browser login** hands gallery-dl `--cookies-from-browser` with your `cookiesFromBrowser` on every fetch, so the login lasts as long as you stay logged in to x.com in that browser and no window is needed; **Connected profile** is the session broker, as before. Left on **Automatic**, it is the browser login when `cookiesFromBrowser` is set and no profile is connected, and the profile otherwise. **Check** says which source is in use, whether an X login is visible in it and when it was last used (the browser's cookies are read from a private copy, never written; this reads Firefox's, and gallery-dl reads Chromium's itself). Needs a rebuild and restart of the editor.
+- **X posts can be kept off every public site.** **Settings → X account session** has a new choice, **Where X posts appear** (`social.x.visibility` in `settings.json`): **Public**, the default, builds an X channel's posts into every site that has the channel, as before; **Private** leaves every X channel out of every public site's build — its posts, its posts manifest entry and its place in the channel list, `site.json` and `corpus.json` — and builds it only into private sites (below). Nothing on disk changes and fetching goes on. A site already published changes on its next build and deploy, and a public site whose only posts were X posts loses its **Posts** box under **Search in**. Choosing the X login source no longer forgets this choice, and choosing this one keeps the login source. Needs a rebuild and restart of the editor, then a rebuild and deploy of every site and the hub.
+- **A site can be private: built for reading on this machine, never deployed and never listed.** A site's settings have a new **Audience** choice (`audience` in `site.json`; only `"private"` is written). A private site is refused by every deploy — **Build & deploy**, **Deploy**, `archilyzer deploy site`, `pnpm ops build-deploy` and `deploy-site`, **Build & deploy all** (which builds it and skips its deploy) and `docker/publish-site.sh` — before anything is uploaded, in a sentence naming the audience; **Build** still builds it. It is left off the homepage, the hub and every other site's footer whatever **List on the Archilyzer homepage and hub** says, it publishes no hub URL, and its `corpus.json` says `"audience": "private"`, so a build of it is refused too if the site is switched back to public before it is rebuilt. Point the MCP at a private site's build to ask about what only it holds. Needs a rebuild and restart of the editor.
+- **A site built on the host right after another site no longer ships that site's channel list.** A site's **Build** (and Build all without containers) composes every site into the same folder, and the step that copies a site's summaries, stats and duplicates was skipped when that site's own data had not changed, even though another site had composed there since. The site then went out with the other site's channel list and stats. Those steps now run again whenever another site composed last. Needs a rebuild and restart of the editor.
+- **The hub no longer ships the data of the last site built before it.** **Build hub** built from the folder a site's build had just filled, so the hub carried that site's summaries, transcripts, posts and other data, and served them. The hub's build now clears every site's data first and uses the global search aliases, and **Deploy hub** refuses a hub build that still carries a site's data. Needs a rebuild and restart of the editor, then a rebuild and deploy of the hub.
+- **The dashboard and `/jobs` keep answering while a channel's report is regenerated.** Regenerating a report walks every video of the channel inside the editor, and two regenerations of channels with a few thousand videos, running side by side, kept `/`, `/channels` and `/jobs` from loading for over an hour. Regenerations now run one at a time, on their own `refresh-report` queue on `/jobs` — a channel's own **Refresh report** included, which now waits its turn there too: it waits up to 15 seconds, then says where its job is instead ("Queued behind 3 report regenerations — the report updates when it finishes (job …).") and the page catches up when it runs, and a regeneration that fails now shows its reason under the button; a channel whose report is already waiting is not queued a second time, whether the request came from a finished job, **Refresh report** or **Update all reports**, and a change made while a channel's report is being regenerated queues one more regeneration after it rather than being missed; and the walk pauses between batches of videos so pages are served in between. **Update all reports** answers as soon as the regenerations are queued, and the reports land one after another; `pnpm ops refresh-report` answers with the job ids for both a single channel and `{"all":true}`, which `--wait` follows. Needs a rebuild and restart of the editor.
+- **The operations pages share one count of the lanes' pending work.** Every open operations page asks for the lanes' status every 3 seconds, and each request used to count every lane's pending videos afresh from every channel's report. That count is now made once and handed to every request in the next 3 seconds. Changing a lane's rules, a focus or a channel's priority counts again at once; otherwise a pending count can be up to 3 seconds behind a report that was just rewritten or a video a lane just picked. A lane's hold, its runner and its picks are still read fresh on every request.
+- **Jobs a stopped editor left "running" are closed when it starts again.** A job that was still running when the editor's process ended (killed, crashed, or shut down before the job had finished unwinding) kept "running" in its record for good, and `/jobs` listed it as archived. On start the editor now marks each one **cancelled**, with "interrupted: the process running it stopped before it finished" as the reason on the job's page, and its end time is the last time its log was written. Nothing is run again; **Retry** works as for any cancelled job. A job that another live process is running, such as `archilyzer run`, is left alone, and the same check now keeps the start-up pass from closing that process's queued jobs. Such leftover jobs never blocked a media move.
## [0.11.0] - 2026-09-30
- **Transcripts that arrived after a video was first seen are counted.** The stats behind the homepage, the hub and every site's charts were cached per video and refreshed only when the video's metadata changed, so a transcript that came later — a Whisper run days after the download, or a video downloaded after the last index build — never reached them, and a video with YouTube captions alone had no transcription date. Counts and charts were low; the homepage could show a site with 0 transcripts, 0 channels and 0 hours while it served its videos. A stat is now also redone whenever the index re-reads the video, every transcript has a date, and a captioned video is dated by when its captions arrived rather than by a later Normalize run, so its place on "Transcribed over time" can move. **After updating, rebuild and restart the editor before anything else:** until then, **Build stats dataset** runs the old code and would undo the new stats, while a site, hub or homepage build already runs the new code — and the first stats build of any kind re-reads every video once (about 10–30 minutes on a large archive; it can be stopped and picks up where it stopped). Then build the index, the stats, the homepage, the hub, and the sites.
diff --git a/editor/app/api/ops/refresh-report/route.ts b/editor/app/api/ops/refresh-report/route.ts
@@ -1,11 +1,11 @@
import { NextResponse } from "next/server";
+import { refreshAllChannelSnapshotsAction } from "../../../channels/actions";
+import { channelExists } from "yt-dlp-transcript-common/controller/channels";
+import { requestRefreshReport } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
- refreshAllChannelSnapshotsAction,
- refreshChannelSnapshotAction,
-} from "../../../channels/actions";
-import {
- actionResponse,
OpsInputError,
+ opsFail,
ops,
optBool,
optString,
@@ -16,10 +16,17 @@ export const dynamic = "force-dynamic";
// POST { slug: string } | { all: true }
//
-// The single-channel form REGENERATES SYNCHRONOUSLY (it is a filesystem scan,
-// not a job) and returns `{ ok: true }` once snapshot.json is on disk. The
-// `all` form queues one refresh-report job per channel and returns the bulk
-// { queued, skipped } the /channels header button shows.
+// BOTH FORMS ANSWER ONCE QUEUED, with the job ids `--wait` follows — the ops
+// rule (_lib.ts: a job-starting route returns a jobId and never streams).
+// Every regeneration runs on the one serial refresh-report queue (release 17),
+// so holding the request until a report is on disk meant waiting for every
+// walk queued ahead of it: past five minutes, and fetch's own header timeout
+// ends the CLI with an error while the job goes on to succeed.
+//
+// `{ slug }` → `{ ok, jobId, started }`: the job this call queued, or the one
+// already queued for the channel (`started: false`; it has not started
+// reading, so it is as fresh). `{ all: true }` → the bulk `{ queued, jobIds,
+// skipped }` the /channels header button shows.
export async function POST(request: Request) {
return ops(request, ["slug", "all"], async (body) => {
const all = optBool(body, "all");
@@ -36,8 +43,17 @@ export async function POST(request: Request) {
}
// Re-read through reqSlug now that we know it is the single-channel form:
// the shape check belongs on the value that reaches a path.join.
- return actionResponse(
- await refreshChannelSnapshotAction(reqSlug(body, "slug")),
- );
+ const checked = reqSlug(body, "slug");
+ const paths = getPaths();
+ if (!(await channelExists(paths, checked))) {
+ return opsFail(`Channel "${checked}" not found`, 404);
+ }
+ const requested = await requestRefreshReport(paths, checked);
+ if (!requested.ok) return opsFail(requested.error);
+ return NextResponse.json({
+ ok: true,
+ jobId: requested.jobId,
+ started: requested.started,
+ });
});
}
diff --git a/editor/app/api/test/invalidate-cache/route.ts b/editor/app/api/test/invalidate-cache/route.ts
@@ -125,6 +125,14 @@ function invalidate() {
// a spec that rewrites a title in place inside one mtime tick would otherwise
// see the previous spec's title.
resetVideoTitleMemo();
+ // And the auto-queue status poll's three-second memo of the snapshot-derived
+ // pending counts (common/views/autoQueueStatus.ts): the next spec's first
+ // poll must count ITS fixture, not the previous spec's. Dropped through the
+ // global like the singletons above, so this route imports nothing that
+ // reaches the runner; a computation still in flight lands in the dropped
+ // object.
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
+ (globalThis as any).__yttAutoQueueStatusMemo__ = undefined;
revalidatePath("/", "layout");
return NextResponse.json({ ok: true });
}
diff --git a/editor/app/api/test/settle-running-metas/route.ts b/editor/app/api/test/settle-running-metas/route.ts
@@ -0,0 +1,26 @@
+import { NextResponse } from "next/server";
+import { settleRunningJobMetas } from "yt-dlp-transcript-common/jobs/bootQueuedJobs";
+import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { testRouteDenied } from "../_guard";
+
+export const dynamic = "force-dynamic";
+
+// E2E test harness only, GUARDED BY `E2E_TEST_ROUTES` like every other
+// /api/test route (see _guard.ts). It runs the boot pass over `running` metas
+// (common/jobs/bootQueuedJobs.ts `settleRunningJobMetas`) NOW, the way
+// editor/instrumentation.ts runs it once at boot: the e2e server boots once
+// for the whole suite, so a spec that plants a ghost meta (a dead process's
+// `running` job) has no other way to watch the pass close it. "This boot" is
+// this instant: every meta from before it, not in the registry, whose writer
+// is gone.
+export async function POST() {
+ const denied = testRouteDenied();
+ if (denied) return denied;
+ const result = await settleRunningJobMetas({
+ paths: getPaths(),
+ bootedAt: Date.now(),
+ isLive: (id) => getRegistry().get(id) !== undefined,
+ });
+ return NextResponse.json(result);
+}
diff --git a/editor/app/channels/[slug]/components/RefreshSnapshotButton.tsx b/editor/app/channels/[slug]/components/RefreshSnapshotButton.tsx
@@ -1,23 +1,54 @@
"use client";
-import { useTransition } from "react";
-import { refreshChannelSnapshotAction } from "../../actions";
+import { useState, useTransition } from "react";
+import {
+ refreshChannelSnapshotAction,
+ type RefreshSnapshotResult,
+} from "../../actions";
+// THE CHANNEL'S "Refresh report". The regeneration runs on the serial
+// refresh-report queue (release 17), so the action answers within ~15 s either
+// way: nothing (the report is on disk; the page re-renders), `{ error }` (the
+// walk could not run — an unmounted drive's sentence), or `{ notice }` (still
+// queued: "Queued behind N report regenerations — …, job …"). Both are drawn
+// under the button; they used to be dropped, so a refusal showed nothing.
export function RefreshSnapshotButton({ slug }: { slug: string }) {
const [pending, startTransition] = useTransition();
+ const [result, setResult] = useState<RefreshSnapshotResult>(undefined);
return (
- <button
- type="button"
- onClick={() =>
- startTransition(async () => {
- await refreshChannelSnapshotAction(slug);
- })
- }
- disabled={pending}
- aria-label="refresh channel report"
- className="self-start px-3 py-1 rounded-md bg-primary text-primary-foreground text-sm font-medium hover:opacity-90 disabled:opacity-50"
- >
- {pending ? "Refreshing…" : "Refresh report"}
- </button>
+ <div className="flex flex-col items-start gap-1">
+ <button
+ type="button"
+ onClick={() =>
+ startTransition(async () => {
+ setResult(undefined);
+ setResult(await refreshChannelSnapshotAction(slug));
+ })
+ }
+ disabled={pending}
+ aria-label="refresh channel report"
+ className="self-start px-3 py-1 rounded-md bg-primary text-primary-foreground text-sm font-medium hover:opacity-90 disabled:opacity-50"
+ >
+ {pending ? "Refreshing…" : "Refresh report"}
+ </button>
+ {result && "error" in result && (
+ <span
+ role="alert"
+ aria-label="refresh report error"
+ className="text-xs text-destructive"
+ >
+ {result.error}
+ </span>
+ )}
+ {result && "notice" in result && (
+ <span
+ role="status"
+ aria-label="refresh report notice"
+ className="text-xs text-muted-foreground"
+ >
+ {result.notice}
+ </span>
+ )}
+ </div>
);
}
diff --git a/editor/app/channels/actions.ts b/editor/app/channels/actions.ts
@@ -24,13 +24,13 @@ import { renameChannel } from "yt-dlp-transcript-common/controller/renameChannel
import { inspectChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia";
import { channelMediaBusyReason } from "./lib/mediaBusy";
import {
- excludedDownloadIdSet,
- generateChannelSnapshot,
-} from "yt-dlp-transcript-common/controller/channelSnapshot";
-import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
-import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
-import { runManagedFunction } from "yt-dlp-transcript-common/jobs/streamCommand";
-import { drainStream } from "yt-dlp-transcript-common/jobs/drainStream";
+ REFRESH_REPORT_ACTIVE,
+ refreshReportWaitNotice,
+ requestChannelSnapshot,
+ requestRefreshReport,
+ startRefreshReport,
+ waitForRefreshReport,
+} from "yt-dlp-transcript-common/jobs/snapshotScheduler";
import {
siteChannelIndex,
type Site,
@@ -349,25 +349,41 @@ export async function gotoVideoAction(
redirect(`/channels/${slug}/videos/${encodeURIComponent(id)}`);
}
+// What a channel's Refresh report answers: nothing when the report is on disk,
+// `{ error }` when it could not be made, `{ notice }` when it is still queued.
+export type RefreshSnapshotResult = ActionResult | { notice: string };
+
export async function refreshChannelSnapshotAction(
slug: string,
-): Promise<ActionResult> {
+): Promise<RefreshSnapshotResult> {
const paths = getPaths();
if (!(await channelExists(paths, slug))) {
return { error: `Channel "${slug}" not found` };
}
- // generateChannelSnapshot now THROWS on a channel whose media is not
- // reachable (guard 3) rather than writing a snapshot that says every video is
- // undownloaded. That is the right behaviour and the wrong exception to let
- // out of a server action: an uncaught throw here reaches the client as a
- // digest-only "an error occurred", and the one thing the operator needs is
- // the sentence naming the unmounted drive. Every sibling in this file returns
+ // THROUGH THE REFRESH-REPORT QUEUE, like every other walk (release 17 slice
+ // D0): a walk run here, in the request, beside a queued one was two walks
+ // side by side — half of the 2026-10-01 outage — and could land an older
+ // read over a newer one. The job is the one this click starts, or the one
+ // already queued for the channel (it has not started reading, so it is as
+ // fresh).
+ //
+ // A BOUNDED WAIT (REFRESH_REPORT_WAIT_MS, 15 s). The queue is serial: behind
+ // a 3,000-video walk, or during Update all reports, this channel's report may
+ // be minutes away. Past the bound the action answers with where the job is
+ // ("Queued behind 3 report regenerations — …, job …") and the page catches
+ // up when it runs (the scheduler's generation moves /api/pulse).
+ //
+ // A FAILED WALK IS AN ERROR, whoever started it: generateChannelSnapshot
+ // throws on a channel whose media is not reachable (guard 3) rather than
+ // writing a snapshot that says every video is undownloaded, and that
+ // sentence — the job log's `[error]` line — is what comes back, not a
+ // digest-only "an error occurred". Every sibling in this file returns
// { error }; so does this.
- try {
- await generateChannelSnapshot(paths, slug);
- } catch (e) {
- return { error: (e as Error).message };
- }
+ const requested = await requestRefreshReport(paths, slug);
+ if (!requested.ok) return { error: requested.error };
+ const wait = await waitForRefreshReport(paths, requested.jobId);
+ if (wait.state === "failed") return { error: wait.error };
+ if (wait.state === "waiting") return { notice: refreshReportWaitNotice(wait) };
revalidatePath(`/channels/${slug}`);
// The Report column on /channels is read off this snapshot, and the row
// action sits next to the marker it flips — so revalidate the list too, not
@@ -387,82 +403,39 @@ export type RefreshAllResult = {
// `QueueOutcome` grew one: without it an HTTP caller could not tell a
// fan-out that started work from an action that started none, so
// `pnpm ops refresh-report --json '{"all":true}' --wait` returned the moment
- // the response arrived. (This action also AWAITS its streams, so by the time
- // it answers the work is done — but the ids are what make the response
- // honest about what it started, and identical in shape to every other
- // fan-out's.)
+ // the response arrived. The action answers once the jobs are QUEUED (release
+ // 17: they run one at a time), so the ids are what `--wait` follows.
jobIds: string[];
};
export async function refreshAllChannelSnapshotsAction(): Promise<RefreshAllResult> {
const paths = getPaths();
const channels = await listChannelConfigs(paths);
- const active = new Set(
- getRegistry()
- .list()
- .filter(
- (j) =>
- j.kind === "refresh-report" &&
- (j.status === "queued" || j.status === "running") &&
- j.channelSlug,
- )
- .map((j) => j.channelSlug as string),
- );
const queued: string[] = [];
const jobIds: string[] = [];
const skipped: { slug: string; reason: string }[] = [];
- const streams: ReadableStream<string>[] = [];
for (const c of channels) {
- if (active.has(c.slug)) {
- skipped.push({ slug: c.slug, reason: "already running" });
- continue;
- }
- const result = await runManagedFunction({
- kind: "refresh-report",
- // Empty queueKey: bypass queue serialization. Snapshot regen is a
- // local filesystem scan that never touches the platform, so there's
- // no reason for it to wait behind sync/download work. See
- // registry.ts:69-72 for the documented escape hatch.
- queueKey: "",
- paths,
- channelSlug: c.slug,
- fn: async (onLog) => {
- onLog(`Regenerating report for ${c.slug}…`);
- const snap = await generateChannelSnapshot(paths, c.slug);
- const excluded = excludedDownloadIdSet(snap);
- const awaitingTranscription = excluded.size
- ? snap.buckets.downloadedNoTranscript.filter(
- (id) => !excluded.has(id),
- ).length
- : snap.buckets.downloadedNoTranscript.length;
- onLog(
- `Done. ${snap.totals.videos} videos · ` +
- `${snap.undownloadedIds.length} undownloaded · ` +
- `${awaitingTranscription} awaiting transcription.`,
- );
- // Deliberately no revalidatePath here — calling it from a
- // background fn races with the in-flight re-render that the action's
- // own revalidatePath triggers. The action's single revalidate at the
- // end picks up every fresh snapshot.
- },
- });
+ // The scheduler's one entry point: the refresh-report queue (one
+ // regeneration at a time — they used to run all at once, in parallel) and
+ // its per-slug dedup, shared with the debounced regeneration.
+ const result = await startRefreshReport(paths, c.slug);
if (!result.ok) {
- skipped.push({ slug: c.slug, reason: result.error });
+ skipped.push({
+ slug: c.slug,
+ reason: result.error === REFRESH_REPORT_ACTIVE ? "already queued" : result.error,
+ });
continue;
}
+ // Nothing reads the stream (the log is on disk).
+ void result.stream.cancel();
queued.push(c.slug);
jobIds.push(result.jobId);
- streams.push(result.stream);
}
- // Wait for all snapshots to finish writing before revalidating so the
- // pages that read the snapshots read fresh counts. With queueKey === "" the
- // jobs all run in parallel, so this waits roughly the time of the
- // slowest snapshot, not the sum.
- await Promise.all(streams.map(drainStream));
- revalidatePath("/channels");
- revalidatePath("/operations/[id]", "page");
- revalidatePath("/cleanup");
- revalidatePath("/");
+ // ANSWERS ONCE QUEUED, NOT ONCE DONE. The regenerations run one at a time,
+ // so waiting for them was waiting for every channel's walk added up —
+ // minutes on a real corpus — behind a button and an ops call that time out.
+ // The pages catch up as each lands: every regeneration moves the
+ // scheduler's generation, which /api/pulse carries.
return { queued, jobIds, skipped };
}
diff --git a/editor/app/components/actions/InlineActionButton.tsx b/editor/app/components/actions/InlineActionButton.tsx
@@ -110,7 +110,12 @@ async function runAction(variant: Variant): Promise<StreamActionResult> {
if (result && "error" in result) {
return { ok: false, error: result.error };
}
- // refreshReport runs no managed job, so synthesize an already-complete result.
+ // Still queued after the action's bounded wait: a neutral line, not a red one.
+ if (result && "notice" in result) {
+ return { ok: false, error: result.notice, info: true };
+ }
+ // The report is on disk (the action waited for its job), so synthesize an
+ // already-complete result.
return {
ok: true,
jobId: "",
diff --git a/editor/app/lib/requestCache.ts b/editor/app/lib/requestCache.ts
@@ -21,6 +21,9 @@ import { getSettings } from "yt-dlp-transcript-common/lib/settings";
// scheduler rewrites these files on a ~1 s debounce from job runners inside
// common/, which cannot import next/cache to invalidate anything. A cache
// nothing can invalidate is just a stale number with extra steps.
+// ONE EXCEPTION, bounded: the operations status poll's 3-second memo of the
+// lanes' pending counts (operations/status.ts, `autoQueueStatusMemo`), keyed by
+// the settings it is folded from and otherwise expired by time.
// Keyed on the `paths` argument by identity, which works because getPaths()
// memoizes its result at module scope and hands back the same object every
// call. Pass it straight through; don't spread or rebuild it.
diff --git a/editor/app/operations/[id]/page.tsx b/editor/app/operations/[id]/page.tsx
@@ -280,9 +280,10 @@ export default async function OperationPage({
const sections = sectionsFor(op.id as SectionConfig["operation"]);
const channelWork =
sections.length > 0 ? (
- // No extra disk walk: the census is built from the request-cached
- // `getChannelBriefs` that `buildAutoQueueStatusPayload` already read
- // above (operations/status.ts:38). Keyed for the same reason
+ // The census reads the request-cached `getChannelBriefs`. It is the
+ // same listing `buildAutoQueueStatusPayload` read above only when that
+ // call missed its memo (operations/status.ts); on a hit the payload's
+ // counts may be up to 3 s older than this table. Keyed for the same reason
// settingsFormFor's elements are — a server element handed to a client
// component lands in its children array with React's dev-only key check
// still to run over it.
diff --git a/editor/app/operations/status.ts b/editor/app/operations/status.ts
@@ -14,9 +14,11 @@ import { LANES } from "yt-dlp-transcript-common/lib/autoQueueTypes";
import { getWorkerPool } from "yt-dlp-transcript-common/jobs/workerPool";
import { buildAutoQueueLanes } from "yt-dlp-transcript-common/views/autoQueueLanes";
import {
+ autoQueueStatusMemo,
buildAutoQueueStatusPayload as build,
type AutoQueueStatusPayload,
} from "yt-dlp-transcript-common/views/autoQueueStatus";
+import type { ChannelBrief } from "yt-dlp-transcript-common/controller/channels";
import { getChannelBriefs } from "../lib/requestCache";
import { readPriorityView } from "./channelPriorityView";
@@ -29,12 +31,25 @@ import { readPriorityView } from "./channelPriorityView";
// times per poll, because `buildKind` did its own reading; and the channel
// briefs are shared with the lanes builder through the per-request cache, and
// with the four computeLeafPending calls through their `shared` argument.
-export async function buildAutoQueueStatusPayload(): Promise<AutoQueueStatusPayload> {
+
+// THE SNAPSHOT-DERIVED HALF, behind the shared single-flight memo
+// (`autoQueueStatusMemo`, common/views/autoQueueStatus.ts): the channel briefs
+// and the four lanes' pending work. It is the expensive half, a fold over every
+// channel's snapshot, and the one every poller used to pay for separately. The
+// memo is KEYED by the settings the fold reads (each lane's policy and tree in
+// `autoQueue`, and `channelPriority`), so an operator's edit to either misses it
+// and is never paired with counts from the tree before; otherwise it holds for
+// AUTO_QUEUE_STATUS_MEMO_MS (3 s) — a snapshot rewritten or a video the runner
+// just picked shows at most one poll late. That window is the one exception to
+// requestCache.ts's "no cache longer than a request" rule, bounded by time.
+type SnapshotHalf = {
+ briefs: ChannelBrief[];
+ pendingByKind: LeafPending[];
+};
+
+async function computeSnapshotHalf(): Promise<SnapshotHalf> {
const paths = getPaths();
- const settings = getSettings();
- const pool = getWorkerPool();
- const [priority, state, briefs] = await Promise.all([
- readPriorityView(),
+ const [state, briefs] = await Promise.all([
readAutoQueueState(paths),
getChannelBriefs(paths),
]);
@@ -47,6 +62,25 @@ export async function buildAutoQueueStatusPayload(): Promise<AutoQueueStatusPayl
computeLeafPending(lane, paths, { configs: briefs, state }),
),
);
+ return { briefs, pendingByKind };
+}
+
+export async function buildAutoQueueStatusPayload(): Promise<AutoQueueStatusPayload> {
+ const paths = getPaths();
+ const settings = getSettings();
+ const pool = getWorkerPool();
+ // FRESH on every call: the priority view, the state document (picks,
+ // cooldowns, deferrals), the settings (holds, policies), the pool and the
+ // runners. The pending counts are memoized under a key of the settings they
+ // are folded from, so an edit to a policy, a tree or a priority recomputes
+ // them at once; only what changes without a settings write (a snapshot, the
+ // runner's in-flight set) can be up to 3 s behind.
+ const memoKey = JSON.stringify([settings.autoQueue, settings.channelPriority]);
+ const [priority, state, { briefs, pendingByKind }] = await Promise.all([
+ readPriorityView(),
+ readAutoQueueState(paths),
+ autoQueueStatusMemo<SnapshotHalf>().get(memoKey, computeSnapshotHalf),
+ ]);
const byLane = <T>(values: readonly T[]): Record<AutoQueueKind, T> =>
Object.fromEntries(LANES.map((lane, i) => [lane, values[i]])) as Record<
AutoQueueKind,
diff --git a/editor/app/settings/components/XPostsVisibilityControl.tsx b/editor/app/settings/components/XPostsVisibilityControl.tsx
@@ -0,0 +1,74 @@
+"use client";
+
+// Where X posts appear — `social.x.visibility` (release 17 slice XP), inside the
+// X account session section. "Public" builds an X channel's posts into every
+// site that has the channel; "Private" keeps them out of every public site and
+// builds them only into private sites (site.json `audience`). The rule is
+// common/lib/postsVisibility.ts; nothing on disk changes and fetching does not.
+
+import { useEffect, useState, useTransition } from "react";
+import { setXPostsVisibilityAction } from "../xSessionActions";
+import type { XPostsVisibility } from "yt-dlp-transcript-common/social/xCookieSource";
+
+export function XPostsVisibilityControl({
+ initial,
+}: {
+ initial: XPostsVisibility;
+}) {
+ const [visibility, setVisibility] = useState<XPostsVisibility>(initial);
+ const [error, setError] = useState<string | null>(null);
+ const [note, setNote] = useState<string | null>(null);
+ const [saving, startSaving] = useTransition();
+
+ // Another tab's choice re-renders the page with a new value; take it.
+ useEffect(() => setVisibility(initial), [initial]);
+
+ const choose = (choice: string) =>
+ startSaving(async () => {
+ setError(null);
+ setNote(null);
+ const res = await setXPostsVisibilityAction(choice);
+ if (res.ok) {
+ setVisibility(res.visibility);
+ setNote("Saved. Published sites change on their next build and deploy.");
+ } else {
+ setError(res.error);
+ }
+ });
+
+ return (
+ <div data-x-posts-visibility="" className="flex flex-col gap-1">
+ <div className="flex flex-wrap items-center gap-2">
+ <label htmlFor="x-posts-visibility" className="text-sm">
+ Where X posts appear
+ </label>
+ <select
+ id="x-posts-visibility"
+ value={visibility}
+ onChange={(e) => choose(e.target.value)}
+ disabled={saving}
+ className="rounded-md border border-border bg-background px-2 py-1 text-sm disabled:opacity-50"
+ >
+ <option value="public">Public — every site that has the channel</option>
+ <option value="private">Private — private sites only</option>
+ </select>
+ </div>
+ <p className="text-xs text-muted-foreground">
+ Private leaves every X channel and its posts out of every public site
+ and builds them only into sites whose audience is private, which are
+ never deployed; nothing is deleted and fetching goes on. Sites already
+ published change on their next build and deploy.
+ </p>
+ {note && (
+ <p role="status" aria-label="x posts visibility saved" className="text-xs text-success">
+ {note}
+ </p>
+ )}
+ {error && (
+ <p role="alert" className="text-xs text-destructive">
+ {error}
+ </p>
+ )}
+ </div>
+ );
+}
diff --git a/editor/app/settings/components/XSessionSection.tsx b/editor/app/settings/components/XSessionSection.tsx
@@ -26,14 +26,19 @@ import {
resolveXCookieSource,
xCookieSourceLabel,
type XCookieSourceView,
+ type XPostsVisibility,
} from "yt-dlp-transcript-common/social/xCookieSource";
+import { XPostsVisibilityControl } from "./XPostsVisibilityControl";
export function XSessionSection({
initial,
initialSource,
+ initialVisibility,
}: {
initial: XSessionStatus;
initialSource: XCookieSourceView;
+ // `social.x.visibility`, resolved (absent = "public"); release 17 slice XP.
+ initialVisibility: XPostsVisibility;
}) {
const [status, setStatus] = useState<XSessionStatus>(initial);
const [source, setSource] = useState<XCookieSourceView>(initialSource);
@@ -250,6 +255,10 @@ export function XSessionSection({
{error}
</p>
)}
+
+ <div className="border-t border-border pt-3">
+ <XPostsVisibilityControl initial={initialVisibility} />
+ </div>
</section>
);
}
diff --git a/editor/app/settings/page.tsx b/editor/app/settings/page.tsx
@@ -5,6 +5,7 @@ import { getSettings } from "yt-dlp-transcript-common/lib/settings";
import { readXSessionStatus } from "yt-dlp-transcript-common/social/xSessionBroker";
import { resolveXCookieSourceFor } from "yt-dlp-transcript-common/social/xBrowserLogin";
import { xCookieSourceView } from "yt-dlp-transcript-common/social/xCookieSource";
+import { xPostsVisibility } from "yt-dlp-transcript-common/lib/postsVisibility";
import { SettingsForm } from "./components/SettingsForm";
import { XSessionSection } from "./components/XSessionSection";
@@ -93,7 +94,11 @@ export default async function SettingsPage() {
</section>
<section className="flex flex-col gap-3 border-t border-border pt-6">
- <XSessionSection initial={xSession} initialSource={xSource} />
+ <XSessionSection
+ initial={xSession}
+ initialSource={xSource}
+ initialVisibility={xPostsVisibility(settings)}
+ />
</section>
<section className="flex flex-col gap-3 border-t border-border pt-6">
diff --git a/editor/app/settings/xSessionActions.ts b/editor/app/settings/xSessionActions.ts
@@ -30,11 +30,25 @@ import {
} from "yt-dlp-transcript-common/social/xBrowserLogin";
import {
isXCookieSource,
+ isXPostsVisibility,
xCookieSourceView,
type XCookieSourceView,
+ type XPostsVisibility,
+ type XSocialSettings,
} from "yt-dlp-transcript-common/social/xCookieSource";
+import { xPostsVisibility } from "yt-dlp-transcript-common/lib/postsVisibility";
import { saveSettings } from "./saveSettings";
+// `social.x` is ONE value to saveSettings, whose merge is one level deep: a
+// patch naming `social: { x }` replaces the whole X block. So each X choice is
+// written over the block as it is now, with only its own key changed — the
+// login source keeps the visibility, and the visibility keeps the source.
+function socialXPatch(
+ change: (x: XSocialSettings) => XSocialSettings,
+): { social: { x: XSocialSettings } } {
+ return { social: { x: change({ ...getSettings().social.x }) } };
+}
+
// Every session action returns the source in use beside the profile's status:
// connecting or forgetting a profile can move the read-time default.
export type XSessionActionResult =
@@ -124,12 +138,42 @@ export async function setXCookieSourceAction(
return { ok: false, error: `Unknown X login source "${choice}".` };
}
try {
- await saveSettings({
- social: { x: choice === "auto" ? {} : { cookieSource: choice } },
- });
+ await saveSettings(
+ socialXPatch(({ cookieSource: _was, ...rest }) =>
+ choice === "auto" ? rest : { ...rest, cookieSource: choice },
+ ),
+ );
} catch (e) {
return { ok: false, error: (e as Error).message };
}
revalidatePath("/settings");
return { ok: true, source: await sourceNow() };
}
+
+// WHERE X POSTS APPEAR — `social.x.visibility` (release 17 slice XP). "public"
+// is the default and is written as no key; "private" keeps every X channel's
+// posts out of every public site build (common/lib/postsVisibility.ts). Nothing
+// on disk changes and fetching does not: a site already published changes on
+// its next build and deploy.
+export type XPostsVisibilityResult =
+ | { ok: true; visibility: XPostsVisibility }
+ | { ok: false; error: string };
+
+export async function setXPostsVisibilityAction(
+ choice: string,
+): Promise<XPostsVisibilityResult> {
+ if (!isXPostsVisibility(choice)) {
+ return { ok: false, error: `Unknown X post visibility "${choice}".` };
+ }
+ try {
+ await saveSettings(
+ socialXPatch(({ visibility: _was, ...rest }) =>
+ choice === "public" ? rest : { ...rest, visibility: choice },
+ ),
+ );
+ } catch (e) {
+ return { ok: false, error: (e as Error).message };
+ }
+ revalidatePath("/settings");
+ return { ok: true, visibility: xPostsVisibility(getSettings()) };
+}
diff --git a/editor/app/sites/actions.ts b/editor/app/sites/actions.ts
@@ -111,6 +111,12 @@ export async function saveSiteAction(
// Listed on the homepage and hub by default: the same opt-out idiom as
// archives below (an unchecked box sends no key → persisted as false).
const listed = formData.get("listed") === "on";
+ // Who the site is built for (release 17 slice XP): only "private" is kept.
+ const audienceRaw = String(formData.get("audience") ?? "public");
+ if (audienceRaw !== "public" && audienceRaw !== "private") {
+ return { ok: false, error: `Unknown audience "${audienceRaw}".`, values };
+ }
+ const isPrivate = audienceRaw === "private";
// Hub parent (per-site override of the family default) + PWA opt-in.
const hubUrlRaw = String(formData.get("hubUrl") ?? "").trim();
@@ -248,6 +254,7 @@ export async function saveSiteAction(
...(siteUrl ? { siteUrl } : {}),
// The Site is rebuilt from the form: a key missing here is dropped on save.
...(listed ? {} : { listed: false }),
+ ...(isPrivate ? { audience: "private" as const } : {}),
...(hubUrl ? { hubUrl } : {}),
...(pwa ? { pwa: true } : {}),
...(archives ? {} : { archives: false }),
diff --git a/editor/app/sites/components/SiteForm.tsx b/editor/app/sites/components/SiteForm.tsx
@@ -16,7 +16,7 @@ import {
toSocialRow,
type SocialRow,
} from "../../components/SocialLinksField";
-import { Field } from "../../components/forms/Field";
+import { Field, SeededSelect } from "../../components/forms/Field";
import {
ControlledCheck,
ControlledSelect,
@@ -341,6 +341,25 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) {
defaultValue={initial.siteUrl ?? ""}
hint="Absolute URL this site is served at (e.g. https://jeralyzer.com). Used so other sites can link to it in their footer. Leave blank to omit this site from cross-site lists."
/>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Audience</span>
+ <SeededSelect
+ state={state}
+ name="audience"
+ aria-label="Audience"
+ initial={initial.audience === "private" ? "private" : "public"}
+ className="w-fit rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ <option value="public">Public — deployed and listed as configured</option>
+ <option value="private">Private — built for reading on this machine, never deployed</option>
+ </SeededSelect>
+ <span className="text-xs text-muted-foreground">
+ A private site is never deployed (Build & deploy and Deploy refuse
+ it; Build still builds it) and never listed on the homepage or the hub,
+ and publishes no hub URL. It is the one kind of site X posts are built
+ into while Settings keeps X posts private.
+ </span>
+ </label>
<label className="flex items-center gap-2 text-sm">
<input
type="checkbox"
diff --git a/editor/app/sites/lib/buildAction.ts b/editor/app/sites/lib/buildAction.ts
@@ -16,6 +16,10 @@ import {
} from "yt-dlp-transcript-common/controller/archiveLiveChat";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { previewBranchProblem } from "yt-dlp-transcript-common/lib/pagesDeploy";
+import {
+ deployAudienceProblem,
+ siteDeployProblem,
+} from "yt-dlp-transcript-common/lib/builtExport";
import { getSite, listSites, type Site } from "yt-dlp-transcript-common/lib/site";
import {
runManagedFunction,
@@ -132,6 +136,10 @@ export async function buildAndDeployAction(
}
const branch = previewBranch?.trim();
const site = getSite(id, paths);
+ // A private site is never deployed (site.json `audience`), so Build & deploy
+ // refuses before the build; its Build button still builds it.
+ const privateProblem = siteDeployProblem(site);
+ if (privateProblem) return { ok: false, error: `${privateProblem}.` };
if (!site.cloudflareProject) {
return {
ok: false,
@@ -155,6 +163,14 @@ export async function buildAndDeployAction(
if (buildCode !== 0) {
throw new Error(`Build failed (exit ${buildCode}) — not deploying.`);
}
+ // Asked again before the upload, over the site as it is now and the
+ // bundle just built: an audience switched to private while this job
+ // waited on the queue is not deployed.
+ const audienceProblem = deployAudienceProblem(
+ getSite(id, paths),
+ resolveOutDir(id, paths),
+ );
+ if (audienceProblem) throw new Error(`${audienceProblem} — not deploying.`);
onLog(branch ? `\n=== Deploy (preview "${branch}") ===\n` : "\n=== Deploy ===\n");
if (branch) onLog(PREVIEW_SHARES_ARCHIVES_NOTICE);
// Push oversize archives to R2 before the Pages deploy (no-op when R2
@@ -401,6 +417,13 @@ async function basicBuildAndDeployAll(
});
continue;
}
+ // A private site is built and never deployed — skipped before the upload.
+ const audienceProblem = deployAudienceProblem(site, basicOut);
+ if (audienceProblem) {
+ onLog(`[${site.siteId}] deploy skipped — ${audienceProblem}`);
+ deploys.push({ siteId: site.siteId, status: "skipped", reason: audienceProblem });
+ continue;
+ }
if (!site.cloudflareProject) {
onLog(`[${site.siteId}] deploy skipped — no Cloudflare project configured`);
deploys.push({
diff --git a/editor/app/sites/lib/deployAction.ts b/editor/app/sites/lib/deployAction.ts
@@ -1,6 +1,10 @@
"use server";
-import { builtSiteProblem } from "yt-dlp-transcript-common/lib/builtExport";
+import {
+ builtAudienceProblem,
+ builtSiteProblem,
+ siteDeployProblem,
+} from "yt-dlp-transcript-common/lib/builtExport";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { previewBranchProblem } from "yt-dlp-transcript-common/lib/pagesDeploy";
import { getSite } from "yt-dlp-transcript-common/lib/site";
@@ -38,6 +42,9 @@ export async function deployExportAction(
}
const branch = previewBranch?.trim();
const site = getSite(siteId.trim(), paths);
+ // A private site is never deployed (site.json `audience`): the first answer.
+ const privateProblem = siteDeployProblem(site);
+ if (privateProblem) return { ok: false, error: `${privateProblem}.` };
if (!site.cloudflareProject) {
return {
ok: false,
@@ -58,6 +65,10 @@ export async function deployExportAction(
const outDir = resolveOutDir(site.siteId, paths);
const builtProblem = builtSiteProblem(outDir, site.siteId);
if (builtProblem) return { ok: false, error: builtProblem };
+ const builtPrivate = builtAudienceProblem(outDir);
+ if (builtPrivate) {
+ return { ok: false, error: `${builtPrivate}. Build ${site.siteId} again, then deploy.` };
+ }
return runManagedFunction({
kind: "deploy-export",
queueKey: DEPLOY_QUEUE,
diff --git a/editor/e2e/bulk-actions.spec.ts b/editor/e2e/bulk-actions.spec.ts
@@ -165,9 +165,15 @@ test("Select failed + Delete directories removes the dirs and queues no job", as
await expect(page.getByLabel("select vidB")).toBeVisible();
// Crucially: NO managed job was queued (no transcode/download triggered).
+ // The report generateReport asked for is a refresh-report job of its own
+ // (release 17: every report walk runs on the refresh-report queue), and
+ // /jobs may draw it before its default filter hides it — not this action's.
await page.goto("/jobs");
await expect(
- page.getByRole("row").filter({ hasText: "test-transcribe" }),
+ page
+ .getByRole("row")
+ .filter({ hasText: "test-transcribe" })
+ .filter({ hasNotText: "refresh-report" }),
).toHaveCount(0);
});
diff --git a/editor/e2e/dashboard-answers.spec.ts b/editor/e2e/dashboard-answers.spec.ts
@@ -0,0 +1,356 @@
+import { link, mkdir, readFile, writeFile } from "node:fs/promises";
+import { join } from "node:path";
+import { test, expect, type APIRequestContext, type Page } from "@playwright/test";
+import { ulid } from "yt-dlp-transcript-common/jobs/ulid";
+import { baseUrl } from "./baseUrl";
+import {
+ channelStage,
+ generateReport,
+ resetData,
+ resolvePath,
+} from "./helpers";
+
+// THE DASHBOARD ANSWERS WHILE A REPORT REGENERATES (release 17 slice D0).
+//
+// On 2026-10-01 the live editor's `/`, `/channels` and `/jobs` gave no
+// response for over an hour while two `refresh-report` jobs walked 2,000- and
+// 3,260-video channels in its own process, side by side, and three more
+// `refresh-report` metas still read `running` hours after the process that ran
+// them was gone. The first case below is the first half: two channels whose
+// reports take several seconds each, regenerated by "Update all reports" — one
+// after the other, never side by side — with `/` and `/jobs` polled every 2 s
+// throughout. The second is the ghost: a `running` meta a dead process left
+// behind is closed as interrupted by the boot pass, one a live process owns is
+// not, and neither stops a media move.
+//
+// THE BUDGET IS 5 s, OR THREE TIMES WHAT THE SAME PAGE TOOK JUST BEFORE THE
+// REGENERATION, WHICHEVER IS LONGER. The suite runs under `next dev` on a
+// machine other suites and builds share; at a load average of 30 a page that
+// renders in 0.3 s on a quiet machine takes 2–4 s with nothing regenerating at
+// all. What this pins is that the regeneration does not starve the pages, not
+// how fast the machine is — so the idle measurement taken a moment before sets
+// the floor, and on a quiet machine the budget is simply 5 s. CAPPED AT 15 s:
+// on a machine so loaded that idle pages take over 5 s, the floor stops
+// growing, and the case fails rather than stretching to hide starvation.
+//
+// What the case pins, honestly: the serial queue (read from the jobs' own
+// records) and starvation at the scale of seconds. The yield between chunks is
+// pinned by a unit test (common/controller/snapshotYield.test.ts): on a dev
+// server under load `main`'s walk answered inside such a budget too.
+
+const OPS_AUTH = { authorization: "Bearer test-worker-token" };
+
+// THE BIG CHANNELS. Every video dir hardlinks ONE ~400 KB metadata.info.json
+// — the size of a long VOD's, the file the walk parses — so 600 of them cost
+// one file's bytes and a second to make. No `webpage_url`: the reconcile pass
+// at the walk's start parses each file for it and, finding none, renames
+// nothing (one shared id would merge every dir into one). The archive names a
+// non-YouTube extractor, so the walk parses each file a second time for its
+// native id — the Rumble-channel path. 600 is several seconds of walk under
+// `next start` on a quiet machine and well over a minute under `next dev` at a
+// load average of 30; 2,000 did not finish inside five minutes there.
+const BIG = ["big-channel-a", "big-channel-b"];
+const BIG_VIDEOS = 600;
+
+async function seedBigChannel(slug: string): Promise<void> {
+ const channelDir = resolvePath(`test-transcripts/channels/${slug}`);
+ const dataDir = join(channelDir, "data");
+ await mkdir(dataDir, { recursive: true });
+ await writeFile(
+ join(channelDir, "config.json"),
+ JSON.stringify({ handling: "youtube", name: "Big Channel" }),
+ );
+ const formats = Array.from({ length: 400 }, (_, i) => ({
+ format_id: `hls-${i}`,
+ url: `https://example.invalid/${"x".repeat(600)}${i}`,
+ ext: "mp4",
+ protocol: "m3u8_native",
+ width: 1280,
+ height: 720,
+ tbr: 1234.5 + i,
+ http_headers: { "User-Agent": `Mozilla/5.0 ${"y".repeat(80)}`, Accept: "*/*" },
+ fragments: [
+ { url: "seg0", duration: 6 },
+ { url: "seg1", duration: 6 },
+ ],
+ }));
+ const template = resolvePath(`test-transcripts/.${slug}-meta.json`);
+ await writeFile(
+ template,
+ JSON.stringify({ id: "native", title: "A long VOD", duration: 3600, formats }),
+ );
+ const ids = Array.from(
+ { length: BIG_VIDEOS },
+ (_, i) => `${slug.slice(-1)}big${String(i).padStart(8, "0")}`,
+ );
+ for (let i = 0; i < ids.length; i += 100) {
+ await Promise.all(
+ ids.slice(i, i + 100).map(async (id) => {
+ const dir = join(dataDir, id);
+ await mkdir(dir);
+ await link(template, join(dir, "metadata.info.json"));
+ }),
+ );
+ }
+ await writeFile(
+ join(channelDir, "archive"),
+ ids.map((id) => `rumble ${id}\n`).join(""),
+ );
+ await writeFile(join(channelDir, "playlist"), "");
+}
+
+type Sample = { path: string; ms: number; status: number };
+
+async function timedGet(
+ request: APIRequestContext,
+ path: string,
+): Promise<Sample> {
+ const t = Date.now();
+ const res = await request.get(`${baseUrl}${path}`, { timeout: 60_000 });
+ // The body too: a page that sends its head and stalls is not an answer.
+ await res.body();
+ return { path, ms: Date.now() - t, status: res.status() };
+}
+
+type JobMetaOnDisk = {
+ id: string;
+ kind: string;
+ channelSlug?: string;
+ status: string;
+ startedAt?: number;
+ endedAt?: number;
+};
+
+test("/ and /jobs answer while two large reports regenerate, one after the other", async ({
+ request,
+}) => {
+ test.setTimeout(420_000);
+ await resetData("empty");
+ for (const slug of BIG) await seedBigChannel(slug);
+
+ // Warm both routes first: under `next dev` the first request compiles the
+ // page, which is the dev server's cost and not what this measures. Then the
+ // idle floor: the slowest of three more rounds, with nothing regenerating.
+ for (const path of ["/", "/jobs"]) {
+ expect((await timedGet(request, path)).status).toBe(200);
+ }
+ // And the ops route that starts the regeneration: its first request compiles
+ // it, which under load held `/` for 15 s in one full-suite run — a dev
+ // server's compile, not a walk. A body naming neither form is refused (400)
+ // after the route has loaded, and starts nothing.
+ const warm = await request.post(`${baseUrl}/api/ops/refresh-report`, {
+ headers: OPS_AUTH,
+ data: {},
+ timeout: 120_000,
+ });
+ expect(warm.status()).toBe(400);
+ const idle: Sample[] = [];
+ for (let i = 0; i < 3; i++) {
+ for (const path of ["/", "/jobs"]) idle.push(await timedGet(request, path));
+ }
+ const budgetMs = Math.min(
+ Math.max(5_000, 3 * Math.max(...idle.map((s) => s.ms))),
+ 15_000,
+ );
+
+ // "Update all reports", through the ops API: a refresh-report job per
+ // channel on the serial queue. The call answers once they are QUEUED, with
+ // their ids; "while they regenerate" lasts until both jobs' records say
+ // they have ended.
+ const startedAt = Date.now();
+ const res = await request.post(`${baseUrl}/api/ops/refresh-report`, {
+ headers: OPS_AUTH,
+ data: { all: true },
+ timeout: 120_000,
+ });
+ expect(res.status()).toBe(200);
+ const done = { body: (await res.json()) as { queued?: string[]; jobIds?: string[] } };
+ expect([...(done.body.queued ?? [])].sort()).toEqual(BIG);
+ expect(done.body.jobIds).toHaveLength(2);
+ const ended = async (): Promise<boolean> => {
+ for (const id of done.body.jobIds ?? []) {
+ const m = await readFile(resolvePath(`test-transcripts/.jobs/${id}.meta.json`), "utf8")
+ .then((raw) => JSON.parse(raw) as JobMetaOnDisk)
+ .catch(() => null);
+ if (!m || m.status === "queued" || m.status === "running") return false;
+ }
+ return true;
+ };
+
+ let finishedAt: number | null = null;
+ const during: Sample[] = [];
+ while (finishedAt === null) {
+ if (Date.now() - startedAt > 360_000) throw new Error("the regenerations did not end in 6 min");
+ const tick = Date.now();
+ for (const path of ["/", "/jobs"]) {
+ const s = await timedGet(request, path);
+ during.push(s);
+ console.log(`[dashboard-answers] +${tick - startedAt} ms ${path} ${s.ms} ms`);
+ }
+ if (await ended()) {
+ finishedAt = Date.now();
+ break;
+ }
+ const wait = 2_000 - (Date.now() - tick);
+ if (wait > 0) await new Promise((r) => setTimeout(r, wait));
+ }
+ for (const slug of BIG) {
+ const snapshot = JSON.parse(
+ await readFile(
+ resolvePath(`test-transcripts/channels/${slug}/snapshot.json`),
+ "utf8",
+ ),
+ ) as { totals: { videos: number } };
+ expect(snapshot.totals.videos).toBe(BIG_VIDEOS);
+ }
+
+ // ONE AFTER THE OTHER: the second started when the first had ended. Read
+ // off the jobs' records on disk; the terminal write lands just after the
+ // job's stream closes, so it is waited for.
+ const readMetas = () =>
+ Promise.all(
+ (done.body.jobIds ?? []).map(
+ async (id) =>
+ JSON.parse(
+ await readFile(resolvePath(`test-transcripts/.jobs/${id}.meta.json`), "utf8"),
+ ) as JobMetaOnDisk,
+ ),
+ );
+ await expect
+ .poll(async () => (await readMetas()).map((m) => `${m.kind} ${m.status}`), {
+ timeout: 10_000,
+ })
+ .toEqual(["refresh-report done", "refresh-report done"]);
+ const metas = await readMetas();
+ const [first, second] = [...metas].sort(
+ (a, b) => (a.startedAt ?? 0) - (b.startedAt ?? 0),
+ );
+ expect(second.startedAt ?? 0).toBeGreaterThanOrEqual(first.endedAt ?? Infinity);
+
+ // The fixture has to make the walk long enough to be polled through, or the
+ // budget below proves nothing.
+ const regenMs = (finishedAt ?? Date.now()) - startedAt;
+ test.info().annotations.push({
+ type: "timings",
+ description:
+ `idle max ${Math.max(...idle.map((s) => s.ms))} ms, budget ${budgetMs} ms; ` +
+ `regeneration ${regenMs} ms; ${during.map((s) => `${s.path} ${s.ms}`).join(", ")}`,
+ });
+ expect(regenMs, "the regenerations took several seconds").toBeGreaterThan(4_000);
+ for (const path of ["/", "/jobs"]) {
+ expect(
+ during.filter((s) => s.path === path).length,
+ `${path} was polled during the regeneration`,
+ ).toBeGreaterThanOrEqual(2);
+ }
+ for (const s of during) {
+ expect(s.status, s.path).toBe(200);
+ expect(s.ms, `${s.path} answered in ${s.ms} ms (budget ${budgetMs} ms)`).toBeLessThan(budgetMs);
+ }
+});
+
+// ---------------------------------------------------------------------------
+// The ghost.
+
+const SLUG = "test-youtube";
+
+async function quiet(page: Page): Promise<void> {
+ await expect
+ .poll(
+ async () => {
+ const res = await page.request.get(`${baseUrl}/api/jobs/active`);
+ const body = await res.json();
+ const jobs: { channelSlug?: string; status: string }[] = Array.isArray(
+ body,
+ )
+ ? body
+ : (body.jobs ?? []);
+ return jobs.filter(
+ (j) =>
+ j.channelSlug === SLUG &&
+ (j.status === "running" || j.status === "queued"),
+ ).length;
+ },
+ { timeout: 30_000 },
+ )
+ .toBe(0);
+}
+
+// A `running` refresh-report meta for SLUG, as a process that died mid-walk
+// leaves it: queued and started an hour ago, its log last written then.
+async function plantGhost(pid: number): Promise<string> {
+ const id = ulid(Date.now() - 60 * 60 * 1000);
+ const jobsDir = resolvePath("test-transcripts/.jobs");
+ await mkdir(jobsDir, { recursive: true });
+ const at = Date.now() - 60 * 60 * 1000;
+ await writeFile(
+ join(jobsDir, `${id}.meta.json`),
+ JSON.stringify({
+ id,
+ kind: "refresh-report",
+ queueKey: "",
+ channelSlug: SLUG,
+ status: "running",
+ queuedAt: at,
+ startedAt: at,
+ pid,
+ }),
+ );
+ await writeFile(
+ join(jobsDir, `${id}.log`),
+ `Regenerating report for ${SLUG}…\n`,
+ );
+ return id;
+}
+
+async function metaOf(id: string): Promise<{ status: string; cancelReason?: string }> {
+ return JSON.parse(
+ await readFile(resolvePath(`test-transcripts/.jobs/${id}.meta.json`), "utf8"),
+ );
+}
+
+test("a ghost running meta from a dead process is closed as interrupted, and does not block a move", async ({
+ page,
+}, testInfo) => {
+ test.setTimeout(180_000);
+ await resetData("one-youtube-channel-with-data");
+ await generateReport(page, SLUG);
+ await quiet(page);
+
+ // Past pid_max: no process has it. And this test runner's own pid: a live
+ // process that is not the editor — an `archilyzer run` beside it.
+ const ghost = await plantGhost(2 ** 22 + 1);
+ const live = await plantGhost(process.pid);
+
+ const res = await page.request.post(`${baseUrl}/api/test/settle-running-metas`);
+ expect(res.ok()).toBe(true);
+ const { interrupted } = (await res.json()) as { interrupted: { id: string }[] };
+ expect(interrupted.map((j) => j.id)).toContain(ghost);
+ expect(interrupted.map((j) => j.id)).not.toContain(live);
+
+ const closed = await metaOf(ghost);
+ expect(closed.status).toBe("cancelled");
+ expect(closed.cancelReason).toMatch(/^interrupted: /);
+ expect((await metaOf(live)).status).toBe("running");
+
+ // /jobs says why, on the job's own page.
+ await page.goto(`/jobs/${ghost}`);
+ await expect(page.getByTestId("cancel-reason")).toContainText("interrupted");
+
+ // Neither the closed ghost nor the live one is a writer on the channel: the
+ // Storage panel offers the move and the move completes.
+ const root = testInfo.outputPath("ghost-root");
+ await mkdir(root, { recursive: true });
+ await page.goto(channelStage(SLUG, "storage"));
+ await page.getByLabel("destination root").fill(root);
+ await page.getByRole("button", { name: "Preview", exact: true }).click();
+ await expect(page.getByLabel("relocation preview")).toBeVisible({
+ timeout: 15_000,
+ });
+ const moveButton = page.getByRole("button", { name: "Move media" });
+ await expect(moveButton).toBeEnabled();
+ await moveButton.click();
+ await expect(page.getByLabel("Move media output")).toContainText("Moved", {
+ timeout: 60_000,
+ });
+});
diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts
@@ -501,23 +501,46 @@ test("refresh-report regenerates snapshot.json", async ({ request }) => {
await rm(resolvePath(SNAP), { force: true });
expect(await pathExists(SNAP)).toBe(false);
- // No page, no click: the route IS the refresh. It regenerates SYNCHRONOUSLY
- // (a filesystem scan, not a job), so { ok: true } means the file is there.
+ // No page, no click: the route IS the refresh. It answers once the
+ // regeneration is QUEUED (release 17: one serial refresh-report queue, and
+ // the ops rule — a job-starting route returns a jobId), so the job is
+ // followed to `done` before the file is read.
+ const doneJob = async (jobId: string | undefined) => {
+ expect(jobId).toBeTruthy();
+ await expect
+ .poll(
+ async () =>
+ (
+ await readJson<{ status: string }>(
+ `test-transcripts/.jobs/${jobId}.meta.json`,
+ ).catch(() => null)
+ )?.status ?? null,
+ { timeout: 30_000 },
+ )
+ .toBe("done");
+ };
const first = await ops(request, "refresh-report", { slug: SLUG });
- expect(first.body).toEqual({ ok: true });
+ expect(first.body.ok).toBe(true);
+ await doneJob(first.body.jobId);
const snapshot = await readJson<{ generatedAt: string; totals: { videos: number } }>(SNAP);
expect(snapshot.generatedAt).toBeTruthy();
expect(snapshot.totals.videos).toBeGreaterThanOrEqual(0);
- // Re-running REWRITES it. Polled through the action itself because two scans
- // of a six-video fixture can land in the same millisecond.
+ // Re-running REWRITES it. Polled because two scans of a six-video fixture
+ // can land in the same millisecond.
await expect
.poll(async () => {
- await ops(request, "refresh-report", { slug: SLUG });
+ const again = await ops(request, "refresh-report", { slug: SLUG });
+ await doneJob(again.body.jobId);
return (await readJson<{ generatedAt: string }>(SNAP)).generatedAt;
})
.not.toBe(snapshot.generatedAt);
+ // An unknown channel is a 404 naming it, and starts nothing.
+ const missing = await ops(request, "refresh-report", { slug: "no-such-channel" });
+ expect(missing.status).toBe(404);
+ expect(missing.body.error).toMatch(/no-such-channel/);
+
// The bulk form queues a job per channel and reports both lists.
const all = await ops(request, "refresh-report", { all: true });
expect(all.status).toBe(200);
diff --git a/editor/e2e/sites-crud.spec.ts b/editor/e2e/sites-crud.spec.ts
@@ -528,3 +528,51 @@ test("brand accent radio group + wordmark lead round-trip to site.json", async (
expect("wordmarkLead" in site).toBe(false);
}).toPass({ timeout: 10_000 });
});
+
+// Who a site is built for — `site.json` `audience` (release 17 slice XP). A
+// private site is the operator's own reading copy: saved from the form, and
+// never deployed — the deploy actions refuse it before a job exists, in words
+// naming the audience. (The ops routes call the same actions the Publish tab's
+// buttons do.)
+test("a private site saves its audience, and every deploy refuses it before any job", async ({
+ page,
+ request,
+}) => {
+ await resetData("empty");
+ await writeSite("privsite", { siteTitle: "Private Site", cloudflareProject: "never-real" });
+
+ await page.goto("/sites/privsite");
+ const audience = page.getByLabel("Audience", { exact: true });
+ await expect(audience).toHaveValue("public");
+ await audience.selectOption("private");
+ await page.getByRole("button", { name: /save site/i }).click();
+ await expect(page.getByRole("status").filter({ hasText: "Saved" })).toBeVisible();
+ await expect(async () => {
+ const site = await readJson<{ audience?: string; cloudflareProject?: string }>(
+ "test-transcripts/sites/privsite/site.json",
+ );
+ expect(site.audience).toBe("private");
+ expect(site.cloudflareProject).toBe("never-real");
+ }).toPass({ timeout: 10_000 });
+ await page.reload();
+ await expect(audience).toHaveValue("private");
+
+ const refusal =
+ 'Site "privsite" is private (audience: private): it is built for reading on this machine and is never deployed. Build it without deploying, or set its audience to public on its Settings tab.';
+ for (const action of ["build-deploy", "deploy-site"]) {
+ const res = await request.post(`/api/ops/${action}`, {
+ headers: { authorization: "Bearer test-worker-token" },
+ data: { siteId: "privsite" },
+ });
+ expect(res.status(), action).toBe(400);
+ expect(((await res.json()) as { error?: string }).error, action).toBe(refusal);
+ }
+
+ // Back to public: the key is gone from the file (public is the default).
+ await audience.selectOption("public");
+ await page.getByRole("button", { name: /save site/i }).click();
+ await expect(async () => {
+ const site = await readJson<Record<string, unknown>>("test-transcripts/sites/privsite/site.json");
+ expect("audience" in site).toBe(false);
+ }).toPass({ timeout: 10_000 });
+});
diff --git a/editor/e2e/x-session.spec.ts b/editor/e2e/x-session.spec.ts
@@ -76,3 +76,44 @@ test("the login source select persists, and Check shows a status line", async ({
await expect(inUse).toHaveText(`In use: Browser login (${spec}) (automatic)`);
expect((await readJson<{ social?: unknown }>("test-settings.json")).social).toEqual({ x: {} });
});
+
+// Where X posts appear — `social.x.visibility` (release 17 slice XP). The two X
+// choices share one settings block, and saveSettings replaces a nested block
+// whole, so each is written over the other: choosing one keeps the other.
+test("where X posts appear persists, beside the login source and without it", async ({ page }) => {
+ await resetData("empty");
+ await page.goto("/settings");
+ const visibility = page.getByLabel("Where X posts appear");
+ const source = page.getByLabel("x cookie source", { exact: true });
+ const social = async () =>
+ (await readJson<{ social?: unknown }>("test-settings.json")).social;
+
+ await expect(visibility).toHaveValue("public");
+ await expect(page.locator("[data-x-posts-visibility]")).toContainText(
+ "Sites already published change on their next build and deploy.",
+ );
+
+ await source.selectOption("profile");
+ await expect(page.getByLabel("x cookie source in use")).toHaveText("In use: Connected profile");
+
+ await visibility.selectOption("private");
+ await expect(page.getByLabel("x posts visibility saved")).toHaveText(
+ "Saved. Published sites change on their next build and deploy.",
+ );
+ expect(await social()).toEqual({ x: { cookieSource: "profile", visibility: "private" } });
+
+ await page.reload();
+ await expect(visibility).toHaveValue("private");
+
+ // The login source back to automatic keeps the visibility…
+ await source.selectOption("auto");
+ await expect(page.getByLabel("x cookie source in use")).toHaveText(
+ "In use: Connected profile (automatic)",
+ );
+ expect(await social()).toEqual({ x: { visibility: "private" } });
+
+ // …and public is the default, written as no key.
+ await visibility.selectOption("public");
+ await expect(page.getByLabel("x posts visibility saved")).toBeVisible();
+ expect(await social()).toEqual({ x: {} });
+});
diff --git a/editor/instrumentation.ts b/editor/instrumentation.ts
@@ -146,7 +146,7 @@ export async function register() {
// common/jobs/bootQueuedJobs.ts. Lazy, voided, best-effort: never blocks
// readiness.
try {
- const { settleAfterStoragePass } = await import(
+ const { settleAfterStoragePass, settleRunningJobMetas } = await import(
"yt-dlp-transcript-common/jobs/bootQueuedJobs"
);
const { getPaths } = await import("yt-dlp-transcript-common/lib/paths");
@@ -154,6 +154,17 @@ export async function register() {
"yt-dlp-transcript-common/jobs/registry"
);
const testServer = process.env.E2E_TEST_ROUTES === "1";
+ // STALE `running` METAS FROM A DEAD PROCESS (release 17 slice D0): closed
+ // `cancelled` as interrupted, never re-run — on every boot, idle and test
+ // server included, because closing one starts nothing. Does not wait for
+ // the storage pass: it touches no channel. A meta another live process
+ // still owns (`archilyzer run`) is left alone — see writerIsGone.
+ void settleRunningJobMetas({
+ paths: getPaths(),
+ bootedAt,
+ isLive: (id) => getRegistry().get(id) !== undefined,
+ log: (line) => console.log(line),
+ }).catch(() => {});
const cancelOnly = idle || testServer;
void settleAfterStoragePass(storagePass, {
paths: getPaths(),
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,6 @@
# Changelog
-## [Unreleased]
+## [0.11.1] - 2026-10-01
- **Use with AI goes to the Archilyzer site's AI and MCP doc; the page on each site is gone.** The header's, the slide-out menu's, the footer's and Ask AI's **Use with AI** keep their label and open https://archilyzer.pages.dev/docs/ai-and-mcp/ in the same tab, on every site and the hub, where one block says how to run Claude Code against any archive (the source, `pnpm install`, `claude mcp add archilyzer`, `/ask`). `/use-with-ai/` is no longer built. `corpus.json`'s `useWithAi` names the doc; `llms.txt`'s Ask AI section lists the site's `/ask/` chat and the doc; the sitemap drops `/use-with-ai`. Needs a rebuild and deploy of each site and the hub.
- **A search with a layer that has nothing to read finishes.** A "Posts" layer under a tag chip, or a "Live chat" layer where no video in the selection has live chat, read "searched N/M…" for ever and never said "No matching videos."; it now finishes at once, having matched nothing. Needs a rebuild and deploy of each site and the hub.
- **A search reads what the visitor ticks under "Search in": Transcripts, Posts and Live chat.** The Filters panel has a new row, **Search in**, beside Type. **Transcripts** and **Posts** are ticked by default and **Live chat** is not; Posts is offered only on a site that has posts, and Live chat only on a site with live chat. The row decides what a plain query reads: with Posts ticked, a plain query now finds posts as well as videos (before, a post was found only by a layer whose scope was "Posts"); with Live chat ticked, it finds live-chat messages too, shown in the same video's card beside the transcript hits, each marked "live chat"; with Transcripts unticked it reads no transcripts. A layer whose scope is picked by name in the query builder ("Live chat", "Posts", "Title / channel", …) reads what it names, whatever the row says. An empty query still lists every video the Type row keeps. With nothing ticked, Search and Apply filters are disabled and the row says "Search in: pick at least one". The Posts box moved here from the Type row, and unticking it no longer empties a layer whose scope is "Posts". Under a tag chip a plain query reads no posts, since a post carries no tags. The row is remembered, and saved with a profile; a shared link does not carry it, so it opens with the reader's own row. A live-chat hit now wears its "live chat" badge wherever it is shown, and the hint under the search bar says to tick Live chat under Search in. Posts unticked is now also remembered after a reload and restored with a profile, which it was not. Needs a rebuild and deploy of each site and the hub.
diff --git a/export/app/offline/page.tsx b/export/app/offline/page.tsx
@@ -1,6 +1,8 @@
import type { Metadata } from "next";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { postsVisibleTo } from "yt-dlp-transcript-common/lib/postsVisibility";
import { currentSite } from "../lib/site";
import { OfflineManager, type OfflineChannel } from "../components/OfflineManager";
@@ -15,12 +17,19 @@ export const metadata: Metadata = {
export default async function OfflinePage() {
const site = currentSite();
const paths = getPaths();
- const channels: OfflineChannel[] = await Promise.all(
- site.channels.map(async ({ slug }) => {
- const config = await readChannelConfig(paths, slug).catch(() => null);
- return { slug, name: config?.name ?? slug };
- }),
+ const settings = getSettings();
+ // The members this build publishes: an X channel is left out of a public
+ // site while X posts are private (lib/postsVisibility.ts, the rule the
+ // index build and compose narrow the site's data by).
+ const members = await Promise.all(
+ site.channels.map(async ({ slug }) => ({
+ slug,
+ config: await readChannelConfig(paths, slug).catch(() => null),
+ })),
);
+ const channels: OfflineChannel[] = members
+ .filter(({ config }) => postsVisibleTo(site, config, settings))
+ .map(({ slug, config }) => ({ slug, name: config?.name ?? slug }));
channels.sort((a, b) => a.name.localeCompare(b.name));
return (
diff --git a/export/e2e/x-posts-private.spec.ts b/export/e2e/x-posts-private.spec.ts
@@ -0,0 +1,104 @@
+import { expect, test, type Page } from "@playwright/test";
+import {
+ POST_CHANNEL,
+ POST_CHANNEL_SLUG,
+ POST_REPLY_ID,
+ POST_ROOT_ID,
+ postsPage,
+} from "./fixtures/data";
+import { installRoutes, openFilters } from "./helpers";
+
+// X posts are private (release 17 slice XP): with `social.x.visibility`
+// "private", a PUBLIC site's build carries no X channel and a PRIVATE site's
+// build carries all of it. The build itself — the index build's per-site posts
+// manifest and compose's posts tree — is pinned through the real code by
+// common/bin/compose-site.postsVisibility.test.ts: this suite's data is
+// route-mocked, never built. Here the two posts manifests that build writes
+// are served, and the visitor's side is checked: the Search in row's Posts box
+// and the post hits come and go with the manifest, with nothing special-cased.
+//
+// The fixture posts say "kappa" (two of them); no video does.
+
+const X_POST_SLUGS = [
+ `${POST_CHANNEL_SLUG}/${POST_ROOT_ID}`,
+ `${POST_CHANNEL_SLUG}/${POST_REPLY_ID}`,
+];
+
+const fulfillJson = (body: unknown) => ({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(body),
+});
+
+// The site posts manifest compose writes: the X channel listed (a private
+// site's build), or no channel at all (a public site whose only posts were X
+// posts). Routed after installRoutes, so these answers win.
+async function servePostsManifest(page: Page, built: "public" | "private") {
+ await page.route("**/posts/manifest.json", (route) =>
+ route.fulfill(
+ fulfillJson({
+ version: 1,
+ channels:
+ built === "private"
+ ? [{ name: POST_CHANNEL, slug: POST_CHANNEL_SLUG, postCount: 4, platform: "twitter" }]
+ : [],
+ totalCount: built === "private" ? 4 : 0,
+ generatedAt: new Date().toISOString(),
+ }),
+ ),
+ );
+ await page.route(/\/posts\/[^/]+\/page-\d+\.json$/, (route) =>
+ route.fulfill(
+ fulfillJson(
+ postsPage().map((p) => ({
+ ...p,
+ platform: "twitter",
+ url: `https://x.com/tester/status/${p.id}`,
+ })),
+ ),
+ ),
+ );
+}
+
+const postsBox = (page: Page) =>
+ page.getByTestId("search-in-row").getByRole("checkbox", { name: "Posts", exact: true });
+
+async function search(page: Page, q: string) {
+ await page.locator('input[data-testid^="leaf-query-"]').first().fill(q);
+ await page.getByTestId("search-submit").click();
+}
+
+test.describe("X posts private", () => {
+ test.beforeEach(async ({ page }) => {
+ await installRoutes(page);
+ });
+
+ test("a public site built with X posts private has no Posts box and finds no X post", async ({
+ page,
+ }) => {
+ await servePostsManifest(page, "public");
+ await page.goto("/");
+ await openFilters(page);
+ await expect(page.getByRole("checkbox", { name: "Transcripts", exact: true })).toBeVisible();
+ await expect(postsBox(page)).toHaveCount(0);
+ await search(page, "kappa");
+ await expect(page.getByText(/^searched \d+\/\d+$/)).toBeVisible({ timeout: 15_000 });
+ await expect(page.getByTestId("results-summary")).toHaveText("Matching videos (0)");
+ await expect(page.locator(`[data-result-slug^="${POST_CHANNEL_SLUG}/"]`)).toHaveCount(0);
+ });
+
+ test("a private site's build shows the X posts", async ({ page }) => {
+ await servePostsManifest(page, "private");
+ await page.goto("/");
+ await openFilters(page);
+ await expect(postsBox(page)).toBeChecked();
+ await search(page, "kappa");
+ const cards = page.locator("[data-card-header]");
+ await expect(async () => {
+ const got = await cards.evaluateAll((els) =>
+ els.map((e) => e.getAttribute("data-result-slug") ?? ""),
+ );
+ expect(got.slice().sort()).toEqual(X_POST_SLUGS.slice().sort());
+ }).toPass({ timeout: 15_000 });
+ });
+});
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -165,7 +165,7 @@ Never name the curated field `tags`. Never assume a `tags.json` is the keyword l
| --- | --- | --- |
| A video's visibility | `common/lib/availability.ts` (the `"unlisted"` state, `isUnlisted`), `common/lib/transcripts{,-server}.ts`, `common/components/shareUrl.ts`, `common/controller/buildIndex.ts` | The platform's own "unlisted" (reachable by link, not listed on the channel). |
| The hub's list has loaded | `export/app/components/hub/useHubSites.ts` — `listed` | `/hub-sites.json` has been answered and `/hub-summary.json` has settled. |
-| **A site the family lists** | `site.json` `listed` (`common/lib/siteSchema.ts` — `isListedSite`, `channelsOnlyOnUnlistedSites`) | Absent = listed; `false` keeps the site off the homepage, the hub and the other sites' footers, and out of the public totals. |
+| **A site the family lists** | `site.json` `listed` (`common/lib/siteSchema.ts` — `isListedSite`, `channelsOnlyOnUnlistedSites`) | Absent = listed; `false` keeps the site off the homepage, the hub and the other sites' footers, and out of the public totals. A PRIVATE site (`audience: "private"`, release 17 XP) is never listed, whatever `listed` says. |
A grep for either word finds all three; read the file before assuming which.
@@ -8251,6 +8251,26 @@ phase deletes from the destination.
**The source, not `cookieMode`, governs the X fetchers**: the browser source passes the spec
whatever the mode; the profile source keeps the old order (the jar, else the `"always"`-mode
spec, else a guest run).
+- **Where X posts may appear, `social.x.visibility`** (release 17 slice XP; `"public"` default |
+ `"private"`, kept by `sanitizeSocial` beside `cookieSource`). **saveSettings' merge is one level
+ deep, so a patch `{ social: { x } }` replaces the whole X block**: both X actions
+ (`xSessionActions.ts`) write over the block as it is (`socialXPatch`). The rule is
+ `common/lib/postsVisibility.ts` (`postsVisibleTo`, `publishedMemberSlugs`): while private, an X
+ channel (social, platform `twitter`) is built only into sites with `site.json` `audience:
+ "private"`; a public site leaves the channel out WHOLE (posts are all it holds) — its posts
+ manifest entry, posts tree, transcripts tree, channel list, `site.json` and `corpus.json` entry,
+ and `channel-sites.json`. Applied in `buildIndex`'s per-site loop (the site fingerprint names the
+ `withheld` members) and in `compose-site` (which prunes a previously shipped tree); the shared
+ posts tree and the LMDB posts sub-DB stay corpus-wide. A private site publishes no `hubUrl`
+ (`resolveHubUrl`), its `corpus.json` says `site.audience: "private"`, and every deploy path
+ refuses it or its bundle before any upload (`lib/builtExport.ts` `deployAudienceProblem`; the
+ bulk deploys skip it). Build & deploy all still BUILDS it.
+- **The hub carries no site's data** (release 17 XP review): `public/` is shared, so a hub built after
+ a site's compose used to ship that site's data trees. `compose-hub` now removes every per-site entry
+ first (`SITE_ONLY_PUBLIC_ENTRIES`) and writes the global `search-aliases.json`; `builtHubProblem`
+ refuses a hub bundle carrying a data tree. **compose-site's own stages (summaries, stats,
+ duplicates) are trusted from its cache only when `public/site.json` names the site**; the
+ per-channel trees keep their signatures across sites.
- **gallery-dl 1.32.9** takes `--cookies-from-browser BROWSER[/DOMAIN][+KEYRING][:PROFILE][::CONTAINER]`
(yt-dlp's syntax plus `/DOMAIN`) and reads every browser it supports, Chromium's encrypted store
included, on each run. The spec is passed verbatim (`galleryDlCookieChoice`).
diff --git a/plans/release-17.md b/plans/release-17.md
@@ -263,6 +263,7 @@ one short Transcribe (the hook on a relocated channel), `/storage`, `df`; then n
| **T3** migration + records | `r17/media-tier-migrate` | `common/bin/migrate-media-tier.ts`, `archilyzer.ts` wiring, fixture tests (tmp "platter"), FACTS "A channel's media is tiered", AGENTS.md's six things → seven, SETTINGS.md/CHANNEL.md regen, the release record, changelog | T1, T2 | dry run; resume from each phase; idempotent rerun; `--reclaim`; refusal on a marker; the free-space stop |
| **U1** umtool roots + `out/` | `r17/umtool-media-root` | `paths.mjs` (`MEDIA_ROOT`, `CACHE_DIR`), `lib/report/storage.mjs`, `driver.mjs`, `build-video.mjs:2615`, `export.mjs`, `kinds.mjs`, `umtool doctor`, `umtool storage move-out`, the e2e env | — (∥ T1) | `test:scripts` (+ mover tests), `next-build-trace.test.mjs`, the capped umtool build with the corpus linked, umtool e2e |
| **U2** deliverables switch | `r17/umtool-deliverables` | manifest `storage` field, `deliverableDir`, `cut.mjs:96`, `deliver.mjs:362`, `umtool storage deliverables`, bench "Move deliverables", `umtool check` | U1 | umtool unit + e2e: cut and share through a linked `clips/` |
+| **XP** X posts are private (operator-requested, beside the media tier) | `r17/x-posts-private` | `common/lib/postsVisibility.ts` (new) + test, `settingsSchema.ts` + `social/xCookieSource.ts` (`social.x.visibility`) + SETTINGS.md, `siteSchema.ts` + `site.ts` (`audience`, `isListedSite`, `resolveHubUrl`) + SITE.md, `buildIndex.ts` (the per-site loop only), `bin/compose-site.ts` + an integration test, `lib/corpus.ts`, `lib/builtExport.ts`, `publish/build.ts` + tests, `controller/poolSummary.ts` + test, `docker/publish-site.sh`, `export/app/offline/page.tsx`, editor `settings/{xSessionActions.ts,page.tsx,components/{XSessionSection,XPostsVisibilityControl}.tsx}`, `sites/{actions.ts,components/SiteForm.tsx,lib/{buildAction,deployAction}.ts}`, e2e `x-session`, `sites-crud`, export `x-posts-private` | — (∥ all) | the compose integration test (public vs private site, flip back, an X-only public site); deploy refusals before wrangler and before the upload |
Order: 0a → D0 ∥ T1 ∥ U1 → T2 ∥ U2 → T3 → parent: records, ONE editor rebuild + restart, umtool rebuild +
restart (the restart is the operator's: the permission layer refuses the `0.0.0.0` bind) → the migration
@@ -321,6 +322,715 @@ hand; a dirent `isFile()` filter over a video dir hides it."** The `.relocating.
- The `en` track → 0 cues bug (index prefers `en` over `en-orig`; some `en` VTTs parse to 0 cues).
- A channel export/import **bundle** built on `mediaTier.ts`'s classifier — the slice after this release.
+## Slice XP — the ruling (2026-10-01)
+
+- **Every X post is hidden from the public, for now; the data is kept, and stays readable by the MCP
+ and umtool for the operator's own questions and tasks.** Fetching is not changed by this slice.
+- **A setting, `social.x.visibility`: `"public"` (default) | `"private"`**, beside
+ `social.x.cookieSource`, chosen on `/settings` in the X account session section as "Where X posts
+ appear", with one sentence saying what private means and that sites already published change on
+ their next build and deploy. `"private"`: every X channel's posts (`sourceKind: "social"`,
+ `platform: "twitter"`) are left out of every PUBLIC site build and built only into PRIVATE sites.
+ Nothing on disk changes; flipping back is a rebuild.
+- **A site audience, `site.json` `audience`: `"public"` (default, absent) | `"private"`**, on the
+ site's form with a sentence. A private site is **never deployed** — every deploy path refuses it
+ with a sentence naming the audience, before any upload, where the release 13 W3 wrong-site guard
+ runs; a build-only still works — and **never listed**: no `hubUrl`, in no homepage or hub listing.
+ Its `corpus.json` says `"audience": "private"`.
+- **One predicate, `postsVisibleTo(site, channelConfig, settings)`**, pure and tested in
+ `common/lib`, called from the index build and from compose. The Search in row's Posts toggle keeps
+ working from what the build shipped (a public site whose only posts were X posts has no posts
+ corpus and no Posts toggle) — verified, not special-cased. The MCP needs no change; umtool's report
+ pipeline is checked for where it reads posts.
+- Not in scope: stopping fetches; a per-platform toggle for Bluesky; deleting anything; editing
+ `transcripts/**` (the rollout — a private site holding every channel, the setting flipped, the
+ public sites rebuilt — is the parent's, through the editor's own writers).
+
## Record
+### Slice U1, as shipped — umtool's render scratch goes to a media root (2026-10-01)
+
+Branch `r17/umtool-media-root` off `main` `7f4901f1`, `main` `90bd8384` (the deck/posts-room merge)
+merged in mid-slice, worktree `~/Projects/homepage-social-visible` (`pnpm wt list` block #11: editor
+4101, test 4111, export 4110), one Opus implementer. Scratch files `U1-*` in the job's `tmp`. The
+ruling is the plan's: render scratch (`out/`) goes to a media root by default; deliverables move per
+project by a switch (slice U2); manifests, `revisions/`, the caches and the cue cache stay put.
+
+**What it does.**
+- **Two roots, one new knob.** `umtool/lib/paths.mjs`: `MEDIA_ROOT = UMTOOL_MEDIA_DIR || REPORTS_ROOT`,
+ `MEDIA_TIERED` (they differ), `mediaMirror(abs, roots?)` (a path under `REPORTS_ROOT` → the same
+ relative path under `MEDIA_ROOT`, null outside; pure). `MEDIA_ROOT` joins `READ_ROOTS`, never
+ `WRITE_ROOTS`. Unset, nothing changes: `out/` is a directory in the project, and `READ_ROOTS` dedupes
+ it away. Every path op carries `turbopackIgnore`.
+- **The cache leaves `SONG_DATA`.** `CACHE_DIR = UMTOOL_CACHE_DIR || $XDG_CACHE_HOME/archilyzer/umtool`
+ (an empty `XDG_CACHE_HOME` is unset, as `common/lib/paths.ts` reads it; default `~/.cache`).
+ `INDEX_DIR`, `MIX_CACHE`, the posters, loudness and clip audio follow it. `OLD_CACHE_DIR`
+ (`<SONG_DATA>/.cache/umtool`) is named only for the doctor. Nothing is migrated: the index is
+ rebuilt by `umtool index` and everything else is remade on demand. The cue cache
+ (`REPORT_CACHE_DIR`, `report-to-video/cues.mjs`) is untouched.
+- **`umtool/lib/report/storage.mjs`** (new; modelled on `common/controller/relocateDir.ts`, not
+ importing it):
+ - `ensureOutDir(projectDir, roots?)`: a real `out/` → kept; a link to a directory → kept; a
+ **dangling link → refused** ("… is a link to …, which is not there — is the media drive mounted?
+ Nothing was written, and nothing was created in its place."); absent and tiered →
+ `mkdir -p <mirror>/out` and an absolute `symlink`; absent and not tiered → `mkdir` as before. The
+ media root itself is **stat'd and never created** (`mediaRootProblem`: missing, not a directory, or
+ inside/around `REPORTS_ROOT`). A project outside `REPORTS_ROOT` is never tiered. EEXIST from a
+ concurrent first writer is accepted when the winner resolves.
+ - `ensureWriteDir(dir)`: a directory a pipeline step writes into; when it is a project's `out` or up
+ to four levels under one, that `out` goes through `ensureOutDir` first, then `mkdir -p`.
+ - `moveDirToMedia(projectDir, name, opts)` / `moveDirToLocal(...)`: by NAME (`out` now; `clips`,
+ `share-*` for U2). Copy (`rsync -a --partial`), mirror toward the copy only (`-a --delete
+ --info=del`; refused when source and copy contain one another), verify (`--dry-run
+ --itemize-changes --delete` empty, one more mirror pass on a difference, a second refuses; equal
+ counts/bytes), then park (`<name>.moved-<ts>`), link, delete the parked copy. Space check on the
+ destination's volume (bytes + 1 GB). Every state is dispatched on the disk, so a cut run is
+ finished by running it again: a link to the mirror → `already` (a leftover parked copy removed);
+ absent with one parked copy → link and delete it; the reverse uses `<name>.incoming`, and after
+ the rename deletes the media copy and every directory above it the move left empty, never the
+ root. `dryRun` measures and changes nothing.
+ - `outDirState`, `pathState`, `measureTree` for readers and the CLI.
+- **Call sites.** `build-video.mjs` (`outRoot`, before any fetch), `check-availability.mjs` (its one
+ write), `render-cards.mjs` (CLI `--out`), `compose-chrome.mjs` (its `out/<variant>` base),
+ `lib/report/onscreen.mjs` (`deckStill`'s scratch) all make `out/` through `ensureWriteDir`.
+ `lib/report/export.mjs` reads only: it now says "out/ is a link to …, which is not there — is the
+ media drive mounted?" instead of "no build" when the link dangles. tmp-then-rename sites (`cut.mjs`,
+ the clip route) are untouched.
+- **The walk.** `kinds.mjs` `SKIP_DIRS` adds `clips` (`out` was already there) and `SKIP_PREFIXES =
+ ["share-"]`, read through `skipsDir(name)` by `walk.mjs`, so the project walk never stats a link
+ into a drive that is not there. The mix picker (`lib/media.ts`) follows a project's `out` link when
+ it points INTO the media root (so a tiered deliverable stays in the picker under its project) and
+ does not walk `MEDIA_ROOT` as a root of its own (it would list every tiered file twice).
+- **CLI.** `umtool doctor` adds `roots` (JSON) / a "roots" block: reports, media (tiered or "=
+ reports"), cache (and whether an index exists), and the old cache with its size while it is there;
+ it exits 1 when the media root is set and missing (the tools' `ok` keeps its meaning). `umtool
+ storage [<project>]` lists every project's `out` (dir, link, DANGLING, none); `umtool storage
+ move-out|move-back <project>|--all [--dry-run] [--json]` runs the movers, one line per project and a
+ total; move-out without `UMTOOL_MEDIA_DIR` refuses once.
+- **e2e env.** The app server and the specs' CLIs get `UMTOOL_CACHE_DIR=<fixture>/cache`
+ (`playwright.config.ts`, `projects.spec.ts`, `report-longform.spec.ts`; the index-deletion spec
+ now removes `cache/index`), so no run writes `~/.cache`. `make-fixture.mjs` adds
+ `storage-fixture` (a cached window, buildable offline), `storage-fresh-fixture` (no `out/`) and the
+ media root `umtool/.e2e-song-media/`, a sibling of the fixture (inside it would be inside
+ `REPORTS_ROOT`, which is refused), reset every run; `.gitignore` and umtool's trace excludes name
+ it. `storage.spec.ts` (new, 5): move-out (dry run first; the index's state and facts unchanged
+ through the link; again → already); **a build the app runs writes through the link and leaves it a
+ link** (the app has no `UMTOOL_MEDIA_DIR` at all); the root renamed away → `storage` says
+ dangling, `check-availability` refuses with the drive sentence on the moved project and with "is
+ not there" on the fresh one, no `out` made, the root not recreated, `doctor` exits 1; the first
+ writer of the fresh project makes the link; move-back → a real `out/`, the project's mirror gone,
+ the root and the other project's mirror kept.
+
+**Commits**
+
+| Commit | What |
+|---|---|
+| `65a3d146` | `umtool:` `MEDIA_ROOT`, `MEDIA_TIERED`, `mediaMirror`; `MEDIA_ROOT` in `READ_ROOTS`; `CACHE_DIR` from `UMTOOL_CACHE_DIR` / `XDG_CACHE_HOME`; `OLD_CACHE_DIR` |
+| `47d6d1b4` | `umtool:` `lib/report/storage.mjs`; the five writers through `ensureWriteDir`; export's dangling sentence; `clips` + `share-*` skips; the picker follows `out` links into the media root; `doctor` roots; `umtool storage` |
+| `4c2cd0c7` | merge of `main` `90bd8384` (the deck/posts-room branch: `build-video.mjs`, `make-fixture.mjs` and more) — one conflict, `export.mjs`'s imports, both kept |
+| `cb08e57a` | `umtool:` `storage.test.mjs` (20); the e2e cache in the fixture; the storage fixtures, media root and `storage.spec.ts`; `umtool storage <project>` status |
+| `efef56b3` | `umtool:` `docs/folders.md` (`MEDIA_ROOT`, `CACHE_DIR`), `docs/cli.md`; two `[Unreleased]` bullets in `editor/CHANGELOG.md` |
+| this commit | `plans:` this section |
+
+#### Gates (logs `$T/U1-*`)
+
+- **tsc** (all workspaces) clean at `47d6d1b4`, at the merge `4c2cd0c7` and at `cb08e57a`.
+- **common:** 2,484/2,484 (300 s, under load). **Editor unit:** 109/109. Neither touched; run on the
+ merged tree.
+- **test:scripts:** 390 tests (the merged `main`'s 370 + `storage.test.mjs`'s 20): 386 passed, 1
+ skipped, 3 failed, then 385/2/3 on a rerun — the three are `queue-lock.test.mjs` timing cases, a
+ different three each time, at a load average of 27–47 (other implementers' suites and builds);
+ `node --test scripts/queue-lock.test.mjs` alone: **11/11**. `storage.test.mjs` **20/20**.
+ `next-build-trace.test.mjs` is in it: **10/10** after each capped build below (its second skip on
+ the rerun is the staleness rule: the baseline run's `git checkout` of `main`'s umtool, below, gave
+ the modules new mtimes after the build).
+- **The capped umtool build with the corpus linked** (`ln -sT <primary>/transcripts transcripts`,
+ 76 channels visible through it; `systemd-run --scope -p MemoryMax=5G -p MemorySwapMax=0`, `timeout
+ -s KILL 240`, the link removed after), at `cb08e57a`: **exit 0, 53 s, 0.83 GB peak**; and once more
+ with `UMTOOL_MEDIA_DIR` set to a scratch directory: **exit 0, 53 s, 0.83 GB**. The two builds'
+ `.nft.json` entries (39,658 each, every route) are **identical** (`diff` empty); none names
+ `transcripts`, the scratch media root or `.e2e-song`. (The worktree carries an old `transcripts/`
+ directory — an `index.mdb` — which `ln -sT` refuses to replace: the script sets it aside for the
+ build and puts it back. A first attempt that did not was stopped before it counted.) The capped
+ editor build was not run: no editor code changed (only `editor/CHANGELOG.md`).
+- **Numbers tool:** none.
+- **umtool e2e** (`SONG_DIR=~/reports/quartering-uh-song/data pnpm --filter umtool run e2e …` from
+ the worktree root; the fixture found song data, `cand2`, `wav48` and `media`, no `asr`, no face
+ detector, so `find.spec`'s 14 skip):
+
+ | Run | At | Specs | Result |
+ |---|---|---|---|
+ | 1 | `efef56b3` | `storage`, `projects`, `report-longform`, `dashboard` | 37 passed, 6 failed, 10.3 min — `storage.spec` **5/5**; the six (`dashboard` ×3, `projects` ×3) are `page.goto: net::ERR_ABORTED` and 30 s timeouts at a load average of 47, and all six pass in run 2 |
+ | 2 | `efef56b3` | the full suite (21 files) | **214 passed**, 17 failed, 14 skipped, 13.8 min of tests (48 min with 34 min in the queue) — `faces` ×4 (`/api/face/detect` 503: no detector here), `triage` ×9 (no `asr` here), `browse:241`, `mix:166`, `mix:201`, `usage:112` |
+ | 3 | `main` `90bd8384`'s umtool, checked out into the worktree and restored after | `browse`, `faces`, `mix`, `triage`, `usage` | 48 passed, 15 failed, 7.2 min — the same `faces` ×4 and `triage` ×9, plus `browse:15`/`:34` (30 s timeouts) |
+ | 4 | `efef56b3` | the same five | 49 passed, 14 failed, 3.2 min — `faces` ×4 and `triage` ×9 as on `main`; `browse:241`, `mix:166`, `mix:201` pass; `usage:112` fails again |
+ | 5 | `efef56b3` | `usage` | **7 passed**, 0 failed, 19.6 s |
+
+ So against `main` on this machine: the `faces` and `triage` failures are the machine's (both
+ missing capabilities fail rather than skip — on `main` too); `mix:166`/`:201` and `browse:241`
+ fail only after the whole suite (the corpus window an earlier spec fetched for `vid1` wins the
+ picker's lookup), and pass in isolation on both; `usage:112` ("confirming the drop writes it
+ through") failed twice when it ran right after the failing `triage` specs on this branch, passed
+ once in that position on `main`, and passes alone — the verdict path it drives reads no cache and
+ no `out/`. Left to the reviewer as an order/timing question, not changed.
+
+#### Found and left
+
+- **Open question 2 — the four `*.mp4` near the project roots** (measured in `~/reports`, depth ≤ 2,
+ outside any `out/`): `kirsche-pippa/latest-contact-2026-06-20.mp4` (7.4 MB) is a cited clip fetched
+ through the MCP's `fetch_clip`, with its `.provenance.json` beside it — the sweep report's evidence,
+ a deliverable of a project that has no `out/`; `quartering-uh-song/jer-metalslug-bg.mp4` (51.5 MB)
+ and `quartering-uh-song/pokemon-no-music-recording.mp4` (3.1 MB) are song-project INPUTS (a song
+ spec's `background.path` names such a file relative to a media root, `song/spec.mjs`); `clips/
+ tim-pool-…mp4` (23.6 MB) is a loose cut at the reports root, in no project. None is render scratch:
+ all four are left untouched, and none is under U2's `clips/` or `share-*/`.
+- **What move-out would move today:** `umtool storage move-out --all --dry-run` against `~/reports`
+ (a scratch media root): **10 projects, 10.6 GB** (quartering-diet 5.0 GB, ferret-rescue 1.5 GB,
+ quartering-employee-count 1.2 GB, elfpire-eva 1.1 GB, …) — less than the plan's "≈ 18 of the 20
+ GB": the rest of `~/reports` is song data and loose files, not project `out/`s.
+- **Rollout, once:** set `UMTOOL_MEDIA_DIR` (the live umtool's environment) to a directory that
+ exists on the media drive, outside `~/reports`; restart umtool; run `umtool index` (the index is
+ rebuilt under `~/.cache/archilyzer/umtool`; until then everything works, slower); `umtool doctor`
+ shows the roots and the old cache (8.7 MB here), which can then be deleted; `umtool storage
+ move-out --all` (when nothing is building) moves the existing `out/`s.
+- **A dangling `out` reads as "no build" to the summary readers** (`lib/projects/report.mjs`'s
+ `stat0(out)`, `readAvailability`): only `export` and the writers say "is the media drive mounted?".
+ `umtool storage` and `umtool doctor` name it. `umtool check` learning it is U2's (plan: "`umtool
+ check` learns the two values").
+- **For U2:** `lib/report/deliver.mjs` `listBatches` filters `isDirectory()` on the project's dirents,
+ so a `share-*` that is a link would vanish from it; `sharedIdsIn`'s walk likewise does not follow a
+ link. The movers take any one-segment name and return `{ state, src, dest|from, bytes, files }`.
+- **A CLI move cannot see the app's jobs** (they live in its memory): the verify refuses when the tree
+ keeps changing, but a write in the instant between the verify and the park would be deleted with
+ the parked copy. The CLI says "run when nothing is building"; U2's bench button runs in the app and
+ can check.
+- **The worktree's stray `transcripts/`** (an `index.mdb` from 2026-09-28) is the trap the rules
+ describe; left in place.
+
+#### Deviations from the plan
+
+- `ensureOutDir` is not called in `driver.mjs`: its step builders are synchronous, are unit-tested with
+ a fake project directory, and only build argv; the call is in the scripts those steps run
+ (`build-video.mjs`, `check-availability.mjs`) through `ensureWriteDir`, which also covers a
+ hand-run script and the three other writers the plan did not list (`render-cards.mjs`,
+ `compose-chrome.mjs`, `onscreen.mjs`).
+- `export.mjs` writes nothing under `out/`, so it does not create it; it reports a dangling link instead.
+- `umtool storage move-back` and the plain `umtool storage [<project>]` listing were added beside
+ `move-out`: the e2e needs the way back, and an operator needs to see which projects moved.
+- `lib/media.ts` (not in the plan) follows `out` links into the media root and skips the root as its
+ own: without it every moved deliverable fell out of the mix picker.
+
+`[Unreleased]` (`editor/CHANGELOG.md`): "umtool can keep each report's render folder on a media
+drive." and "umtool's cache moves to `~/.cache/archilyzer/umtool`".
+
+#### Review (SHIP AFTER FIXES) and the fixes
+
+| Finding | Fix |
+|---|---|
+| F1 — the e2e app and the spec CLIs spread the shell's environment, so a shell exporting `UMTOOL_MEDIA_DIR` would put every fixture build's `out/` on the real media drive | `683e0a0f`: `UMTOOL_MEDIA_DIR=` (empty = unset under `\|\|`) in the webServer command; `UMTOOL_MEDIA_DIR: ""` in the `projects`, `report-longform` and `dashboard` CLI envs (`dashboard`'s doctor also gets the fixture cache); `storage.spec` keeps its own |
+| L1 — `doctor --json`'s `ok` was the tools' verdict while the exit status also counted the roots | `683e0a0f`: `ok = tools && roots`, `toolsOk` = the tools alone, `roots.ok` kept |
+| L2 — a CLI move cannot see the app's jobs | `683e0a0f`: `move-out`/`move-back` (one project or `--all`) skip, as `busy` with the pids, any project a running pipeline script (`build-video`, `check-availability`, `render-cards`, `compose-chrome`, `verify-build`, `fetch-via-editor`, `resolve-windows`, `cut-from-cache`, `share-batch`) names on its command line (`/proc/*/cmdline`; none elsewhere). It does not see the app's in-process deck previews; U2's in-app button can ask the app's jobs |
+| L3 — `dropMediaCopy` deleted any target inside "the media root", which untiered is the reports root | `a6926e17`: deletes only the project's own mirror, only when tiered and only when that is what came home; anything else is left and returned as `mediaCopyLeft` (the CLI prints "left in place … remove it by hand once checked"). A resumed move-back (the link already gone) reports the mirror, never deletes it |
+| L4 — a move-back cut after its rename orphaned the media copy silently | `a6926e17`: "already" reports the project's mirror as `mediaCopyLeft` while it exists |
+| L5 — a cut move's leftovers were not a guard | `a6926e17`: while `out.moved-*` or `out.incoming` exists, `ensureOutDir` (absent `out`) and both movers' directory branches refuse, naming the leftover and the move that finishes it; move-out refuses a lone `.incoming`, move-back a parked copy. `683e0a0f`: `folders.md` — the media root is a directory inside the drive, never the mountpoint; the leftovers rule |
+| N1 — the `paths.mjs` comment named the wrong reason for `READ_ROOTS` | `683e0a0f`: it names `/api/mix/{media,track}` and the lexical write check |
+| N2 — the walk did not skip `*.moved-*`/`*.incoming` | `683e0a0f`: `skipsDir` does |
+| N3 — `isMediaLink` realpathed every link it met | `683e0a0f`, `45cf9946`: only an entry named `out` (U2 adds its names to `MEDIA_LINKS`) |
+| N4 — the changelog bullet | `683e0a0f`: "in umtool's environment (restart umtool after setting it), to a directory inside that drive" |
+| N5 — an empty mirror for a project that does not exist | `a6926e17`: `ensureOutDir` checks the project directory first |
+
+`683e0a0f` left `report-to-video`'s tsc red (the `isMediaLink` parameter type); `45cf9946` restored it.
+
+**Re-review (SHIP AFTER FIXES, no further round):**
+- R1 — a real `out/` beside an `out.moved-*`/`out.incoming` sent each move to the other, which refused again → `4862ec4c`: both movers say both exist, that the leftover holds the moved data, to keep one and remove the other by hand, then run the move; `folders.md` says the same; the leftovers unit case asserts it for both movers.
+- R2 — `ensureOutDir`'s project check used `lstat`, refusing a project directory that is itself a link → `4862ec4c`: `stat`; a unit case links a project in.
+- N6 — the walk's leftover skip matched any `*.incoming`/`*.moved-*` folder → `4862ec4c`: only `out`, `clips` or `share-*` followed by one.
+- Gates: umtool and `report-to-video` tsc clean, all workspaces clean; `storage.test.mjs` 24/24; `test:scripts` 394: 392 passed, 2 skipped (LIVE, and `next-build-trace`'s staleness skip — `storage.mjs` changed after the last build; its path ops gained one `stat`, with `turbopackIgnore`), 0 failed.
+
+**Gates after the fixes:** tsc clean at `45cf9946`. `storage.test.mjs` 24/24 (+4: the untiered
+hand-made link, a tiered foreign link, the leftovers guard both ways, the ghost project; the
+resumed-move-back case now expects the mirror reported and kept). `test:scripts` **394: 393 passed, 1
+skipped (LIVE), 0 failed**, `next-build-trace` passing against a fresh build. Capped umtool build with
+the corpus linked (76 channels; the fixes touch `storage.mjs`'s path ops): exit 0, 39 s, 0.83 GB.
+umtool e2e at `45cf9946`: `storage`, `projects`, `report-longform`, `dashboard` — **42 passed**, 1 failed, 2.3 min (after 45 min in the queue; `storage.spec` 5/5) — the one is `dashboard:14`, the run's first test, a 30 s timeout on the cold first page; `dashboard.spec.ts` alone right after: **7 passed**, 0 failed, 42 s.
+
+**Still left (follow-ups):** the project summary (`lib/projects/report.mjs`) stats `out/<slug>.mp4`
+and reads `out/availability.json` through the link, so a **stalled** media drive blocks those reads,
+libuv's threadpool and the project list, and `availability.json` — small hot text — now lives on the
+media tier; a dangling link still reads as "no build" there (`umtool check` learning it is U2's). The
+`usage.spec:112` order question (after the `triage` specs, at load) is for a quiet-machine run of
+`triage.spec.ts usage.spec.ts` at integration. If a song project ever grows an `out/` and is moved, a
+mix render into it lands on the media root through the link (the write check is lexical; that matches
+the semantics).
+
+### Slice XP, as shipped — X posts are private (2026-10-01)
+
+Branch `r17/x-posts-private` off `main` `90bd8384`, worktree `~/Projects/r13-lows-export` (editor 5501,
+test 5511, export 5510), one Opus implementer, beside the media-tier slices. Scratch files `XP-*` in the
+job's `tmp`. The ruling is above ("Slice XP — the ruling").
+
+**The listing side is release 14 slice HS's.** HS shipped `site.json` `listed` (absent = listed), the
+one predicate `isListedSite`, and every listing that reads it: the homepage summary,
+`channel-sites.json`, the pooled stats, the hub's `hub-sites.json` (and so its `corpus.json` and
+`llms.txt`), every footer, and the form's **List on the Archilyzer homepage and hub** checkbox. This
+slice adds no listing plumbing of its own: `isListedSite` gains one clause — a private site is never
+listed, whatever `listed` says — and the checkbox stays as it is. What is new is the deploy refusal,
+the private site's empty `hubUrl` and its `corpus.json` word, and "private content is built only into
+private sites".
+
+**What it does.**
+- **`social.x.visibility`: `"public"` (default, absent) | `"private"`**, beside `social.x.cookieSource`
+ (`XSocialSettings` and `sanitizeSocial` in `common/social/xCookieSource.ts`, the social block's home;
+ its doc in `settingsSchema.ts`; SETTINGS.md regenerated). On `/settings`, at the foot of the X account
+ session section, **Where X posts appear** (`XPostsVisibilityControl.tsx`): "Public — every site that
+ has the channel" | "Private — private sites only", with: "Private leaves every X channel and its posts
+ out of every public site and builds them only into sites whose audience is private, which are never
+ deployed; nothing is deleted and fetching goes on. Sites already published change on their next build
+ and deploy." Written by `setXPostsVisibilityAction` through `saveSettings`; "public" is written as no
+ key.
+- **`site.json` `audience`: `"public"` (default, absent) | `"private"`** (`siteSchema.ts`, only
+ `"private"` written, `isPrivateSite` the one predicate; SITE.md regenerated). The site form has an
+ **Audience** select with a sentence; `saveSiteAction` keeps only `"private"`.
+- **The rule, `common/lib/postsVisibility.ts`** (pure, tested): `postsVisibleTo(site, config, settings)`
+ — an X channel (`sourceKind: "social"`, `platform: "twitter"`) goes only to a private site while the
+ setting is private; every other channel is untouched — and `publishedMemberSlugs`, a site's members
+ narrowed by it. **A public site leaves the X channel out whole**: posts are all a social channel holds,
+ so without them it would be an empty checkbox and a name in `corpus.json`. Narrowed by the same call in
+ `buildIndex`'s **per-site loop** (the summaries, subs, posts and digests manifests; the site
+ fingerprint gains `withheld`, only when non-empty, so flipping the setting or the audience rebuilds
+ the site and nothing else moves) and in `compose-site` (every shared tree it copies, so a tree a
+ public site shipped before is pruned). The shared posts tree and the LMDB posts sub-DB stay
+ corpus-wide. `channel-sites.json` maps a channel only to the sites whose build carries it
+ (`channelSitesOf` takes the narrowing) and the export's `/offline` page lists the same members.
+- **A private build says so and belongs under no hub**: `corpus.json`'s `site.audience` is `"private"`;
+ `resolveHubUrl` gives a private site no `hubUrl` (absent from its `site.json` and `corpus.json`).
+- **Never deployed.** `lib/builtExport.ts`: `siteDeployProblem(site)` — `Site "x" is private (audience:
+ private): it is built for reading on this machine and is never deployed. Build it without deploying,
+ or set its audience to public on its Settings tab` — `builtAudienceProblem(outDir)` (a bundle whose
+ `corpus.json` says private, so a site switched back to public cannot ship its private build) and
+ `deployAudienceProblem`, both. Asked first, before any upload, where release 13 W3's
+ `builtBundleProblem` is asked: `runDeployIntoLog` (the last word before wrangler),
+ `runDockerDeployAllPhase` (skipped with the sentence, not failed, so Build & deploy all is not red
+ while a private site exists — the site is still built), `deploySite` (`archilyzer deploy site`, before
+ the R2 upload), the editor's `deployExportAction` and `buildAndDeployAction` (before a job exists;
+ Build & deploy asks again before its upload), the host Build & deploy all fallback
+ (`basicBuildAndDeployAll`, skipped before the upload), and `docker/publish-site.sh` (before building,
+ from `site.json`, and over the built `corpus.json` before publishing). The ops routes `build-deploy`
+ and `deploy-site` answer the actions' sentence (400). **Build** still builds a private site.
+- **The Search in row's Posts toggle**: no code changed. `hasPostsCorpus` is "the site posts manifest
+ lists a channel", so a public site whose only posts were X posts ships `channels: []` and no Posts
+ box — pinned by the integration test (no `postScheme` either) and the export spec.
+
+**Where the rule lives — one deviation.** The prompt named the social-channel branch of `scanSource`
+(`buildIndex.ts:333-341`). That branch is corpus-wide: it feeds the shared posts tree that every site,
+and the MCP over a private build, read, so it must keep building X posts. The rule sits in the per-site
+loop (`buildIndex.ts` ~1943 and the fingerprint), a different hunk from T1's `scanSource` guard.
+
+**Found on the way, fixed here.**
+- **`saveSettings` merges one level deep, so a patch `{ social: { x } }` replaced the whole X block**:
+ choosing a login source would have erased the visibility, and the reverse. Both X actions now write
+ over the block as it is (`socialXPatch`); the e2e case pins both directions.
+- **compose-site trusted its per-site cache after ANOTHER site's compose.** `public/` is one directory
+ every site composes into in turn (the basic build); the cache is per site, and a stage whose source
+ had not changed was skipped. So composing site B, then site A again with no new data, shipped B's
+ summaries — its whole channel list — as A's: a public site composed after a private one holding every
+ channel would have listed the private site's channels. The site's own stages (summaries, stats,
+ duplicates) are now trusted only when `public/site.json` names the site; `site.json` is cleared at
+ the start of a compose and written at its end, so a compose cut short leaves nothing to trust. The
+ per-channel trees keep their signatures (copies of the shared trees, the same bytes whichever site
+ copied them). Docker builds have a per-site `public/` and were not affected.
+- **compose never ships a posts tree the site's posts manifest does not list** (the index build's
+ word), so a channel config compose fails to read — read as visible — does not ship X posts.
+
+**The MCP and umtool.**
+- **The MCP needs no change**: it reads a composed export (`mcp/src/sources.ts`, `--local <dir>` |
+ `TRANSCRIPT_LOCAL_DIR`). A private site is composed into a directory of its own, without touching
+ `export/public`, from the checkout root:
+ ```sh
+ pnpm archilyzer index # or any editor build: the index build writes the per-site manifests
+ BUILD_ARCHIVES=0 EXPORT_PUBLIC_DIR="$HOME/archives/<private-id>" \
+ EXPORT_INDEX_DIR="$PWD/export/.export-index" pnpm archilyzer compose site <private-id>
+ claude mcp remove archilyzer -s local # the name must be free; use the scope it was added in
+ claude mcp add archilyzer \
+ --env ARCHILYZER_EDITOR_URL=http://localhost:3001 \
+ --env WORKER_TOKEN=… \
+ -- pnpm --silent -C "$PWD" archilyzer mcp --local "$HOME/archives/<private-id>"
+ ```
+ The two `--env` lines are what `fetch_clip` needs (the editor's own `WORKER_TOKEN`, from
+ `editor/.env`); the name stays `archilyzer` for `/ask` and `/sweep` (AGENTS.md).
+ (`EXPORT_PUBLIC_DIR` must be absolute — the command runs in `common/`; the compose cache lands beside
+ it, in `$HOME/archives/.compose-cache/`.) The site's editor **Build** works too: it composes into
+ `export/public` and builds into `export/out`. **The current registration, `--local
+ <checkout>/export/public`, reads whatever site was composed there last**: after a public site's build
+ it has no X posts, after the private site's it has them.
+- **umtool is unaffected**: the report pipeline's posts (the deck's posts room) are carried whole in the
+ report manifest — `posts[]` with `platform`, `date`, `text`, `url` (`validatePosts`,
+ `umtool/report-to-video/deck.mjs`), "added by editing the manifest" — and its cues come from the local
+ corpus or the archive `provenance.siteOrigin` names (videos only). It reads no public build's posts. A
+ sweep that looks posts up for a manifest goes through the MCP, pointed at the private build as above.
+
+**Commits**
+
+| Commit | What |
+|---|---|
+| `3ea76853` | `common:` `postsVisibility.ts` + test; `social.x.visibility`; `site.json` `audience` (`isListedSite`, `resolveHubUrl`); the per-site loop and compose narrowed; `corpus.json` `audience`; the deploy refusals (`builtExport`, `publish/build`) + tests; `channel-sites.json`; `/offline`; `publish-site.sh`; SETTINGS.md, SITE.md; the compose integration test |
+| `0195d52a` | `editor:` Where X posts appear; the X block written whole; the Audience select; the deploy actions refuse a private site |
+| `75ccbef3` | `editor(e2e), export(e2e):` `x-session` and `sites-crud` cases; `export/e2e/x-posts-private.spec.ts` |
+| `30c14aa0` | `common:` compose never ships a posts tree the index withheld; the cache trusted only over the site's own last compose |
+| `63002de3` | `common:` the per-channel trees keep their signatures across sites |
+| `4ed7d410` | `plans:` this section, the ruling, the slices row; FACTS; the editor changelog |
+| `6466a68e` | `common:` the hub carries no site's data; `builtHubProblem` refuses one that does (review HIGH 1) |
+| `385e4eb1` | `common:` a compose over a stale index lists no withheld channel; the comments (LOW 3, NIT 7) |
+| `8e448fde` | `editor:` the changelog (LOW 5) |
+| `da5c2912` | `plans:` the review, its record, the rollout steps and the gates after it |
+| `837d630a` | merge of `main` `bb877f93` (slice U1: umtool, `.gitignore`, and the two shared records — both sides kept, U1's section before this one). Re-gated on the merged tree: tsc clean; common **2,501/2,501**; export unit 98/98; homepage unit 23/23; editor unit 109/109; test:scripts 392 passed, 0 failed, 2 skipped (394). The merge touched no file the editor, export or hub e2e lists cover (umtool only), so they were not re-run |
+| this commit | `plans:` the merge in this table |
+
+#### Gates (logs `$T/XP-*.log`)
+
+- **tsc** (all workspaces) clean before every commit; last at `63002de3`'s tree (`XP-tsc5.log`).
+- **common:** **2,499/2,499** at `63002de3`, 161 s (`main`'s count + the new `postsVisibility.test.ts`
+ 7, `compose-site.postsVisibility.test.ts` 4, `build.test.ts` +3, `poolSummary.test.ts` +1); 2,498 at
+ `75ccbef3`. `archilyzer docs files --check` and `settings example --check` exit 0. **Editor unit:**
+ 109/109. **Export unit:** 98/98. **Homepage unit:** 23/23. **mcp:** 271/271 (no change there).
+- **test:scripts:** 367 passed, 1 failed, 2 skipped of 370 (`XP-scripts2.log`; the first run had 2
+ failed). The failures are `scripts/queue-lock.test.mjs`'s "prints a banner naming the holder while
+ waiting" (and once "serves waiters in arrival order"): timing cases (200 ms and 150 ms staggers) run
+ at a load average of 23–30 while other suites held the e2e queue; alone, the banner case still fails
+ under that load (10/11). This slice does not touch `scripts/` (`git diff 90bd8384 -- scripts/` is
+ empty).
+- **Builds** at `75ccbef3` (the later commits touch only `compose-site`, a bin no Next app bundles): the
+ capped editor build with the corpus linked (`ln -sT`, `systemd-run --scope -p MemoryMax=6G`, the link
+ removed after) exit 0, 172 s; `pnpm --filter export exec next build` exit 0, 83 s (the `export/public`
+ links made, 0 dangling); `pnpm --filter homepage run build:nodata` exit 0, 47 s.
+- **Numbers tool:** none. **Privacy gate:** `git diff main --name-only | xargs grep -lc …` names one
+ file, `plans/FACTS.md`, whose 3 matches are all on `main` already; 0 in this slice's added lines.
+
+ | Run | At | Specs | Result |
+ |---|---|---|---|
+ | 1 (editor) | `75ccbef3` | `sites-crud`, `settings`, `x-session`, `forms-keep-input`, `deploy-page`, `site-publish-preview`, `sites-homepage` | **62 passed**, 0 failed, 4.0 min (after 35 min in the queue) |
+ | 2 (export) | `63002de3` | `x-posts-private` (new, 2), `search-in`, `posts-search` | **20 passed**, 0 failed, 3.0 min (after 7 min in the queue) |
+
+ New cases: `x-session` "where X posts appear persists, beside the login source and without it" (the
+ whole-block write both ways); `sites-crud` "a private site saves its audience, and every deploy
+ refuses it before any job" (the form, the file, `build-deploy` and `deploy-site` answering the
+ sentence with 400, back to public with the key gone); export `x-posts-private` — the posts manifest a
+ public site built with X private ships (no channel: no Posts box, no X post for "kappa") and the one a
+ private site ships (the Posts box, both X posts). **The export suite's data is route-mocked, never
+ built**, so the build rule itself is proved by `common/bin/compose-site.postsVisibility.test.ts`
+ through the real `buildIndex` and compose: a public and a private site over one video, one X and one
+ Bluesky channel; the flip back; a public site whose only posts were X posts (empty manifest, no
+ `postScheme`); a config compose cannot read; a public site composed after the private one.
+
+#### Found and left
+
+- **An X channel's manifest-only transcripts tree** (pageCount 0) would still be copied into a public
+ site when compose fails to read the channel's config: a directory named for the channel, holding no
+ content. The posts tree, the posts manifest, the channel list and `corpus.json` follow the index build
+ and do not carry it.
+- The `/sites` list does not mark a private site; its form and every deploy refusal do.
+- **The posts reconcile ships only the channels the site's posts manifest lists, on every build**: a
+ social channel with 0 posts no longer gets a posts folder. Nothing advertised that folder (no posts
+ manifest entry, no `corpus.json` posts link), so nothing reads the difference.
+- **A private site changes the homepage's and the hub's totals.** A private site is unlisted, and
+ `channelsOnlyOnUnlistedSites` keeps a channel only unlisted sites expose out of every public total. A
+ private site holding every channel turns every pool-only channel (on no public site) into "only on
+ unlisted sites", so the homepage's and the hub's instance-wide totals drop by those channels at the
+ next homepage and hub build. Channels a public site also has are unaffected.
+- **Cloudflare Pages keeps old builds reachable.** A production redeploy replaces what the production
+ URL serves, nothing else: every earlier deployment stays live at its own
+ `<hash>.<project>.pages.dev`, and a preview alias (`<branch>.<project>.pages.dev`) keeps its last
+ build. The review found `tags-exclude.anilyzer.pages.dev/posts/manifest.json` still listing an X
+ channel. So after the rollout, X posts are gone from the production URLs only; whether to delete
+ the old deployments is the operator's call (rollout step 5).
+- `queue-lock.test.mjs`'s two timing cases failed under the machine's load (below); not this slice's file.
+
+#### Decisions the operator could overturn
+
+| What I did | The alternative |
+|---|---|
+| A public site leaves an X channel out WHOLE (posts, posts manifest, channel list, `site.json`, `corpus.json`, `channel-sites.json`) | Keep the channel listed with no posts: an empty checkbox and a name in `corpus.json` |
+| Build & deploy all builds a private site and SKIPS its deploy, with the sentence | Fail its deploy: the run goes red every time while a private site exists |
+| A bundle whose `corpus.json` says private is refused even when the site is public now | Trust the site's current audience only (a stale private build could ship) |
+| `docker/publish-site.sh` refuses a private site (the `site` service is the host's public face) | Let it publish locally behind Caddy |
+| `social.x.visibility` lives in `xCookieSource.ts` with the rest of the social block | A module of its own |
+| The hub's `search-aliases.json` is the global dictionary (it was whichever site composed last) | Ship none: hub-wide search in a reader (`reader-hub.ts`) would have no aliases |
+
+#### Rollout for this slice (the parent's, through the editor's own writers)
+
+1. Rebuild and restart the editor (the setting, the Audience field, the deploy refusals, the compose
+ and hub fixes).
+2. Create the private site (Sites → New, **Audience: Private**, every channel), then set **Where X posts
+ appear: Private** on `/settings`.
+3. Rebuild and deploy every public site that has an X channel (each by name; Build & deploy all skips the
+ private site's deploy and says why).
+4. **Rebuild and deploy the hub** — the live hub serves the last-built site's data, X posts included,
+ until it is rebuilt from this branch (the review's HIGH 1). Then the homepage, for the totals.
+5. **Old deployments** (the operator decides): every earlier production deployment and every preview
+ alias of a site that had X posts still serves them. To remove them: the Cloudflare dashboard →
+ Workers & Pages → the project → Deployments → a deployment's ⋯ menu → Delete deployment (a preview
+ alias's branch deployments are listed there too); or `pnpm dlx wrangler pages deployment list
+ --project-name <project>` then `pnpm dlx wrangler pages deployment delete <deployment-id>
+ --project-name <project>` (check `--help` for the force flag an aliased deployment needs). The
+ current production deployment cannot be deleted, and needs none.
+6. Build the private site (its **Build**, or the compose-to-a-directory commands above) and point the
+ MCP at it.
+
+#### Review
+
+**Verdict: SHIP AFTER FIXES** (`XP-review.md` in the job's scratch): two Highs, four Lows, three nits.
+
+| Finding | Where |
+|---|---|
+| HIGH 1: the hub bundle carried the last-composed site's data trees (the live hub serves jeralyzer's posts manifest, two X channels), through no gate | `6466a68e`: `compose-hub` removes every per-site entry from `public/` first (`SITE_ONLY_PUBLIC_ENTRIES`: summaries, transcripts, subs, posts, digests, stats, archives, `site.json`, `tags.json`, `duplicates.json`, `search-aliases.json`, `chart-templates.json`, `sitemap.xml`; a worktree link by the link only) and writes the global alias dictionary as the hub's; `builtHubProblem` refuses a hub bundle that still carries a data tree, so Deploy hub refuses one. Tests: `compose-hub.test.ts` +1, `builtExport.test.ts` extended. Rollout step 4 |
+| HIGH 2: Pages preview aliases and old deployments keep serving X posts | Record: "Found and left" and rollout step 5 (the operator decides; the dashboard and wrangler paths) |
+| LOW 3: a compose over an index built before the flip listed the X channel's name and count | `385e4eb1`: the served posts manifest and summaries manifest are narrowed to the published members (`site.json`, `corpus.json` follow); a narrowed summaries copy is not trusted by the next compose. Test: "a compose over an index built before the setting flipped lists no X channel anywhere" |
+| LOW 4: the MCP line lost `fetch_clip`'s env, and `add` fails over a registered name | This commit: `claude mcp remove archilyzer -s local` first, the two `--env` lines kept (`WORKER_TOKEN=…`) |
+| LOW 5: the changelog missed the compose-cache fix and the restart notes | `8e448fde`: a bullet for the compose-cache fix, one for the hub, and the restart/redeploy notes |
+| LOW 6: a private site holding every channel moves the homepage's and hub's totals | Record: "Found and left" |
+| NIT 7: the deploy-all comment said the bundle check runs first | `385e4eb1`: the comment says the audience check runs first and what that means for a private wrong-site bundle |
+| NIT 8: the posts reconcile drops a 0-post social channel's folder | Record: "Found and left" |
+| NIT 9: `/sites` does not mark a private site | Already listed as left |
+
+#### Gates after the review
+
+- **tsc** (all workspaces) clean at `385e4eb1`'s tree (`XP-tsc6.log`).
+- **common:** 2,500 passed, 1 failed of 2,501 (`XP-common4.log`; +2 since the first gates: the hub
+ clearing and the stale-index compose). The one failure is `relocateChannelMedia.test.ts`'s "reconcile:
+ an extra and a changed file on the destination are settled" (its diff listing counted `./` as
+ changed — a directory mtime); 3/3 alone, and the file is not this slice's. **Export unit:** 98/98.
+ **Homepage unit:** 23/23. **Editor unit:** 109/109.
+- **No rebuild:** the fixes touch two bins (`compose-hub`, `compose-site`), `builtExport.ts` and
+ comments in `publish/build.ts`; no Next app bundles a changed module beyond `builtExport` (the
+ editor's hub action reads `builtHubProblem`, a pure function, tsc-checked).
+
+ | Run | At | Specs | Result |
+ |---|---|---|---|
+ | 3 (editor) | `8e448fde` | the run-1 list | **62 passed**, 0 failed, 3.0 min (after 45 min in the queue) |
+ | 4 (export) | `8e448fde` | the run-2 list | **20 passed**, 0 failed, 1.1 min (after 50 min in the queue) |
+ | 5 (hub) | `8e448fde` | `e2e:hub`, the whole suite | **39 passed**, 0 failed, 1.6 min (after 12 min in the queue) |
+
+ The hub suite runs `next dev` in hub mode over `export/public` and never composes, so it shows the
+ hub app is unchanged; the clearing itself is pinned by `compose-hub.test.ts`.
+
+### Slice D0, as shipped — the dashboard answers while a snapshot regenerates (2026-10-01)
+
+Branch `r17/dashboard-answers` off `main` `7f4901f1`, worktree `~/Projects/r13-lows-editor` (editor 5401,
+test 5411, export 5410 — `pnpm wt list`'s block #24), one Opus implementer. Scratch files `D0-*` in the
+job's `tmp`. The plan is Step 0b above.
+
+**What was found before building** (the corpus only read; one video dir's text copied to scratch for the
+profile).
+- **The walk did yield — on every file read.** Each video's unit awaits its reads, so the loop turned
+ between them; what the walk does ON the loop is parse. A CPU profile of `generateChannelSnapshot` over
+ 300 copies of one omnibased video dir's text (metadata 0.62 MB, cues 0.62 MB, two 2.9 MB VTTs):
+ `readNormalizedTranscript` (the cues parse, `readTranscriptCoverage`) 1,037 ms self, `readWebpageUrl`
+ 791 ms self, everything else in the walk under 70 ms. `readWebpageUrl` is the reconcile pass at the
+ walk's start (`reconcileVideoDirs.ts`): it reads and parses EVERY `metadata.info.json` for its
+ `webpage_url`, one at a time, on every regeneration.
+- **One walk alone does not starve the pages.** Built and served by `next start` (loopback, a scratch
+ corpus of one 2,000-video channel, load average 24), with this slice: the regeneration took 26 s, and
+ `/` answered in 0.28–1.6 s and `/jobs` in 0.18–0.81 s, polled every 2 s throughout. Under `next dev`
+ at a load average of 32–35 the same walk took 118 s and the answers ran 0.3–28 s; with `main`'s five
+ files swapped back, 2.0–12.5 s. On this machine, under the dev server, one regeneration does not
+ separate the two, and an isolated tsx micro-benchmark (800 such dirs, three interleaved pairs, load
+ 25–35) put the chunked walk and `main`'s inside each other's noise (loop delay max 300–860 ms either
+ way).
+- **So the hour-long outage needed more than one walk**, and the live metas show the rest: two walks side
+ by side (the empty queue key), the omnibased one on the platter through `onDrive` (four slots per
+ location, so every page read of that drive queued behind the walk's units), two regenerations of
+ omnibased four seconds apart in the restarted process (`01M3WHY8…`, `01M3WHYC…` — two passes racing
+ between the registry check and the record's registration), and the auto-queue status poll recomputing
+ behind all of it. D0 closes the three it owns: one queue, a per-slug dedup that covers the enqueue in
+ flight, and the status memo. The platter half is T1's (text reads leave `onDrive`).
+- **The ghosts never held a move.** `channelWriters` reads this process's registry only, never a meta,
+ and the registry is in memory: a meta a dead process left `running` was never a writer. `/jobs`
+ listed such a job as `archived` (`listJobs`: a non-terminal meta not in the registry). Harmless to a
+ move, misleading on `/jobs`; the boot pass now closes them.
+- **What a SIGTERM'd process leaves behind today** (`shutdownCancel.ts`, "graceful shutdown only"): the
+ reaper cancels every live job and re-raises the signal at once, without waiting. A queued job keeps
+ `queued` on disk on purpose (the boot pass settles it). A running managed job's terminal meta is
+ written only when its function returns (`streamCommand.ts`, the `.finally`); `generateChannelSnapshot`
+ takes no signal, so a regeneration never returns before the exit, and its meta stays `running`. A
+ SIGKILL leaves the same, and orphans children.
+
+**What it does.**
+- **The walk yields between chunks.** `generateChannelSnapshot` maps its video dirs through
+ `mapInYieldingChunks` — chunks of `SNAPSHOT_YIELD_EVERY` (32: two full waves of the 16-wide limit), one
+ `setImmediate` between chunks. The unit body is untouched: the diff in that file is the helper, the
+ constant and the call's two lines, so T1's merge is trivial.
+- **One regeneration at a time, once per channel.** `snapshotScheduler.ts` gains `REFRESH_REPORT_QUEUE`
+ (`"refresh-report"`: a non-empty key is the registry's existing concurrency-1 serialization, so no new
+ mechanism) and `startRefreshReport(paths, slug)`, the one entry point: a slug with a `refresh-report`
+ queued, running or being enqueued (a `starting` set on the scheduler's global state, held across
+ `runManagedFunction`'s awaits) is answered `info` with `REFRESH_REPORT_ACTIVE`. The debounced pass and
+ **Update all reports** (`refreshAllChannelSnapshotsAction`, `editor/app/channels/actions.ts` — not in
+ the slice's file list and owned by no other slice) both go through it, so they dedup against each
+ other; the action still reports a skip as "already running". The job body is the scheduler's (it bumps
+ `generation`, which the action's copy did not).
+- **The auto-queue status poll shares one fold.** `common/views/autoQueueStatus.ts` gains
+ `singleFlightMemo` (concurrent callers share the computation in flight; a landed value is reused for
+ `AUTO_QUEUE_STATUS_MEMO_MS` = 3 s from when it LANDED; a rejection is not memoized; `clear()` detaches
+ an in-flight computation) and `autoQueueStatusMemo()` on `globalThis`. `buildAutoQueueStatusPayload`
+ is unchanged. The shell (`editor/app/operations/status.ts`) memoizes the snapshot-derived half only —
+ the channel briefs and the four lanes' `computeLeafPending` — and reads the priority view, the state
+ document, settings, the pool and the runners fresh. The e2e reset route drops the memo with the other
+ singletons.
+- **Boot closes the running metas a dead process left.** A meta now records `pid` (`jobMeta.ts`, owned by
+ no slice). `bootQueuedJobs.ts` gains `settleRunningJobMetas`, run from `instrumentation.ts` on every
+ boot (idle and test server too: it re-queues nothing, and it does not wait for the storage pass): a
+ `running` meta from before this boot, not in the registry, whose writer is gone is closed `cancelled`
+ with `cancelReason` "interrupted: the process running it stopped before it finished", `endedAt` = its
+ log's mtime (the last moment it is known to have run), else now. **The status:** `cancelled` +
+ `cancelReason` already is this file's terminal state for "the server went down under it", and every
+ reader (listJobs, `/jobs`, the job page's "Cancelled because", Retry) handles it — so there is no new
+ `interrupted` status. **The dead-process test** (`writerIsGone`): metas carried no pid or host, and
+ "predates this boot" alone is not reliable here, because `archilyzer run` (`bin/run-operation.ts`) runs
+ jobs offline into the same `.jobs/`. Gone = no pid (written before this release), or this process's
+ own pid (a container's editor restarts as the same pid; this process's own metas are already excluded
+ by `bootedAt` and `isLive`), or `kill(pid, 0)` → ESRCH (EPERM counts as alive). A live pid is left
+ alone. The queued pass now applies the same check. `channelWriters.ts` gains a comment saying why a
+ ghost cannot reach it, and a test pins it.
+- **e2e** `dashboard-answers.spec.ts` (two cases) and `/api/test/settle-running-metas` (POST, guarded by
+ `E2E_TEST_ROUTES`), which runs the running pass on demand: the e2e server boots once per suite.
+ (a) Two 600-video channels (one hardlinked ~400 KB `metadata.info.json` per dir, a Rumble archive so
+ each file is parsed twice) regenerated by Update all reports through the ops API while `/` and `/jobs`
+ are polled every 2 s: both reports land, the second job started when the first had ended (read from
+ their records), and every answer is inside the budget. (b) A `running` refresh-report meta with a dead
+ pid is closed as interrupted, one with a live pid (the test runner's) is left `running`, the job page
+ says "interrupted", and the channel's Move media completes.
+
+**Commits**
+
+| Commit | What |
+|---|---|
+| `8ce3b24d` | `common:` the chunked walk; `REFRESH_REPORT_QUEUE`, `startRefreshReport`, the `starting` set; `snapshotScheduler.test.ts` (2); Update all reports through it |
+| `56efcd74` | `common, editor:` `singleFlightMemo` + `autoQueueStatusMemo` + 5 tests; the shell memoizes the snapshot half; the reset route drops it |
+| `36babcad` | `common, editor:` meta `pid`; `writerIsGone`, `processIsAlive`, `settleRunningJobMetas`; the queued pass's writer check; boot wiring; 4 boot tests + 1 `channelWriters` test |
+| `fef9ebc2` | `editor(e2e):` `dashboard-answers.spec.ts`; `/api/test/settle-running-metas` |
+| `fe1676c3` | `editor(e2e):` the spec reworked after run 1 (two 600-video channels, the serial check, the idle-floored budget) |
+| `3c90b8e9` | `plans:` this section; the editor changelog |
+
+#### Gates (logs `$T/D0-*.log`)
+
+- **tsc** (all workspaces) clean at every commit.
+- **common:** **2,496/2,496** — new: `snapshotScheduler.test.ts` 2, `autoQueueStatus.test.ts` +5,
+ `bootQueuedJobs.test.ts` +4, `channelWriters.test.ts` +1. **Editor unit:** 109/109.
+ **test:scripts:** run 1: 301 passed, 1 failed, 2 skipped (304); rerun: 300 passed, 2 failed, 2
+ skipped. Every failure is in `queue-lock.test.mjs` ("prints a banner naming the holder while waiting",
+ "serves waiters in arrival order (FIFO)"), timing cases run at load averages of 16–30; the file run
+ alone passes 11/11, and the slice touches nothing under `scripts/`.
+- **Build:** the capped editor build with the corpus linked (`ln -sT`, `systemd-run --scope -p
+ MemoryMax=6G`, the link removed after): exit 0, 123 s, at `fef9ebc2`.
+- **e2e** (`jobs`, `channels`, `channel-storage`, `dashboard-answers`; `next dev`):
+ - run 1 at `fef9ebc2`: **23 passed, 2 failed, 16.7 min**. `dashboard-answers` (a) timed out at 5 min:
+ 2,000 videos under `next dev` at a load average of ~30 did not finish (the scratch-server measurement
+ above: 118 s for the walk alone, every answer slower). `channel-storage` "relocate a channel's media
+ to another root, and move it back" — the run's first test, on a cold dev server — timed out at 90 s
+ on its first request to the file route.
+ - run 2 at `fe1676c3`: **25 passed, 0 failed, 3.7 min** (34 min with the queue wait). (a) took 23.8 s:
+ nine polls of each page while the two regenerations ran, `/` 0.14–1.9 s and `/jobs` 0.34–1.6 s
+ (budget 5 s); (b) 7.8 s. The relocate case passed — run 1's failure was the cold first test.
+- **Numbers tool:** none.
+
+**Deviations from the plan, one sentence each.**
+- The chunk is 32, not 25, so every wave of the 16-wide limit is full (25 left each chunk's second wave
+ nine wide).
+- The memo covers the shell's snapshot-derived half, not the whole payload: memoizing it all would hold a
+ lane's hold, runner and picks up to 3 s behind a click, and specs read them straight after one.
+- The running-meta pass is in `bootQueuedJobs.ts` beside the queued one, not in `registry.ts`, which is
+ in memory and holds no metas.
+- "Interrupted" is `cancelled` + `cancelReason` (the existing terminal state for a job the server went
+ down under), not a new status.
+- `channelWriters.ts` gets a comment and a test, no code: a ghost cannot reach it.
+- The e2e budget is 5 s or three times the slowest idle answer measured just before, whichever is longer:
+ the dev server on this shared machine at load 30 takes 2–4 s a page with nothing regenerating, and the
+ spec pins starvation, not the machine's speed.
+- The spec regenerates two 600-video channels, not one large one: 2,000 did not finish in five minutes
+ under the dev server here, and two prove the serial queue from the jobs' records.
+
+**Found and left.**
+- The reconcile pass parses every `metadata.info.json` in full, on every regeneration, for one field — the
+ largest single cost of a walk. Reading only the head, or skipping a dir whose name is already canonical,
+ is a follow-up (`reconcileVideoDirs.ts` is not D0's).
+- `generateChannelSnapshot` takes no abort signal, so a Cancel or a graceful shutdown cannot stop a walk;
+ the chunk boundary is now the natural place to check one — a follow-up.
+- A meta whose pid has since been reused by an unrelated live process stays `running` on `/jobs` until
+ that process exits; a start-time check (`/proc/<pid>/stat`) would close it. Not worth it today.
+- `editor/app/jobs/[id]/page.tsx` and `JobRow.tsx` describe the cancel reason as the one "a restart left
+ queued"; it now also covers a dead process's running job. Comments only, left.
+
+**Changelog** (`editor/CHANGELOG.md` `[Unreleased]`): three bullets — the dashboard and `/jobs` answer
+while reports regenerate (one at a time, deduped; Update all reports takes the sum), the operations pages
+share one pending count (up to 3 s old; holds, runners and picks fresh), and jobs a stopped editor left
+"running" are closed at start.
+
+**Rollout note.** No `archilyzer run` may be in flight across the first editor restart after this ships:
+every meta written before it has no `pid`, so the boot pass reads its writer as gone and would close a
+still-running offline job's meta as interrupted (the live job rewrites it at its next meta write, but
+`/jobs` offers Retry meanwhile). The same holds across pid namespaces — an editor in a container judging
+a pid a host-side `archilyzer run` wrote, or the reverse, sees ESRCH. The release's rollout stops the
+editor for the migration anyway.
+
+#### Review (SHIP AFTER FIXES) and the fixes
+
+| Finding | Fix |
+|---|---|
+| H1 — the memo served pending counts folded from settings (lane policy and tree, `channelPriority`) up to 3 s stale against a fresh tree; the payload-reading specs were not run | `5b236f8c`: `singleFlightMemo.get(key, compute)`; the shell keys it on `JSON.stringify([settings.autoQueue, settings.channelPriority])`, time the only other expiry; only the latest-started computation stores (an old key landing late never overwrites); comments say what can be one poll late (a rewritten snapshot, the runner's in-flight set) and what is fresh; +1 test. The status-reading specs ran in the full suite below |
+| L2 — a change during a RUNNING regeneration was dropped until the next change (as on main) | `616a9ce2`: dedup only against a queued or starting regeneration (`isRefreshReportPending`), so a running walk gets one queued successor and no second; fire()'s trailing-edge comment is now true; +1 test |
+| L3 — the per-channel Refresh report walked in the request, outside the queue | `5509ccf4`: `refreshChannelSnapshotAction` (Refresh report, ops `{slug}`, e2e `generateReport`) starts a job through `startRefreshReport` and awaits it — or polls the one already queued for the channel — and returns the walk's error sentence (`onError`) as its `{error}`; the ops route's comment follows |
+| L4 — a queued refresh-report holds the Storage panel's Move (`mediaBusy.ts` counts queued jobs) | Not changed in D0, as ruled: T1's `mediaOnly` (refresh-report does not need media) clears it when T2's courtesy check passes it; to confirm at the T1/T2 merges |
+| L3 follow-on — `generateReport` (85 specs) now queues a job, which `/jobs` may draw before its default filter hides it | `7ba8cd41`: `bulk-actions.spec` "queues no job" leaves `refresh-report` rows out; `dashboard-answers` compiles the ops route (a 400 on `{}`) before measuring — its first compile held `/` 15 s once under load |
+| L5 — nothing pinned the yield; the e2e floor could grow without bound | `fea5b203`: `snapshotYield.test.ts` — a `setImmediate` probe queued during the first chunk runs before the second chunk's first unit (it fails with the yield removed, checked); 2 tests. The e2e budget is `min(max(5 s, 3 × slowest idle), 15 s)`, and the spec header says it pins the serial queue and gross starvation, not the yield |
+| L6 — the first boot after this ships closes a live pre-slice `archilyzer run`'s meta | The rollout note above |
+| N7 — `REFRESH_REPORT_QUEUE` beside the other queue keys | `616a9ce2`: defined in `lib/queueKeys.ts`, re-exported by the scheduler |
+| N8 — `operations/[id]/page.tsx`'s census comment claimed the payload's listing | `5b236f8c`: says it is the same listing only on a memo miss |
+| N9 — `requestCache.ts` said "no exception" | `5b236f8c`: one paragraph naming the memo and its bound |
+| N10 — `channelWriters.test.ts` conflicts with T1 (both append) | Keep both, at the merge |
+| the record and changelog | this commit: this subsection, the rollout note; the changelog's first two bullets say Refresh report waits its turn, a change during a regeneration queues one more, and a rules/focus/priority edit recounts at once |
+
+**Gates after the fixes.** tsc clean at every commit. **common 2,500/2,500** (+4: memo key 1, the
+successor 1, the yield 2). **Editor unit 109/109.** **e2e** — L3 makes every `generateReport` (85 specs) a
+queued job, so the whole editor suite rather than the two lists (it contains both), in three runs on a
+machine that ran out of memory (15 GB used, swap 19/19 GB; a parakeet transcription of the live editor
+beside several suites):
+ - run 3, the full suite (702 tests) at `fea5b203`: stopped at 374 passed, 8 failed, ~50 min, when
+ pages began to crash (`page.goto: Page crashed`). Two failures were D0's and are fixed in `7ba8cd41`
+ (`bulk-actions` "queues no job"; `dashboard-answers` (a): one `/` of 14.9 s at the moment the ops
+ route compiled, every other answer ≤ 4.6 s). The other six passed in run 4.
+ - run 4 at `7ba8cd41` (run 3's failures plus every spec file from `new-channel-onboarding` on, and both
+ lists — 69 files): **295 passed, 154 failed, 35.1 min** (54 with the queue). Every spec in both lists
+ passed — `jobs` 2, `channels` 8, `channel-storage` 13, `dashboard-answers` 2, `focus-banner` 3,
+ `auto-queue` 24, `lane-runner` 5, `channel-priority` 9, `operation-settings` 7, `backfill` 20,
+ `channel-work` 11, `ops-api` 23 — except `view-route`, which ran after a machine-wide OOM kill at
+ 22:41 took the test server (the kernel log names it, among browser tabs, the desktop session and the
+ live editor's transcriber): every test from the 300th on failed in under 2 s against a dead server.
+ Before it, four failures in `site-scope` (2), `sites-crud` and `social-channel` (the known 22–26 s
+ fetch-posts case).
+ - run 5 at `7ba8cd41`, the 25 spec files with a failure in run 4 (`view-route` included): **183 passed, 0
+ failed, 12.2 min** (13 with the queue).
+ So every editor spec has passed with the fixes in, across runs 3–5, and both lists in full.
+
+**test:scripts** was not re-run: the fixes touch nothing under `scripts/`.
+
+#### Re-review (SHIP AFTER FIXES) and the fixes
+
+| Finding | Fix |
+|---|---|
+| R1 — Refresh report and both ops forms waited for the whole serial queue, with no bound and no feedback (the CLI's fetch gives up at 300 s; the button dropped the action's result) | `5287e8c2`: `requestRefreshReport` (start, or find the queued job) and `waitForRefreshReport` (up to `REFRESH_REPORT_WAIT_MS` = 15 s → done, failed, or waiting with the regenerations ahead), `refreshReportWaitNotice`; `e491ec14`: the action returns `{ notice }` ("Queued behind N report regenerations — the report updates when it finishes (job …).") past the bound, `RefreshSnapshotButton` draws an error and the notice, `InlineActionButton` shows the notice neutrally, and Update all reports answers once queued; `7ae184bd`: ops `{ slug }` returns `{ ok, jobId, started }` (404 for an unknown channel) and `{ all }` returns the ids at once, as `_lib.ts` rules — `--wait` follows them |
+| R2 — the "already queued" branch reported success when that job failed, or when the slug was mid-enqueue | `5287e8c2`: the wait reads any job's end — a failure returns the job log's `[error]` sentence, whoever started it; a slug mid-enqueue is waited for (2 s) until its id exists; +2 tests |
+| R3 — `started.stream` left open | `5287e8c2`: `requestRefreshReport` cancels every stream it starts |
+| R4 — run 4's count | this commit: 154 failed, not 68 (the log's tally) |
+| the changelog | this commit: bullet 1 says what a long wait shows |
+
+**Gates after the re-review fixes.** tsc clean at every commit; **common 2,502/2,502** (+2); **editor unit
+109/109**; e2e `dashboard-answers`, `channels`, `ops-api`, `channel-work` (only these, as asked): run 6 at `7ae184bd`: **44 passed, 0 failed, 4.7 min**
+(dashboard-answers (a): 30 polls of each page while the two regenerations ran, every answer ≤ 3.1 s).
+
+**Merge of `main` `1d5c33bf`** (U1, export 0.11.1, XP) at `52446024`: two conflicts, both appends —
+this file keeps U1's and XP's record sections before D0's under `## Record`, and `editor/CHANGELOG.md`
+keeps every `[Unreleased]` bullet (main's, then D0's). Re-gated on the merged tree: tsc clean; **common
+2,519/2,519**; **editor unit 109/109**; e2e `dashboard-answers`, `channels`, `jobs`, `channel-storage`
+(run 7): **25 passed, 0 failed, 6.4 min** (23 with the queue wait; every dashboard answer ≤ 3.7 s).
+
## Rollout
diff --git a/umtool/bin/umtool.mjs b/umtool/bin/umtool.mjs
@@ -27,11 +27,15 @@
// umtool new <slug> [--kind report-video] [--from <report.md>|<share URL>|<channel>/<id>]
// [--site-origin URL] [--seed chapters] [--brand archilyzer-media]
// umtool doctor [--json] exit 1 if the report pipeline is missing a tool
+// or the media root is set and not there
+// umtool storage [move-out|move-back <project>|--all] [--dry-run] [--json]
+// where each project's out/ lives; move it
// umtool snapshot <project> [--label L] copy the manifest into revisions/
// umtool diff <project> <snapshot> what changed since that snapshot
// umtool export <project> --format toc-bbcode|toc-markdown|description|chapters [--variant V]
// umtool check-sources [<project>…] prints the re-check chain
import process from "node:process";
+import { readdirSync, readFileSync } from "node:fs";
import {
PROJECT_KINDS,
REPORTS_ROOT,
@@ -60,6 +64,8 @@ import { buildSteps, checkSourcesSteps, PRESETS } from "../lib/report/driver.mjs
import { openIndex, signRecord } from "../lib/projects/index-db.mjs";
import { probeTools } from "../lib/tools.mjs";
import { scaffoldReportVideo } from "../lib/projects/scaffold.mjs";
+import { CACHE_DIR, INDEX_DIR, MEDIA_ROOT, MEDIA_TIERED, OLD_CACHE_DIR } from "../lib/paths.mjs";
+import { measureTree, mediaRootProblem, moveDirToLocal, moveDirToMedia, outDirState, pathState } from "../lib/report/storage.mjs";
const argv = process.argv.slice(2);
const cmd = argv.find((a) => !a.startsWith("-")) ?? "help";
@@ -338,11 +344,37 @@ function cmdKinds() {
}
}
+/**
+ * The roots, as the doctor reports them: where projects are read, where their
+ * out/ goes, where the cache is -- and a cache left where it used to live
+ * (under SONG_DATA, before release 17), which is derived and safe to delete
+ * once `umtool index` has rebuilt the new one.
+ */
+async function rootsReport() {
+ const isDir = async (p) => (await pathState(p)).kind === "dir";
+ const problem = await mediaRootProblem();
+ const oldPresent = OLD_CACHE_DIR !== CACHE_DIR && (await isDir(OLD_CACHE_DIR));
+ return {
+ ok: !problem,
+ reports: { path: REPORTS_ROOT, present: await isDir(REPORTS_ROOT) },
+ media: { path: MEDIA_ROOT, tiered: MEDIA_TIERED, present: await isDir(MEDIA_ROOT), problem },
+ cache: { path: CACHE_DIR, present: await isDir(CACHE_DIR), index: await isDir(INDEX_DIR) },
+ oldCache: oldPresent
+ ? { path: OLD_CACHE_DIR, present: true, bytes: (await measureTree(OLD_CACHE_DIR)).bytes }
+ : { path: OLD_CACHE_DIR, present: false },
+ };
+}
+
+const mb = (n) => `${(n / 1024 ** 2).toFixed(1)} MB`;
+
async function cmdDoctor() {
// The one command that shells out on purpose. Seven version flags, ~100 ms.
const r = await probeTools();
+ const roots = await rootsReport();
if (json) {
- out(r);
+ // `ok` is what the exit status says (a script gates on either); the tools'
+ // own verdict stays readable as toolsOk.
+ out({ ...r, ok: r.ok && roots.ok, toolsOk: r.ok, roots });
} else {
for (const t of r.tools) {
const mark = t.present ? "ok " : t.required ? "MISSING" : "absent";
@@ -356,8 +388,134 @@ async function cmdDoctor() {
? "\nthe report pipeline can build here"
: "\nthe report pipeline is MISSING a tool it cannot run without",
);
+ console.log("\nroots");
+ console.log(` reports ${roots.reports.path}${roots.reports.present ? "" : " (not there)"}`);
+ console.log(
+ roots.media.tiered
+ ? ` media ${roots.media.path} — every project's out/ is linked here (UMTOOL_MEDIA_DIR)` +
+ (roots.media.problem ? `\n MISSING ${roots.media.problem}` : "")
+ : ` media = reports (UMTOOL_MEDIA_DIR unset): out/ stays in each project`,
+ );
+ console.log(
+ ` cache ${roots.cache.path}` +
+ (roots.cache.index ? "" : " (no index yet — `umtool index` builds it; everything works without)"),
+ );
+ if (roots.oldCache.present) {
+ console.log(
+ ` old cache ${roots.oldCache.path} ${mb(roots.oldCache.bytes ?? 0)} — the cache's old place, ` +
+ "no longer read; derived, safe to delete",
+ );
+ }
+ }
+ process.exit(r.ok && roots.ok ? 0 : 1);
+}
+
+// ---------------------------------------------------------------------------
+// storage: where each project's out/ lives, and moving it (release 17).
+//
+// umtool storage every project's out/: dir | link | DANGLING | none
+// umtool storage move-out <p>|--all out/ to the media root, a link left in its place
+// umtool storage move-back <p>|--all out/ back to a real directory in the project
+// --dry-run measure and say; change nothing
+//
+// The movers are lib/report/storage.mjs's, which `umtool` and the app share.
+// Run a move when nothing is building: the app's jobs live in its memory, so
+// this cannot see them -- the verify refuses when the tree keeps changing, but
+// a write in the last instant before the swap would be lost with the parked copy.
+// ---------------------------------------------------------------------------
+/**
+ * The pids of report-pipeline processes whose command line names this project
+ * (its manifest, its out/, or the directory itself). Linux /proc; elsewhere,
+ * none. Cheap and coarse: it sees the pipeline's scripts, not the app's
+ * in-process deck previews.
+ */
+function pipelineProcessesFor(projectDir) {
+ const SCRIPTS = /(build-video|check-availability|render-cards|compose-chrome|verify-build|fetch-via-editor|resolve-windows|cut-from-cache|share-batch)\.mjs/;
+ const pids = [];
+ let entries = [];
+ try {
+ entries = readdirSync("/proc").filter((n) => /^\d+$/.test(n));
+ } catch {
+ return pids;
+ }
+ for (const pid of entries) {
+ if (Number(pid) === process.pid) continue;
+ let args;
+ try {
+ args = readFileSync(`/proc/${pid}/cmdline`, "utf8").split("\0");
+ } catch {
+ continue;
+ }
+ if (!args.some((a) => SCRIPTS.test(a))) continue;
+ if (args.some((a) => a === projectDir || a.startsWith(projectDir + "/"))) pids.push(Number(pid));
+ }
+ return pids;
+}
+
+async function cmdStorage() {
+ const sub = positional[0];
+ const dryRun = has("--dry-run");
+ const log = (m) => (json ? console.error(m) : console.log(m));
+
+ if (sub !== "move-out" && sub !== "move-back") {
+ // `umtool storage`, `umtool storage <project>`, `umtool storage status [<project>]`.
+ const one = sub === "status" ? positional[1] : sub;
+ const refs = one ? [await pick(one)] : await projectRefs();
+ const rows = [];
+ for (const p of refs) rows.push({ id: p.id, ...(await outDirState(p.dir)) });
+ if (json) return out({ media: { path: MEDIA_ROOT, tiered: MEDIA_TIERED }, projects: rows });
+ console.log(MEDIA_TIERED ? `media root ${MEDIA_ROOT}` : "media root unset (UMTOOL_MEDIA_DIR): out/ stays in each project");
+ for (const r of rows) {
+ const what = { absent: "none", dir: "dir", link: "link", dangling: "DANGLING", other: "OTHER" }[r.state] ?? r.state;
+ console.log(`${what.padEnd(9)} ${r.id}${r.target ? ` -> ${r.target}` : ""}`);
+ }
+ return;
+ }
+
+ const move = sub === "move-out" ? moveDirToMedia : moveDirToLocal;
+ if (sub === "move-out" && !MEDIA_TIERED) {
+ die("UMTOOL_MEDIA_DIR is not set: there is no media root to move out/ to. Set it to a directory on the media drive (outside the reports root).");
+ }
+ const all = has("--all");
+ if (!all && !positional[1]) die(`which project? \`umtool storage ${sub} <project>\` or --all`);
+ const refs = all ? await projectRefs() : [await pick(positional[1])];
+
+ const results = [];
+ let failed = 0;
+ for (const p of refs) {
+ // The app's jobs live in its memory, but the pipeline runs as processes:
+ // one whose command line names this project is building it now.
+ const busy = pipelineProcessesFor(p.dir);
+ if (busy.length) {
+ results.push({ id: p.id, state: "busy", pids: busy });
+ if (!json) console.log(`${"busy".padEnd(12)} ${p.id} — a pipeline process is writing it (pid ${busy.join(", ")}); skipped`);
+ continue;
+ }
+ try {
+ const r = await move(p.dir, "out", { dryRun, log });
+ results.push({ id: p.id, ...r });
+ if (!json && r.state !== "absent") {
+ const size = r.bytes !== undefined ? ` ${r.files} file(s), ${mb(r.bytes)}` : "";
+ console.log(`${r.state.padEnd(12)} ${p.id}${size}`);
+ if (r.mediaCopyLeft) console.log(` left in place: ${r.mediaCopyLeft} (not deleted; remove it by hand once checked)`);
+ }
+ } catch (e) {
+ failed += 1;
+ results.push({ id: p.id, state: "failed", error: e?.message ?? String(e) });
+ if (!json) console.log(`${"FAILED".padEnd(12)} ${p.id}\n ${e?.message ?? e}`);
+ }
+ }
+ const bytes = results.reduce((n, r) => n + (r.bytes ?? 0), 0);
+ if (json) out({ ok: failed === 0, dryRun, bytes, results });
+ else {
+ const moved = results.filter((r) => r.state === "moved" || r.state === "would-move").length;
+ console.log(
+ `\n${moved} project(s) ${dryRun ? "would move" : "moved"}, ${mb(bytes)}` +
+ (failed ? `; ${failed} FAILED` : "") +
+ (dryRun ? " — dry run, nothing changed" : ""),
+ );
}
- process.exit(r.ok ? 0 : 1);
+ if (failed) process.exit(1);
}
async function cmdSnapshot() {
@@ -462,6 +620,10 @@ function usage() {
" umtool new <slug> [--from <report.md>|<share URL>|<channel>/<id>] [--site-origin URL] [--seed chapters]",
" [--brand archilyzer-media] render.brand: the report-to-video brand preset",
" umtool doctor [--json] exit 1 if the report pipeline is missing a tool",
+ " or the media root is set and not there; the roots, and a leftover old cache",
+ " umtool storage [<project>] where each project's out/ lives (dir, link, DANGLING)",
+ " umtool storage move-out|move-back <project>|--all [--dry-run]",
+ " out/ to the media root (UMTOOL_MEDIA_DIR) and back; run when nothing is building",
" umtool snapshot <project> [--label L] copy the manifest into revisions/",
" umtool diff <project> <snapshot> what changed since that snapshot",
" umtool export <project> --format toc-bbcode|toc-markdown|description|chapters [--variant V]",
@@ -488,6 +650,7 @@ const COMMANDS = {
folders: cmdFolders,
kinds: cmdKinds,
doctor: cmdDoctor,
+ storage: cmdStorage,
snapshot: cmdSnapshot,
diff: cmdDiff,
export: cmdExport,
diff --git a/umtool/docs/cli.md b/umtool/docs/cli.md
@@ -25,7 +25,9 @@ decisions inbox cannot disagree about what is wrong with one.
| `build <project> [--preset preview\|fast\|final] [--only ID]` | **prints** the chain |
| `index [--rebuild] [--prune] [--since MS] [--json]` | the cache |
| `new <slug> [--from <report.md>\|<share URL>\|<channel>/<id>] [--site-origin URL] [--seed chapters]` | scaffold |
-| `doctor [--json]` | which tools are on this machine; **exit 1** if the report pipeline is missing one |
+| `doctor [--json]` | which tools are on this machine and where the roots are; **exit 1** if the report pipeline is missing one, or `UMTOOL_MEDIA_DIR` is set and not there |
+| `storage [<project>] [--json]` | where each project's `out/` lives: dir, link, DANGLING, none |
+| `storage move-out\|move-back <project>\|--all [--dry-run] [--json]` | `out/` to the media root (a link left behind) and back — copied, mirrored, verified first; run when nothing is building |
| `snapshot <project> [--label L]` | copy the manifest into `revisions/` |
| `diff <project> <snapshot>` | added / removed / moved / window / retyped, by entry id |
| `export <project> --format toc-bbcode\|toc-markdown\|description\|chapters [--variant V]` | the posting artifacts, from the build's chapter offsets |
@@ -36,7 +38,8 @@ projects answering to one name is reported, never resolved by picking one.
## Environment
-`REPORTS_DIR`, `SONG_REPORTS_DIR`, `SONG_DIR`, `CHANNELS_DIR`, `UMTOOL_INDEX_DIR`
+`REPORTS_DIR`, `SONG_REPORTS_DIR`, `SONG_DIR`, `CHANNELS_DIR`, `UMTOOL_INDEX_DIR`,
+`UMTOOL_CACHE_DIR`, `UMTOOL_MEDIA_DIR`
— which is how it is tested against the e2e fixture. The path defaults
(`lib/paths.mjs`, `song/paths.mjs`):
@@ -45,6 +48,8 @@ projects answering to one name is reported, never resolved by picking one.
| `SONG_DIR` | `~/.local/share/archilyzer/song`, through its realpath — a symlink there is the supported way to keep the data where it is |
| `SONG_REPORTS_DIR` | `~/reports/quartering-uh-song` |
| `REPORTS_DIR` | `~/reports` (the parent of `SONG_REPORTS_DIR` when that is set) |
+| `UMTOOL_MEDIA_DIR` | unset = `REPORTS_DIR`: `out/` stays in each project. Set, each project's `out` is a link to the same path under it ([folders.md](folders.md)) |
+| `UMTOOL_CACHE_DIR` | `$XDG_CACHE_HOME/archilyzer/umtool`, else `~/.cache/archilyzer/umtool` (it was `<SONG_DIR>/.cache/umtool`) |
| `CHANNELS_DIR` | `$TRANSCRIPTS_DIR/channels`, else the checkout's `transcripts/channels` (found by walking up from the cwd to `pnpm-workspace.yaml`) |
| `VIDEO_ROOT` (`song/spec.mjs`, `song/video-dir.mjs`) | `~/reports/quartering-uh-song/videos` |
diff --git a/umtool/docs/folders.md b/umtool/docs/folders.md
@@ -9,6 +9,42 @@ environment variable.
`SONG_REPORTS` (the um-song deliverables) keeps its exact previous default and is
now a *subdirectory* of `REPORTS_ROOT` rather than the widest root there is.
+## `MEDIA_ROOT`
+
+Where a project's render scratch lives (release 17). `UMTOOL_MEDIA_DIR`, else
+`REPORTS_ROOT` — and then nothing is different: `out/` is a directory in the
+project. Set, a project's `out` is an absolute link to the same project-relative
+path under it (`<REPORTS_ROOT>/a/b/out -> <MEDIA_ROOT>/a/b/out`), made by the
+first writer (`lib/report/storage.mjs` `ensureOutDir`) or by
+`umtool storage move-out`. The manifest, `revisions/`, notes and sources stay put.
+
+- It must already exist, outside `REPORTS_ROOT`: umtool never creates it, so an
+ unmounted drive is a loud refusal ("is the media drive mounted?"), never a new
+ tree on the main disk. A dangling `out` link refuses the same way.
+- Make it a directory **inside** the drive (`<mount>/umtool`), never the
+ mountpoint itself: a mountpoint that stays behind as an empty directory when
+ the drive is unmounted passes the check, and the first build of a new project
+ would make its tree on the main disk.
+- A cut move leaves `out.moved-<stamp>` or `out.incoming` beside the project's
+ `out`. While one exists, no writer makes a new `out/` and both moves refuse,
+ naming it: run the move it names again to finish it. When a real `out/` exists
+ beside the leftover too (a writer made a fresh one after the cut), the moves
+ refuse and say so: the leftover holds the moved data — keep one, remove the
+ other by hand, then run the move.
+- It is a READ root (a realpath through the link lands under it), never a write root.
+- The walk skips `out`, `clips` and `share-*`, so it never stats a link into a
+ drive that is not there.
+- `umtool storage` lists every project's `out` (dir, link, DANGLING, none);
+ `umtool doctor` names the roots and exits 1 when this one is set and missing.
+
+## `CACHE_DIR`
+
+`UMTOOL_CACHE_DIR`, else `$XDG_CACHE_HOME/archilyzer/umtool`, else
+`~/.cache/archilyzer/umtool`: the project index, posters, the mix bench's
+analyses, sliced audio. Derived, safe to delete. Until release 17 it was
+`<SONG_DIR>/.cache/umtool`; `umtool doctor` reports a leftover one, and the first
+`umtool index` rebuilds the index in the new place.
+
## The walk
Two rules do almost all the work.
diff --git a/umtool/e2e/dashboard.spec.ts b/umtool/e2e/dashboard.spec.ts
@@ -51,7 +51,14 @@ test("umtool doctor reports a deliberately bad path as absent, and exits 1", ()
stdout = execFileSync("node", ["bin/umtool.mjs", "doctor", "--json"], {
cwd: UMTOOL,
encoding: "utf8",
- env: { ...process.env, YTDLP_BIN: path.join(FIXTURE, "bin", "definitely-not-here"), QRENCODE_BIN: path.join(FIXTURE, "bin", "qrencode") },
+ env: {
+ ...process.env,
+ YTDLP_BIN: path.join(FIXTURE, "bin", "definitely-not-here"),
+ QRENCODE_BIN: path.join(FIXTURE, "bin", "qrencode"),
+ // The fixture's roots, never this shell's media root or cache (empty = unset).
+ UMTOOL_MEDIA_DIR: "",
+ UMTOOL_CACHE_DIR: path.join(FIXTURE, "cache"),
+ },
});
} catch (e) {
const err = e as { status: number; stdout: string };
diff --git a/umtool/e2e/fixtures/make-fixture.mjs b/umtool/e2e/fixtures/make-fixture.mjs
@@ -1711,6 +1711,34 @@ copyFileSync(
path.join(DASH, "out", "clips-raw", "vid1_0.00-9.00.mp4"),
);
+// -- the media root (release 17) ------------------------------------------------
+//
+// storage.spec.ts moves storage-fixture's out/ to a media root and back, builds
+// through the link, and unplugs the root. Its CLI is given UMTOOL_MEDIA_DIR =
+// this directory; the APP is not (every other spec's out/ stays a directory).
+// A SIBLING of the fixture, not inside it: REPORTS_ROOT is the fixture root, and
+// a media root inside the tree it mirrors is refused. Reset here, every run.
+// storage-fixture dash-fixture's shape: a cached window, buildable offline
+// storage-fresh-fixture no out/ at all: the first writer makes the link
+const MEDIA = `${dest}-media`;
+rmSync(MEDIA, { recursive: true, force: true });
+mkdirSync(MEDIA, { recursive: true });
+for (const slug of ["storage-fixture", "storage-fresh-fixture"]) {
+ const dir = writeProject(
+ slug,
+ manifest(slug, "The Storage Fixture", { siteOrigin: "https://archive.example" }, [
+ { type: "clip", id: "c01", video: "vid1", start: 3.0, end: 6.0, cite: 3, section: 0, lock: true, quote: "and because" },
+ { type: "clip", id: "c02", video: "vid1", start: 9.0, end: 12.0, cite: 9, section: 0, lock: true, quote: "another whole sentence" },
+ ]),
+ );
+ if (slug !== "storage-fixture") continue;
+ mkdirSync(path.join(dir, "out", "clips-raw"), { recursive: true });
+ copyFileSync(
+ path.join(REPORT, "out", "clips-raw", "vid1_0.00-9.00.mp4"),
+ path.join(dir, "out", "clips-raw", "vid1_0.00-9.00.mp4"),
+ );
+}
+
console.log(`fixture at ${dest}`);
if (planned) console.log(` planned clip (used in a build): ${planned}`);
console.log(` videos/: alpha (4 cuts, 3 variants), beta (2 cuts), deck (1 cut, 2 variants)`);
@@ -1725,6 +1753,8 @@ console.log(` flagged source: ${flagged ? flagged.video : "none — no asr/"}`)
console.log(` SONG_CODE_DIR=${path.join(dest, "code")}`);
console.log(` SONG_DIR=${path.join(dest, "data")}`);
console.log(` SONG_REPORTS_DIR=${reports}`);
+console.log(` UMTOOL_CACHE_DIR=${path.join(dest, "cache")} (removed with the fixture; never ~/.cache)`);
+console.log(` storage spec media root: ${MEDIA} (UMTOOL_MEDIA_DIR on its CLI only; reset here)`);
console.log(` YTDLP_BIN=${path.join(BIN, "yt-dlp")} QRENCODE_BIN=${path.join(BIN, "qrencode")} HYPERFRAMES_BIN=${path.join(BIN, "hyperframes")}`);
console.log(` CHANNELS_DIR=${CHANNELS} (testchan/vid1 punctuated, vid2 not; vid3/vid4/vid5 for the editor fetch)`);
console.log(` projects: report-fixture (4 clips, 1 mid-sentence), no-origin-fixture,`);
diff --git a/umtool/e2e/projects.spec.ts b/umtool/e2e/projects.spec.ts
@@ -318,6 +318,10 @@ const cliEnv = {
SONG_REPORTS_DIR: path.join(FIXTURE, "reports"),
SONG_DIR: path.join(FIXTURE, "data"),
CHANNELS_DIR: path.join(FIXTURE, "channels"),
+ // The fixture's cache, as playwright.config.ts gives the app (never ~/.cache).
+ UMTOOL_CACHE_DIR: path.join(FIXTURE, "cache"),
+ // Never the real media root, whatever this shell exports (empty = unset).
+ UMTOOL_MEDIA_DIR: "",
};
const umtool = (args: string[]) =>
execFileSync("node", ["bin/umtool.mjs", ...args], { cwd: UMTOOL, encoding: "utf8", env: cliEnv });
@@ -415,7 +419,7 @@ test("deleting the index changes nothing but latency", async ({ request }) => {
// CACHE_DIR is documented as derived output, safe to delete at any time. This
// is that promise, tested.
- rmSync(path.join(FIXTURE, "data", ".cache", "umtool", "index"), {
+ rmSync(path.join(FIXTURE, "cache", "index"), {
recursive: true,
force: true,
});
diff --git a/umtool/e2e/report-longform.spec.ts b/umtool/e2e/report-longform.spec.ts
@@ -21,6 +21,10 @@ const cliEnv = {
SONG_REPORTS_DIR: path.join(FIXTURE, "reports"),
SONG_DIR: path.join(FIXTURE, "data"),
CHANNELS_DIR: path.join(FIXTURE, "channels"),
+ // The fixture's cache, as playwright.config.ts gives the app (never ~/.cache).
+ UMTOOL_CACHE_DIR: path.join(FIXTURE, "cache"),
+ // Never the real media root, whatever this shell exports (empty = unset).
+ UMTOOL_MEDIA_DIR: "",
YTDLP_BIN: path.join(FIXTURE, "bin", "yt-dlp"),
};
const umtool = (args: string[]) =>
diff --git a/umtool/e2e/storage.spec.ts b/umtool/e2e/storage.spec.ts
@@ -0,0 +1,180 @@
+import { test, expect, type APIRequestContext } from "@playwright/test";
+import { execFileSync, spawnSync } from "node:child_process";
+import { existsSync, lstatSync, readdirSync, readlinkSync, renameSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+// ---------------------------------------------------------------------------
+// A project's render scratch on a media root (release 17, slice U1).
+//
+// With UMTOOL_MEDIA_DIR set, a project's out/ is a link to the same
+// project-relative path under it: made by the first writer, or moved there by
+// `umtool storage move-out`. Only THIS spec's CLI is given the variable (the
+// media root is make-fixture's `<fixture>-media`, reset every run); the app is
+// not, which is the point of half of it -- a reader, and a build the app runs,
+// go through the link without knowing a media root exists.
+//
+// The tests run in order and hand the project's state on: moved out, built
+// through, unplugged, plugged back, moved back.
+// ---------------------------------------------------------------------------
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const UMTOOL = path.join(HERE, "..");
+const FIXTURE = path.join(UMTOOL, ".e2e-song");
+const MEDIA = `${FIXTURE}-media`;
+const UNPLUGGED = `${MEDIA}.unplugged`;
+const PROJECT = "reports/storage-fixture";
+const FRESH = "reports/storage-fresh-fixture";
+const dirOf = (id: string) => path.join(FIXTURE, id);
+const mirrorOf = (id: string) => path.join(MEDIA, id);
+
+const env = {
+ ...process.env,
+ SONG_REPORTS_DIR: path.join(FIXTURE, "reports"),
+ SONG_DIR: path.join(FIXTURE, "data"),
+ CHANNELS_DIR: path.join(FIXTURE, "channels"),
+ UMTOOL_CACHE_DIR: path.join(FIXTURE, "cache"),
+ YTDLP_BIN: path.join(FIXTURE, "bin", "yt-dlp"),
+ UMTOOL_MEDIA_DIR: MEDIA,
+};
+const umtool = (args: string[]) =>
+ JSON.parse(execFileSync("node", ["bin/umtool.mjs", ...args, "--json"], { cwd: UMTOOL, encoding: "utf8", env }));
+// The pipeline's first writer, as a build's step 1 runs it.
+const checkAvailability = (id: string) =>
+ spawnSync(
+ "node",
+ [
+ path.join(UMTOOL, "report-to-video", "check-availability.mjs"),
+ path.join(dirOf(id), "video.manifest.json"),
+ "--out",
+ path.join(dirOf(id), "out"),
+ "--allow-missing",
+ ],
+ { cwd: path.join(UMTOOL, "report-to-video"), encoding: "utf8", env },
+ );
+
+const isLink = (p: string) => existsSync(path.dirname(p)) && lstatSync(p, { throwIfNoEntry: false })?.isSymbolicLink() === true;
+const isRealDir = (p: string) => lstatSync(p, { throwIfNoEntry: false })?.isDirectory() === true;
+const parked = (id: string) => readdirSync(dirOf(id)).filter((n) => n.startsWith("out.moved-") || n === "out.incoming");
+
+async function projectRow(request: APIRequestContext, id: string) {
+ const j = (await (await request.get("/api/browse/projects")).json()) as {
+ projects: { id: string; state: string; facts: string[] }[];
+ };
+ return j.projects.find((p) => p.id === id);
+}
+
+test.describe.configure({ mode: "serial" });
+
+test.afterAll(() => {
+ // A failure mid-way must not leave the root unplugged for the next run's
+ // reader of this file -- make-fixture resets it anyway.
+ if (existsSync(UNPLUGGED) && !existsSync(MEDIA)) renameSync(UNPLUGGED, MEDIA);
+});
+
+test("move-out leaves a link to the media root, and readers see the same project", async ({ request }) => {
+ const out = path.join(dirOf(PROJECT), "out");
+ expect(isRealDir(out)).toBe(true);
+ const before = await projectRow(request, PROJECT);
+ expect(before).toBeTruthy();
+
+ // A dry run measures and changes nothing.
+ const dry = umtool(["storage", "move-out", PROJECT, "--dry-run"]);
+ expect(dry.results[0].state).toBe("would-move");
+ expect(dry.results[0].bytes).toBeGreaterThan(0);
+ expect(isRealDir(out)).toBe(true);
+ expect(existsSync(mirrorOf(PROJECT))).toBe(false);
+
+ const moved = umtool(["storage", "move-out", PROJECT]);
+ expect(moved.ok).toBe(true);
+ expect(moved.results[0].state).toBe("moved");
+ expect(isLink(out)).toBe(true);
+ expect(readlinkSync(out)).toBe(path.join(mirrorOf(PROJECT), "out"));
+ expect(existsSync(path.join(mirrorOf(PROJECT), "out", "clips-raw", "vid1_0.00-9.00.mp4"))).toBe(true);
+ expect(parked(PROJECT)).toEqual([]);
+
+ // The index and the page read <project>/out by path; the link changes nothing.
+ const after = await projectRow(request, PROJECT);
+ expect(after?.state).toBe(before?.state);
+ expect(after?.facts).toEqual(before?.facts);
+
+ // Again: nothing to do.
+ expect(umtool(["storage", "move-out", PROJECT]).results[0].state).toBe("already");
+ const status = umtool(["storage", PROJECT]);
+ expect(status.projects[0].state).toBe("link");
+});
+
+test("a build the app runs writes through the link, and leaves it a link", async ({ request }) => {
+ const start = await request.post("/api/report/build", { data: { project: PROJECT, preset: "fast" } });
+ expect(start.ok()).toBeTruthy();
+ const { job } = (await start.json()) as { job: { id: string } };
+ let state = "running";
+ for (let i = 0; i < 150 && state === "running"; i += 1) {
+ const j = (await (await request.get("/api/jobs")).json()) as { jobs: { id: string; state: string }[] };
+ state = j.jobs.find((x) => x.id === job.id)?.state ?? "running";
+ if (state === "running") await new Promise((r) => setTimeout(r, 200));
+ }
+ expect(state).toBe("done");
+
+ const out = path.join(dirOf(PROJECT), "out");
+ expect(isLink(out)).toBe(true);
+ expect(existsSync(path.join(mirrorOf(PROJECT), "out", "storage-fixture.mp4"))).toBe(true);
+ expect(existsSync(path.join(mirrorOf(PROJECT), "out", "availability.json"))).toBe(true);
+});
+
+test("an unplugged media root refuses loudly and materialises nothing", () => {
+ renameSync(MEDIA, UNPLUGGED);
+ try {
+ expect(umtool(["storage", PROJECT]).projects[0].state).toBe("dangling");
+
+ // A moved project: the link dangles, and the first writer says why.
+ const r = checkAvailability(PROJECT);
+ expect(r.status).not.toBe(0);
+ expect(r.stderr).toContain("is the media drive mounted?");
+ expect(isLink(path.join(dirOf(PROJECT), "out"))).toBe(true);
+
+ // A project with no out/ yet: the root is stat'd, never created.
+ const fresh = checkAvailability(FRESH);
+ expect(fresh.status).not.toBe(0);
+ expect(fresh.stderr).toContain("is not there");
+ expect(existsSync(path.join(dirOf(FRESH), "out"))).toBe(false);
+
+ // Neither recreated the root on the disk it was "on".
+ expect(existsSync(MEDIA)).toBe(false);
+
+ // The doctor says so, and exits 1.
+ const doctor = spawnSync("node", ["bin/umtool.mjs", "doctor", "--json"], { cwd: UMTOOL, encoding: "utf8", env });
+ expect(doctor.status).toBe(1);
+ const roots = JSON.parse(doctor.stdout).roots;
+ expect(roots.media.tiered).toBe(true);
+ expect(roots.media.problem).toContain("is not there");
+ expect(roots.cache.path).toBe(path.join(FIXTURE, "cache"));
+ } finally {
+ renameSync(UNPLUGGED, MEDIA);
+ }
+});
+
+test("the first writer of a project with no out/ makes the link", () => {
+ const r = checkAvailability(FRESH);
+ expect(r.status, r.stderr).toBe(0);
+ const out = path.join(dirOf(FRESH), "out");
+ expect(isLink(out)).toBe(true);
+ expect(readlinkSync(out)).toBe(path.join(mirrorOf(FRESH), "out"));
+ expect(existsSync(path.join(mirrorOf(FRESH), "out", "availability.json"))).toBe(true);
+});
+
+test("move-back makes out/ a real directory again and removes the media copy", () => {
+ const back = umtool(["storage", "move-back", PROJECT]);
+ expect(back.ok).toBe(true);
+ expect(back.results[0].state).toBe("moved");
+ const out = path.join(dirOf(PROJECT), "out");
+ expect(isRealDir(out)).toBe(true);
+ expect(existsSync(path.join(out, "storage-fixture.mp4"))).toBe(true);
+ expect(parked(PROJECT)).toEqual([]);
+ // The project's mirror is gone; the root, and the other project's, are not.
+ expect(existsSync(mirrorOf(PROJECT))).toBe(false);
+ expect(existsSync(MEDIA)).toBe(true);
+ expect(isLink(path.join(dirOf(FRESH), "out"))).toBe(true);
+
+ expect(umtool(["storage", "move-back", PROJECT]).results[0].state).toBe("already");
+});
diff --git a/umtool/lib/media.ts b/umtool/lib/media.ts
@@ -2,10 +2,10 @@ import { createHash } from "node:crypto";
import { spawn } from "node:child_process";
import { execFile } from "node:child_process";
import { promisify } from "node:util";
-import { mkdir, readdir, readFile, rename, stat, writeFile } from "node:fs/promises";
+import { mkdir, readdir, readFile, realpath, rename, stat, writeFile } from "node:fs/promises";
import { existsSync } from "node:fs";
import path from "node:path";
-import { MEDIA_ROOTS, MIX_CACHE, REPORTS_ROOT, SONG_REPORTS, labelFor } from "./paths";
+import { MEDIA_ROOT, MEDIA_ROOTS, MEDIA_TIERED, MIX_CACHE, REPORTS_ROOT, SONG_REPORTS, inside, labelFor } from "./paths";
import { brightnessCurve, brightnessSteps } from "../song/flatness.mjs";
const run = promisify(execFile);
@@ -79,6 +79,22 @@ const SCRATCH_FILE = /^(poly-song-|polytmp-|seg_|i_|o_|ms\d?seg|out\.raw)/;
/** Below this is a fragment, a probe or a one-note extraction, not a track. */
const MIN_INTERESTING = 256 * 1024;
+/**
+ * A project's `out` linked to the media root (UMTOOL_MEDIA_DIR, release 17) is
+ * walked like the directory it replaced, so a tiered deliverable stays in the
+ * picker under its project. Only a link INTO the media root: every other link
+ * stays unfollowed, as it always was (SONG_DATA's 39 GB are links).
+ */
+/** The project directories that may be links to the media root (U2 adds deliverables). */
+const MEDIA_LINKS = new Set(["out"]);
+
+async function isMediaLink(e: { name: string; isSymbolicLink(): boolean }, abs: string): Promise<boolean> {
+ if (!MEDIA_TIERED || !e.isSymbolicLink() || !MEDIA_LINKS.has(e.name)) return false;
+ const real = await realpath(abs).catch(() => null);
+ if (!real || !inside(MEDIA_ROOT, real)) return false;
+ return stat(real).then((s) => s.isDirectory(), () => false);
+}
+
export type MediaRow = { path: string; label: string; size: number; mtimeMs: number };
/**
@@ -103,7 +119,7 @@ export async function listMediaUnder(root: string, maxDepth = 2, limit = 200): P
for (const e of entries) {
if (e.name.startsWith(".")) continue;
const abs = path.join(dir, e.name);
- if (e.isDirectory()) {
+ if (e.isDirectory() || (await isMediaLink(e, abs))) {
if (depth > 0 && !SCRATCH_DIR.test(e.name)) await walk(abs, depth - 1);
continue;
}
@@ -139,7 +155,7 @@ export async function listMedia(limit = 400): Promise<MediaRow[]> {
for (const e of entries) {
if (e.name.startsWith(".")) continue;
const abs = path.join(dir, e.name);
- if (e.isDirectory()) {
+ if (e.isDirectory() || (await isMediaLink(e, abs))) {
if (depth > 0 && !SCRATCH_DIR.test(e.name)) await walk(abs, depth - 1);
continue;
}
@@ -168,7 +184,13 @@ export async function listMedia(limit = 400): Promise<MediaRow[]> {
// to find nothing anybody would load, which is the opposite of the problem
// this is fixing.
const depthFor = (r: string) => (r === REPORTS_ROOT || r === SONG_REPORTS ? 2 : 1);
- for (const r of MEDIA_ROOTS) await walk(r, depthFor(r));
+ // The media root is reached through the projects' `out` links, under the
+ // project's own path; walked as a root as well, every tiered deliverable
+ // would be listed twice and the second copy would land in "other".
+ for (const r of MEDIA_ROOTS) {
+ if (MEDIA_TIERED && r === MEDIA_ROOT) continue;
+ await walk(r, depthFor(r));
+ }
out.sort((a, b) => b.mtimeMs - a.mtimeMs);
return out.slice(0, limit);
}
diff --git a/umtool/lib/paths.mjs b/umtool/lib/paths.mjs
@@ -14,9 +14,25 @@ import { SONG_DATA, SONG_REPORTS } from "../song/paths.mjs";
// record paths relative to it (make-thumb, accept-thumb) import only siblings.
export { SONG_DATA, SONG_REPORTS };
-// Derived output (sliced mp3s, waveform peaks, the project index). Lives with
-// the data, not in the repo, and is safe to delete at any time.
-export const CACHE_DIR = path.join(SONG_DATA, ".cache", "umtool");
+// Derived output (sliced mp3s, waveform peaks, the project index, posters, the
+// mix bench's analyses). Not in the repo, and safe to delete at any time.
+//
+// It used to live under SONG_DATA (`<SONG_DIR>/.cache/umtool`), which tied
+// every project's index and every report's derived files to wherever the song
+// project's 39 GB happened to sit -- a report-only machine, or one whose song
+// data is on a drive that is not mounted, had its cache follow it there
+// (release 17). It is a cache, so it goes where caches go:
+// `UMTOOL_CACHE_DIR`, else `$XDG_CACHE_HOME/archilyzer/umtool` (an empty
+// XDG_CACHE_HOME is unset, as common/lib/paths.ts reads it), else
+// `~/.cache/archilyzer/umtool`. Nothing is migrated: the first `umtool index`
+// rebuilds the index there, and every other file is re-made on demand.
+// OLD_CACHE_DIR is only for `umtool doctor`, which reports a leftover one.
+const XDG_CACHE = process.env.XDG_CACHE_HOME || path.join(os.homedir(), ".cache");
+export const CACHE_DIR = path.resolve(
+ /* turbopackIgnore: true */
+ process.env.UMTOOL_CACHE_DIR || path.join(/* turbopackIgnore: true */ XDG_CACHE, "archilyzer", "umtool"),
+);
+export const OLD_CACHE_DIR = path.join(/* turbopackIgnore: true */ SONG_DATA, ".cache", "umtool");
/**
* A file in CACHE_DIR, by name. A route names its cache files through this
@@ -61,6 +77,53 @@ export const REPORTS_ROOT = path.resolve(
const dedupe = (list) => [...new Set(list.map((p) => path.resolve(p)))];
// ---------------------------------------------------------------------------
+// MEDIA_ROOT -- where a project's RENDER SCRATCH (`out/`) lives (release 17).
+//
+// A report project's manifest, revisions/, notes and sources are small text and
+// stay under REPORTS_ROOT. Its `out/` -- fetched windows, segments, the
+// deliverable, ~18 of the 20 GB in ~/reports -- is bulk that can be re-made,
+// and belongs on a media drive. With UMTOOL_MEDIA_DIR set, a project's `out` is
+// an absolute SYMLINK to the same project-relative path under it:
+//
+// <REPORTS_ROOT>/<folder>/<project>/out -> <MEDIA_ROOT>/<folder>/<project>/out
+//
+// made by the first writer (lib/report/storage.mjs ensureOutDir) or moved there
+// by `umtool storage move-out`. Every reader keeps opening `<project>/out/...`
+// by path; the link is the only place the media drive is named.
+//
+// UNSET, MEDIA_ROOT is REPORTS_ROOT, MEDIA_TIERED is false, and nothing changes:
+// `out/` is a plain directory in the project, as it always was.
+//
+// MEDIA_ROOT is READABLE -- so a client may name a media file by its real path
+// (one taken through a project's `out` link lands under it) to the mix bench's
+// /api/mix/{media,track} -- and never WRITABLE by a client-named path: what a
+// render may write to is still WRITE_ROOTS, judged lexically, so a write to
+// `<project>/out/...` is judged by the project's place, never the link's target.
+// ---------------------------------------------------------------------------
+export const MEDIA_ROOT = path.resolve(
+ /* turbopackIgnore: true */
+ process.env.UMTOOL_MEDIA_DIR || REPORTS_ROOT,
+);
+
+/** True when render scratch goes to a media root of its own. */
+export const MEDIA_TIERED = MEDIA_ROOT !== REPORTS_ROOT;
+
+/**
+ * Where `abs` (a path under REPORTS_ROOT) is mirrored under MEDIA_ROOT, or null
+ * when it is not under REPORTS_ROOT -- a project somewhere else is never tiered.
+ * Pure: it never touches the disk. The roots are parameters so a test can name
+ * its own; the defaults are the process's.
+ */
+export function mediaMirror(abs, { reportsRoot = REPORTS_ROOT, mediaRoot = MEDIA_ROOT } = {}) {
+ const rel = path.relative(
+ /* turbopackIgnore: true */ path.resolve(/* turbopackIgnore: true */ reportsRoot),
+ path.resolve(/* turbopackIgnore: true */ abs),
+ );
+ if (rel === "" || rel.startsWith("..") || path.isAbsolute(rel)) return null;
+ return path.join(/* turbopackIgnore: true */ path.resolve(/* turbopackIgnore: true */ mediaRoot), rel);
+}
+
+// ---------------------------------------------------------------------------
// READ vs WRITE, and why they are two lists.
//
// resolveInRoots() guards both what may be OPENED and what may be RENDERED TO.
@@ -148,7 +211,7 @@ export const CHANNELS_DIR = path.resolve(
export const READ_ROOTS = dedupe(
process.env.MIX_ROOTS
? process.env.MIX_ROOTS.split(":").filter(Boolean)
- : [SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH, CHANNELS_DIR],
+ : [SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH, CHANNELS_DIR, MEDIA_ROOT],
);
export const WRITE_ROOTS = dedupe(
diff --git a/umtool/lib/paths.ts b/umtool/lib/paths.ts
@@ -8,6 +8,7 @@ import path from "node:path";
// SONG_REPORTS the um-song deliverables -- quartering-*.mp4 and their .plan.json
// SONG_SCRATCH render scratch, ABOVE SONG_DATA
// REPORTS_ROOT the tree every PROJECT hangs off -- songs AND report videos
+// MEDIA_ROOT where a project's out/ is linked to (UMTOOL_MEDIA_DIR); = REPORTS_ROOT when unset
//
// Everything the bench reads or writes must resolve inside one of these. Not
// because this is exposed -- it is a local tool on a loopback port -- but
@@ -21,7 +22,9 @@ export {
CACHE_DIR,
cacheFile,
INDEX_DIR,
+ MEDIA_ROOT,
MEDIA_ROOTS,
+ MEDIA_TIERED,
MIX_CACHE,
READ_ROOTS,
REPORTS_ROOT,
@@ -31,6 +34,7 @@ export {
WRITE_ROOTS,
inside,
labelFor,
+ mediaMirror,
resolveInRoots,
} from "./paths.mjs";
diff --git a/umtool/lib/projects/kinds.mjs b/umtool/lib/projects/kinds.mjs
@@ -47,8 +47,31 @@ export const SKIP_DIRS = new Set([
// never reaches inside one -- but a snapshot directory left behind by a
// deleted manifest must not read as a project either.
"revisions",
+ // A report's cut clips. Like out/, it may be a link to the media root
+ // (release 17), and the walk follows a link to a directory with a stat --
+ // which, on a media drive that is unplugged or stalled, is a hang or a
+ // miss, never a project.
+ "clips",
]);
+/**
+ * Name PREFIXES the walk never descends into, for the same reason as `clips`:
+ * `share-<x>/` (a deliver's zip and its staging) may be a link to the media
+ * root, and none of them can contain a project.
+ */
+export const SKIP_PREFIXES = ["share-"];
+
+/**
+ * What a cut move leaves beside the directory it moved (lib/report/storage.mjs):
+ * `out.moved-<stamp>`, `out.incoming` -- only for the names that move, so a
+ * folder of projects that merely ends in `.incoming` is still walked.
+ */
+const MOVE_LEFTOVER = /^(out|clips|share-[^/]*)\.(moved-[^/]*|incoming)$/;
+
+/** Whether the walk skips a directory entry by its name. */
+export const skipsDir = (name) =>
+ SKIP_DIRS.has(name) || SKIP_PREFIXES.some((p) => name.startsWith(p)) || MOVE_LEFTOVER.test(name);
+
const has = (names, n) => names.has(n);
const someMatch = (names, re) => [...names].some((n) => re.test(n));
diff --git a/umtool/lib/projects/walk.mjs b/umtool/lib/projects/walk.mjs
@@ -15,7 +15,7 @@
// project. The 3.1 GB is never touched.
import { readdir, readFile, realpath, stat } from "node:fs/promises";
import path from "node:path";
-import { RESERVED_BROWSE, SKIP_DIRS, detectKind } from "./kinds.mjs";
+import { RESERVED_BROWSE, detectKind, skipsDir } from "./kinds.mjs";
/** A single safe path segment: no separators, no traversal, no dotfiles. */
export const isSegment = (v) => /^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(v) && !v.includes("..");
@@ -103,7 +103,7 @@ export async function walkProjects(root, { maxDepth = MAX_DEPTH } = {}) {
for (const e of entries) {
if (e.name.startsWith(".")) continue;
- if (SKIP_DIRS.has(e.name)) continue;
+ if (skipsDir(e.name)) continue;
let isDir = e.isDirectory();
if (!isDir && e.isSymbolicLink()) {
isDir = await stat(path.join(abs, e.name)).then((s) => s.isDirectory(), () => false);
diff --git a/umtool/lib/report/export.mjs b/umtool/lib/report/export.mjs
@@ -19,6 +19,7 @@ import path from "node:path";
import { DEFAULT_VARIANT, selectVariant, segmentOffsets } from "umtool-report-to-video/build-video";
import { citeUrlFor, channelFor, readAvailability, readManifest } from "../projects/report.mjs";
import { teaserTitle } from "umtool-report-to-video/deck";
+import { outDirState } from "./storage.mjs";
export const EXPORT_FORMATS = ["toc-bbcode", "toc-markdown", "description", "chapters"];
@@ -72,6 +73,13 @@ const titleOf = (e, i) =>
export async function chapterOffsets(dir, manifest, variant) {
const entries = manifest.timeline ?? [];
const outDir = path.join(dir, "out");
+ // Read through, never made: an export writes nothing under out/. But a link
+ // to a media root that is not mounted would read below as "no build", which
+ // sends somebody off to rebuild a cut that is sitting on an unplugged drive.
+ const out = await outDirState(dir);
+ if (out.state === "dangling") {
+ return { error: `out/ is a link to ${out.target}, which is not there — is the media drive mounted?` };
+ }
const ffmetaCandidates = [path.join(outDir, variant, "chapters.ffmeta"), path.join(outDir, "chapters.ffmeta")];
for (const file of ffmetaCandidates) {
const text = await readFile(file, "utf8").catch(() => null);
diff --git a/umtool/lib/report/onscreen.mjs b/umtool/lib/report/onscreen.mjs
@@ -15,7 +15,7 @@
//
// The preview never renders and never touches the build's project or cache:
// compose-chrome's `preview: true` writes out/<variant>/chrome/deck-preview/.
-import { mkdir, readFile, rm } from "node:fs/promises";
+import { readFile, rm } from "node:fs/promises";
import path from "node:path";
import { composeChrome } from "umtool-report-to-video/compose-chrome";
import { selectVariant } from "umtool-report-to-video/build-video";
@@ -42,6 +42,7 @@ import {
} from "../projects/report.mjs";
import { normalizePostPatches } from "./manifest.mjs";
import { deckPreviewDir, feedPreviewDir, postsPreviewDir } from "./serve.mjs";
+import { ensureWriteDir } from "./storage.mjs";
/** The schedule document deck.mjs defines, built or estimated. */
/** @typedef {ReturnType<typeof estimateSchedule>} DeckSchedule */
@@ -517,7 +518,7 @@ export async function deckStill(project, variant, schedule, t) {
const want = deckPreviewDir(project.dir, variant);
const stills = path.join(outDir, "chrome", "deck-stills");
return serialised(want, async () => {
- await mkdir(stills, { recursive: true });
+ await ensureWriteDir(stills); // through ensureOutDir: out/ may be a link to the media root
const png = path.join(stills, `still-${process.pid}-${Math.random().toString(36).slice(2, 8)}.png`);
try {
/** @type {Record<string, unknown>} */
diff --git a/umtool/lib/report/storage.mjs b/umtool/lib/report/storage.mjs
@@ -0,0 +1,569 @@
+// Where a project's bulk lives: its render scratch (`out/`) on a media root, the
+// manifest and everything small beside it (release 17).
+//
+// lib/paths.mjs names the roots: REPORTS_ROOT holds every project; MEDIA_ROOT
+// (UMTOOL_MEDIA_DIR) holds the bulk; MEDIA_TIERED says they differ. A tiered
+// project's `out` is an ABSOLUTE SYMLINK to the same project-relative path
+// under MEDIA_ROOT, so every reader keeps opening `<project>/out/...` by path
+// and none of them knows the media drive exists.
+//
+// ensureOutDir the first writer's call: makes `out/` -- a link when
+// tiered, a directory when not -- and refuses loudly on a
+// dangling link instead of building a new tree beside it
+// moveDirToMedia an existing directory of the project to MEDIA_ROOT:
+// copy, mirror, verify, park, link, delete the parked copy
+// moveDirToLocal the reverse: back to a real directory in the project
+//
+// The movers take a NAME, not "out": `umtool storage move-out` moves `out/`
+// with them, and a project's deliverables (`clips/`, `share-*/`) are moved by
+// the same two calls behind the deliverables switch.
+//
+// Modelled on the editor's common/controller/relocateDir.ts, not imported from
+// it (umtool's scripts are plain .mjs under node, and that is a TypeScript
+// controller): the source is never touched until the copy verifies; `--delete`
+// only ever points at the copy under construction; every step past the copy is
+// dispatched on what is ON DISK, so a run cut at any point is finished by
+// running it again.
+//
+// THE MEDIA ROOT IS NEVER CREATED HERE. A media drive that is not mounted
+// leaves its mountpoint as an empty directory or no directory at all, and a
+// recursive mkdir would build the tree on the root filesystem and fill it. So
+// the root is stat'd first and must already be a directory; everything BELOW
+// it may be made with a recursive mkdir.
+//
+// Every path and fs call carries `turbopackIgnore`: lib/report/onscreen.mjs
+// imports this module, and the app's routes import that (plans/FACTS.md, "A
+// path joined from `process.cwd()` …").
+import { spawn } from "node:child_process";
+import { lstat, mkdir, readdir, readlink, rename, rm, rmdir, stat, statfs, symlink, unlink } from "node:fs/promises";
+import path from "node:path";
+import { MEDIA_ROOT, REPORTS_ROOT, inside, mediaMirror } from "../paths.mjs";
+
+/** The free space a move keeps on the volume it copies to, beyond the copy. */
+export const MOVE_MARGIN_BYTES = 1024 ** 3;
+
+/** The project directories the movers may move. One path segment, never a dot name. */
+const assertName = (name) => {
+ if (typeof name !== "string" || !name || name.startsWith(".") || /[\/\\\0]/.test(name)) {
+ throw new Error(`not a project directory name: ${JSON.stringify(name)}`);
+ }
+};
+
+/** The roots a call works against: the process's, unless a test names its own. */
+function rootsOf(opts = {}) {
+ const reportsRoot = path.resolve(/* turbopackIgnore: true */ opts.reportsRoot ?? REPORTS_ROOT);
+ const mediaRoot = path.resolve(/* turbopackIgnore: true */ opts.mediaRoot ?? MEDIA_ROOT);
+ return { reportsRoot, mediaRoot, tiered: mediaRoot !== reportsRoot };
+}
+
+/**
+ * What a path is RIGHT NOW. lstat, never stat: a dangling link -- the state an
+ * unmounted media drive leaves behind -- reads as absent through stat.
+ * @returns {Promise<{ kind: "missing" } | { kind: "dir" } | { kind: "link", target: string, targetIsDir: boolean } | { kind: "other" }>}
+ */
+export async function pathState(p) {
+ let st;
+ try {
+ st = await lstat(/* turbopackIgnore: true */ p);
+ } catch {
+ return { kind: "missing" };
+ }
+ if (st.isSymbolicLink()) {
+ const raw = await readlink(/* turbopackIgnore: true */ p).catch(() => "");
+ const target = path.resolve(/* turbopackIgnore: true */ path.dirname(/* turbopackIgnore: true */ p), raw);
+ const targetIsDir = await stat(/* turbopackIgnore: true */ p).then((s) => s.isDirectory(), () => false);
+ return { kind: "link", target, targetIsDir };
+ }
+ if (st.isDirectory()) return { kind: "dir" };
+ return { kind: "other" };
+}
+
+/**
+ * Why render scratch cannot go to the media root now, or null when it can (or
+ * when nothing is tiered). The root must already exist as a directory -- it is
+ * never created -- and must neither sit inside REPORTS_ROOT nor contain it (a
+ * mirror inside the tree it mirrors would be walked as projects).
+ */
+export async function mediaRootProblem(opts = {}) {
+ const { reportsRoot, mediaRoot, tiered } = rootsOf(opts);
+ if (!tiered) return null;
+ if (inside(reportsRoot, mediaRoot) || inside(mediaRoot, reportsRoot)) {
+ return `the media root ${mediaRoot} (UMTOOL_MEDIA_DIR) must be outside the reports root ${reportsRoot}, and must not contain it`;
+ }
+ const st = await stat(/* turbopackIgnore: true */ mediaRoot).catch(() => null);
+ if (!st) {
+ return `the media root ${mediaRoot} (UMTOOL_MEDIA_DIR) is not there — is its drive mounted? Nothing was created.`;
+ }
+ if (!st.isDirectory()) return `the media root ${mediaRoot} (UMTOOL_MEDIA_DIR) is not a directory`;
+ return null;
+}
+
+/**
+ * The state of a project's `out`, for a reader that wants to say WHY a build's
+ * files are not there rather than "no build": `dangling` is a link whose target
+ * is gone (the media drive is not mounted).
+ * @returns {Promise<{ state: "absent" | "dir" | "link" | "dangling" | "other", target?: string }>}
+ */
+export async function outDirState(projectDir) {
+ const s = await pathState(path.join(/* turbopackIgnore: true */ projectDir, "out"));
+ if (s.kind === "missing") return { state: "absent" };
+ if (s.kind === "link") return { state: s.targetIsDir ? "link" : "dangling", target: s.target };
+ return { state: s.kind };
+}
+
+/**
+ * Make sure `<projectDir>/out` exists, and return its path.
+ *
+ * a directory -> it, as it is (an untiered project, or one not moved yet)
+ * a link to a directory -> it
+ * a DANGLING link -> throws: the media drive is not there, and writing
+ * through the link would fail anyway -- but a
+ * recursive mkdir under it would fail with a
+ * message about some subdirectory, so this says
+ * what is actually wrong. Nothing is created.
+ * absent, tiered -> `<MEDIA_ROOT>/<project-relative>/out`, then the link
+ * absent, not tiered -> a directory, as every writer always made it
+ *
+ * A project outside REPORTS_ROOT (a hand-run script's `--out` somewhere else)
+ * is never tiered.
+ */
+export async function ensureOutDir(projectDir, opts = {}) {
+ const out = path.join(/* turbopackIgnore: true */ projectDir, "out");
+ const s = await pathState(out);
+ if (s.kind === "dir") return out;
+ if (s.kind === "link") {
+ if (s.targetIsDir) return out;
+ throw new Error(
+ `${out} is a link to ${s.target}, which is not there — is the media drive mounted? ` +
+ `Nothing was written, and nothing was created in its place.`,
+ );
+ }
+ if (s.kind === "other") throw new Error(`${out} exists and is not a directory`);
+
+ // A cut move leaves `out.moved-<ts>` or `out.incoming` beside a missing
+ // `out`. Making a fresh, empty out/ there would let the next move-out mirror
+ // it over the complete media copy (review L5): refuse, and say how to finish.
+ await assertNoLeftovers(projectDir, "out", "nothing was created");
+
+ const roots = rootsOf(opts);
+ const mirror = roots.tiered ? mediaMirror(projectDir, roots) : null;
+ if (!mirror) {
+ await mkdir(/* turbopackIgnore: true */ out, { recursive: true });
+ return out;
+ }
+ const problem = await mediaRootProblem(roots);
+ if (problem) throw new Error(`cannot make ${out}: ${problem}`);
+ // The project must exist before its mirror is made: otherwise the symlink
+ // fails and leaves an empty mirror on the media root (review N5).
+ // stat, not lstat: a project directory may itself be a link (the walk follows them).
+ if (!(await stat(/* turbopackIgnore: true */ projectDir).then((st) => st.isDirectory(), () => false))) {
+ throw new Error(`cannot make ${out}: ${projectDir} is not a directory`);
+ }
+ const target = path.join(/* turbopackIgnore: true */ mirror, "out");
+ // Recursive is safe here: the root itself was just seen to exist.
+ await mkdir(/* turbopackIgnore: true */ target, { recursive: true });
+ try {
+ await symlink(/* turbopackIgnore: true */ target, out, "dir");
+ } catch (e) {
+ // Two writers starting at once: whichever linked first won, and a link (or
+ // directory) that now resolves is as good as ours.
+ if (e?.code !== "EEXIST") throw e;
+ const again = await pathState(out);
+ if (again.kind === "dir" || (again.kind === "link" && again.targetIsDir)) return out;
+ throw e;
+ }
+ return out;
+}
+
+/**
+ * Make a directory a pipeline step writes into, and return it.
+ *
+ * When it is a project's `out` or lies under one (`out/<variant>`,
+ * `out/<variant>/chrome/deck-stills`), that `out` is made by ensureOutDir FIRST
+ * -- a link when tiered -- and only then is the rest made under it. A plain
+ * recursive mkdir of `out/<variant>/segments` was how every script made `out/`,
+ * and it would make a real directory where the link belongs. Any other
+ * directory (a hand-run script's `--out` somewhere else) is made as it always
+ * was.
+ */
+export async function ensureWriteDir(dir) {
+ let d = path.resolve(/* turbopackIgnore: true */ dir);
+ for (let i = 0; i < 4; i++) {
+ if (path.basename(/* turbopackIgnore: true */ d) === "out") {
+ await ensureOutDir(path.dirname(/* turbopackIgnore: true */ d));
+ break;
+ }
+ const up = path.dirname(/* turbopackIgnore: true */ d);
+ if (up === d) break;
+ d = up;
+ }
+ await mkdir(/* turbopackIgnore: true */ dir, { recursive: true });
+ return dir;
+}
+
+// ---------------------------------------------------------------------------
+// Measuring
+// ---------------------------------------------------------------------------
+
+/** Files and bytes under a directory. Follows no symlink. */
+export async function measureTree(dir) {
+ let bytes = 0;
+ let files = 0;
+ const stack = [dir];
+ while (stack.length) {
+ const cur = stack.pop();
+ let entries;
+ try {
+ entries = await readdir(/* turbopackIgnore: true */ cur, { withFileTypes: true });
+ } catch {
+ continue;
+ }
+ for (const e of entries) {
+ const p = path.join(/* turbopackIgnore: true */ cur, e.name);
+ if (e.isDirectory()) stack.push(p);
+ else if (e.isFile()) {
+ const st = await stat(/* turbopackIgnore: true */ p).catch(() => null);
+ if (st) {
+ bytes += st.size;
+ files += 1;
+ }
+ }
+ }
+ }
+ return { bytes, files };
+}
+
+async function freeBytes(dir) {
+ const s = await statfs(/* turbopackIgnore: true */ dir).catch(() => null);
+ return s ? Number(s.bavail) * Number(s.bsize) : null;
+}
+
+async function assertRoom(dir, bytes) {
+ const free = await freeBytes(dir);
+ if (free !== null && free < bytes + MOVE_MARGIN_BYTES) {
+ throw new Error(
+ `not enough room on ${dir}: ${gb(free)} free, the copy needs ${gb(bytes)} plus ${gb(MOVE_MARGIN_BYTES)} to spare. Nothing moved.`,
+ );
+ }
+}
+
+const gb = (n) => `${(n / 1024 ** 3).toFixed(2)} GB`;
+
+// ---------------------------------------------------------------------------
+// rsync
+// ---------------------------------------------------------------------------
+
+/** `rsync <args> <src>/ <dest>/` -- the trailing slashes copy the CONTENTS. */
+function rsync(bin, args, src, dest, log) {
+ return new Promise((resolve, reject) => {
+ const argv = [...args, `${src}/`, `${dest}/`];
+ log(`$ ${bin} ${argv.join(" ")}`);
+ const child = spawn(bin, argv, { stdio: ["ignore", "pipe", "pipe"] });
+ let output = "";
+ child.stdout.on("data", (c) => (output += c));
+ child.stderr.on("data", (c) => (output += c));
+ child.on("error", reject);
+ child.on("close", (code) => resolve({ code: code ?? 1, output }));
+ });
+}
+
+// What a dry run itemizes that is not a difference: rsync's own chatter.
+const driftLines = (output) =>
+ output
+ .split("\n")
+ .map((l) => l.trim())
+ .filter((l) => l && !l.startsWith("sending incremental") && !/^(sent|total size)/.test(l));
+
+/**
+ * COPY, MIRROR, VERIFY -- the source is not touched by any of it.
+ *
+ * 1. `rsync -a --partial`: the bytes, resumable.
+ * 2. `rsync -a --delete --info=del` toward the COPY: what changed on the
+ * source meanwhile is re-sent, what it no longer has leaves the copy
+ * (each removal in the log). Never pointed the other way: the copy is
+ * checked to be neither the source nor inside it, nor around it.
+ * 3. `rsync -a --dry-run --itemize-changes --delete` must list nothing, and
+ * the two trees must measure the same. One more mirror pass if the source
+ * moved under the check; a second difference refuses.
+ */
+async function copyMirrorVerify(src, dest, { rsyncBin, log }) {
+ if (inside(src, dest) || inside(dest, src)) {
+ throw new Error(`refusing to copy ${src} into ${dest}: one is inside the other. Nothing moved.`);
+ }
+ const copied = await rsync(rsyncBin, ["-a", "--partial"], src, dest, log);
+ if (copied.code !== 0) throw new Error(`rsync failed (exit ${copied.code}): ${copied.output.trim().split("\n").pop() ?? ""}`);
+ for (let pass = 0; ; pass++) {
+ const mirrored = await rsync(rsyncBin, ["-a", "--delete", "--info=del"], src, dest, log);
+ if (mirrored.output.trim()) log(mirrored.output.trim());
+ if (mirrored.code !== 0) throw new Error(`the mirror pass failed (exit ${mirrored.code}). The source has NOT been touched.`);
+ const check = await rsync(rsyncBin, ["-a", "--dry-run", "--itemize-changes", "--delete"], src, dest, () => {});
+ if (check.code !== 0) throw new Error(`the verify failed (exit ${check.code}). The source has NOT been touched.`);
+ const drift = driftLines(check.output);
+ if (!drift.length) break;
+ if (pass >= 1) {
+ throw new Error(
+ `the copy still differs from the source after a second mirror pass (${drift.length} item(s), first: ${drift[0]}) — ` +
+ `something is still writing into ${src}. Stop it and run the move again. The source has NOT been touched.`,
+ );
+ }
+ log(`the source changed during the copy (${drift.length} item(s)) — one more mirror pass`);
+ }
+ const [a, b] = await Promise.all([measureTree(src), measureTree(dest)]);
+ if (a.files !== b.files || a.bytes !== b.bytes) {
+ throw new Error(
+ `the copy does not measure the same: ${a.files} file(s)/${a.bytes} B against ${b.files}/${b.bytes} B. The source has NOT been touched.`,
+ );
+ }
+ return a;
+}
+
+// ---------------------------------------------------------------------------
+// The movers
+// ---------------------------------------------------------------------------
+
+const stamp = () => new Date().toISOString().replace(/[-:]/g, "").replace(/\.\d+Z$/, "Z");
+
+/** `<name>.moved-<stamp>` siblings: a move-out parked them and was cut before deleting. */
+async function parkedOf(projectDir, name) {
+ const names = await readdir(/* turbopackIgnore: true */ projectDir).catch(() => []);
+ return names
+ .filter((n) => n.startsWith(`${name}.moved-`))
+ .sort()
+ .map((n) => path.join(/* turbopackIgnore: true */ projectDir, n));
+}
+
+/** `<name>.incoming`, when a move-back left it (a cut between its copy and its rename). */
+async function incomingOf(projectDir, name) {
+ const p = path.join(/* turbopackIgnore: true */ projectDir, `${name}.incoming`);
+ return (await pathState(p)).kind === "dir" ? p : null;
+}
+
+/** Every leftover of a cut move of `<name>`: parked copies and an incoming copy. */
+export async function leftoversOf(projectDir, name) {
+ const incoming = await incomingOf(projectDir, name);
+ return [...(await parkedOf(projectDir, name)), ...(incoming ? [incoming] : [])];
+}
+
+/**
+ * Refuse while a cut move of `<name>` has left something behind. A writer that
+ * made a fresh `<name>/` there, and a move that then mirrored it over the
+ * complete copy, would lose everything but the leftover nobody names.
+ */
+async function assertNoLeftovers(projectDir, name, what, { beside = false } = {}) {
+ const left = await leftoversOf(projectDir, name);
+ if (!left.length) return;
+ const names = left.map((p) => path.basename(/* turbopackIgnore: true */ p)).join(", ");
+ if (beside) {
+ // A real `<name>/` AND a leftover: running a move again would only refuse
+ // again (re-review R1). Only a person can say which one to keep.
+ throw new Error(
+ `${name}/ and ${names} both exist in ${projectDir}: a move of ${name}/ was cut, and something made a fresh ${name}/ since. ` +
+ `The leftover holds the moved data. Keep one and remove the other by hand, then run the move; ${what}`,
+ );
+ }
+ throw new Error(
+ `a move of ${name}/ in ${projectDir} was cut (left: ${left.map((p) => path.basename(/* turbopackIgnore: true */ p)).join(", ")}) — ` +
+ `run \`umtool storage ${left.some((p) => p.endsWith(".incoming")) ? "move-back" : "move-out"} <project>\` to finish it; ${what}`,
+ );
+}
+
+const defaults = (opts) => ({
+ rsyncBin: opts.rsyncBin ?? process.env.RSYNC_BIN ?? "rsync",
+ log: opts.log ?? (() => {}),
+ dryRun: !!opts.dryRun,
+});
+
+/**
+ * Move `<projectDir>/<name>` to the same project-relative path under the media
+ * root and leave an absolute link in its place.
+ *
+ * Dispatched on the disk, so running it again finishes a run that was cut:
+ *
+ * a link to the mirror -> "already" (and a parked copy left by a cut
+ * between the link and its deletion is removed)
+ * a link anywhere else -> refused; it is not this media root's
+ * absent, a parked copy -> the cut was between the park and the link
+ * (the copy had verified): link, delete the parked copy
+ * absent, nothing parked -> "absent", nothing to move
+ * a directory -> copy, mirror, verify, rename it to
+ * `<name>.moved-<stamp>`, link, delete the parked copy
+ * a directory + a leftover -> refused: a parked copy or `<name>.incoming`
+ * beside a real directory means a writer made a
+ * fresh one after a cut move (review L5)
+ * absent + `<name>.incoming` -> refused: a cut move-back is finished by move-back
+ *
+ * `dryRun` measures and changes nothing. Nothing in the project may be writing
+ * into `<name>` while it runs (the app's jobs are in its own memory, so a CLI
+ * caller cannot see them); the verify refuses if the tree keeps changing.
+ *
+ * @param {string} projectDir
+ * @param {string} name one path segment: "out", "clips", "share-<x>"
+ * @param {{ dryRun?: boolean, log?: (m: string) => void, rsyncBin?: string, reportsRoot?: string, mediaRoot?: string }} [opts]
+ * @returns {Promise<{ state: "moved" | "already" | "absent" | "would-move" | "would-finish", src: string, dest: string, bytes?: number, files?: number, resumed?: boolean }>}
+ */
+export async function moveDirToMedia(projectDir, name, opts = {}) {
+ assertName(name);
+ const { rsyncBin, log, dryRun } = defaults(opts);
+ const roots = rootsOf(opts);
+ if (!roots.tiered) {
+ throw new Error(`UMTOOL_MEDIA_DIR is not set: there is no media root to move ${name}/ to`);
+ }
+ const mirror = mediaMirror(projectDir, roots);
+ if (!mirror) throw new Error(`${projectDir} is not under the reports root ${roots.reportsRoot}`);
+ const src = path.join(/* turbopackIgnore: true */ projectDir, name);
+ const dest = path.join(/* turbopackIgnore: true */ mirror, name);
+ const s = await pathState(src);
+
+ if (s.kind === "link") {
+ if (s.target !== dest) {
+ throw new Error(`${src} is already a link, to ${s.target} — not to ${dest}. Nothing moved.`);
+ }
+ const parked = await parkedOf(projectDir, name);
+ if (!dryRun && s.targetIsDir) {
+ for (const p of parked) await rm(/* turbopackIgnore: true */ p, { recursive: true, force: true });
+ }
+ return { state: "already", src, dest };
+ }
+ if (s.kind === "other") throw new Error(`${src} is not a directory. Nothing moved.`);
+
+ if (s.kind === "missing") {
+ const parked = await parkedOf(projectDir, name);
+ if (!parked.length) {
+ // A cut move-BACK is finished by move-back, never overtaken by a move-out.
+ if (await incomingOf(projectDir, name)) await assertNoLeftovers(projectDir, name, "nothing moved");
+ return { state: "absent", src, dest };
+ }
+ if (parked.length > 1) {
+ throw new Error(`${src} is missing and there are ${parked.length} parked copies (${parked.join(", ")}) — settle them by hand`);
+ }
+ // A parked copy exists only after the copy verified, so the media side is
+ // complete -- provided it is reachable.
+ if ((await pathState(dest)).kind !== "dir") {
+ throw new Error(`${src} was parked at ${parked[0]}, but ${dest} is not there — is the media drive mounted? Nothing changed.`);
+ }
+ if (dryRun) return { state: "would-finish", src, dest };
+ log(`finishing a cut move: linking ${src} -> ${dest}`);
+ await symlink(/* turbopackIgnore: true */ dest, src, "dir");
+ await rm(/* turbopackIgnore: true */ parked[0], { recursive: true, force: true });
+ return { state: "moved", src, dest, resumed: true };
+ }
+
+ // A real directory: the move itself -- unless a cut move left something
+ // behind, in which case this directory is a writer's fresh one.
+ await assertNoLeftovers(projectDir, name, "nothing moved", { beside: true });
+ const problem = await mediaRootProblem(roots);
+ if (problem) throw new Error(`cannot move ${src}: ${problem}`);
+ const measured = await measureTree(src);
+ if (dryRun) return { state: "would-move", src, dest, ...measured };
+ await assertRoom(roots.mediaRoot, measured.bytes);
+ await mkdir(/* turbopackIgnore: true */ dest, { recursive: true });
+ const verified = await copyMirrorVerify(src, dest, { rsyncBin, log });
+ const parked = path.join(/* turbopackIgnore: true */ projectDir, `${name}.moved-${stamp()}`);
+ await rename(/* turbopackIgnore: true */ src, parked);
+ await symlink(/* turbopackIgnore: true */ dest, src, "dir");
+ await rm(/* turbopackIgnore: true */ parked, { recursive: true, force: true });
+ log(`moved ${src} -> ${dest} (${verified.files} file(s), ${gb(verified.bytes)})`);
+ return { state: "moved", src, dest, ...verified };
+}
+
+/**
+ * The reverse: `<projectDir>/<name>`, a link into the media root, becomes a
+ * real directory in the project again, and the media copy is deleted.
+ *
+ * a directory -> "already"
+ * absent, `<name>.incoming` -> the cut was between removing the link and
+ * renaming the verified copy: rename it, and
+ * report (never delete) the project's mirror
+ * absent, nothing incoming -> "absent"
+ * a link whose target is gone -> refused: the drive is not mounted
+ * a link -> copy the target into `<name>.incoming`,
+ * mirror, verify, remove the link, rename,
+ * then delete the copy only when it is this
+ * project's own mirror under a tiered media
+ * root (and the directories above it the move
+ * left empty, never the root); any other
+ * target is left and reported (mediaCopyLeft)
+ * a parked `<name>.moved-*` -> refused: a cut move-out is finished first
+ * a directory + a leftover -> refused (a writer's fresh directory)
+ *
+ * @param {string} projectDir
+ * @param {string} name
+ * @param {{ dryRun?: boolean, log?: (m: string) => void, rsyncBin?: string, reportsRoot?: string, mediaRoot?: string }} [opts]
+ * @returns {Promise<{ state: "moved" | "already" | "absent" | "would-move" | "would-finish", src: string, from?: string, bytes?: number, files?: number, resumed?: boolean }>}
+ */
+export async function moveDirToLocal(projectDir, name, opts = {}) {
+ assertName(name);
+ const { rsyncBin, log, dryRun } = defaults(opts);
+ const roots = rootsOf(opts);
+ const src = path.join(/* turbopackIgnore: true */ projectDir, name);
+ const incoming = path.join(/* turbopackIgnore: true */ projectDir, `${name}.incoming`);
+ const s = await pathState(src);
+
+ const mirror = roots.tiered ? mediaMirror(projectDir, roots) : null;
+ const ownCopy = mirror ? path.join(/* turbopackIgnore: true */ mirror, name) : null;
+ if (s.kind === "dir") {
+ await assertNoLeftovers(projectDir, name, "nothing moved", { beside: true });
+ // A move-back cut after its rename leaves the media copy behind (review
+ // L4): report it, never delete it blind.
+ const left = ownCopy && (await pathState(ownCopy)).kind === "dir" ? ownCopy : undefined;
+ return { state: "already", src, ...(left ? { mediaCopyLeft: left } : {}) };
+ }
+ if (s.kind === "other") throw new Error(`${src} is not a directory. Nothing moved.`);
+ // A cut move-OUT is finished by move-out first.
+ if ((await parkedOf(projectDir, name)).length) await assertNoLeftovers(projectDir, name, "nothing moved");
+
+ if (s.kind === "missing") {
+ if ((await pathState(incoming)).kind !== "dir") return { state: "absent", src };
+ if (dryRun) return { state: "would-finish", src };
+ // `.incoming` outlives the link only once it verified.
+ log(`finishing a cut move: ${incoming} -> ${src}`);
+ await rename(/* turbopackIgnore: true */ incoming, src);
+ // The link is gone, so what was copied cannot be told from the disk: the
+ // project's own mirror is reported, never deleted (review L3).
+ const left = ownCopy && (await pathState(ownCopy)).kind === "dir" ? ownCopy : undefined;
+ return { state: "moved", src, resumed: true, ...(left ? { mediaCopyLeft: left } : {}) };
+ }
+
+ // A link.
+ const from = s.target;
+ if (!s.targetIsDir) {
+ throw new Error(`${src} is a link to ${from}, which is not there — is the media drive mounted? Nothing moved.`);
+ }
+ const measured = await measureTree(from);
+ if (dryRun) return { state: "would-move", src, from, ...measured };
+ await assertRoom(projectDir, measured.bytes);
+ await mkdir(/* turbopackIgnore: true */ incoming, { recursive: true });
+ const verified = await copyMirrorVerify(from, incoming, { rsyncBin, log });
+ await unlink(/* turbopackIgnore: true */ src); // the link, not what it points at
+ await rename(/* turbopackIgnore: true */ incoming, src);
+ const left = await dropMediaCopy(from, ownCopy, roots.mediaRoot);
+ if (left) log(`left ${left} in place: it is not this project's own copy on the media root`);
+ log(`moved ${from} -> ${src} (${verified.files} file(s), ${gb(verified.bytes)})`);
+ return { state: "moved", src, from, ...verified, ...(left ? { mediaCopyLeft: left } : {}) };
+}
+
+/**
+ * Delete a media copy that has been brought home, then every directory above
+ * it the move left empty, stopping at -- never removing -- the media root.
+ *
+ * ONLY this project's own mirror (`ownCopy`, null when nothing is tiered), and
+ * only when the copy that came home IS that mirror (review L3). Without
+ * UMTOOL_MEDIA_DIR the "media root" would be the reports root itself, and a
+ * hand-made `out` link into another project's real out/ would be deleted after
+ * the copy. Anything else is left where it is, and its path returned so the
+ * caller can say so.
+ * @returns {Promise<string | undefined>} the path left in place, if any
+ */
+async function dropMediaCopy(from, ownCopy, mediaRoot) {
+ if (!ownCopy || from !== ownCopy || !inside(mediaRoot, from) || from === mediaRoot) return from;
+ if ((await pathState(from)).kind !== "dir") return undefined;
+ await rm(/* turbopackIgnore: true */ from, { recursive: true, force: true });
+ for (let d = path.dirname(/* turbopackIgnore: true */ from); d !== mediaRoot && inside(mediaRoot, d); d = path.dirname(/* turbopackIgnore: true */ d)) {
+ try {
+ await rmdir(/* turbopackIgnore: true */ d);
+ } catch {
+ break; // not empty: something else lives there
+ }
+ }
+ return undefined;
+}
diff --git a/umtool/lib/report/storage.test.mjs b/umtool/lib/report/storage.test.mjs
@@ -0,0 +1,469 @@
+// A project's render scratch on a media root (release 17): the roots in
+// lib/paths.mjs, ensureOutDir / ensureWriteDir, the two movers, and the walk's
+// skip rules for a media drive that is not there.
+//
+// Every call here names its own roots (`reportsRoot`, `mediaRoot`), so nothing
+// depends on, or touches, the process's REPORTS_ROOT. The env-derived roots are
+// tested in a child process, where the module is evaluated fresh.
+//
+// Run with: pnpm test:scripts
+import assert from "node:assert/strict";
+import { execFileSync } from "node:child_process";
+import { lstat, mkdir, mkdtemp, readFile, readdir, readlink, rename, rm, symlink, writeFile } from "node:fs/promises";
+import { existsSync } from "node:fs";
+import { homedir, tmpdir } from "node:os";
+import path from "node:path";
+import test from "node:test";
+import { fileURLToPath } from "node:url";
+
+import { mediaMirror } from "../paths.mjs";
+import { SKIP_DIRS, skipsDir } from "../projects/kinds.mjs";
+import { walkProjects } from "../projects/walk.mjs";
+import {
+ ensureOutDir,
+ ensureWriteDir,
+ mediaRootProblem,
+ moveDirToLocal,
+ moveDirToMedia,
+ outDirState,
+} from "./storage.mjs";
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+
+/** A reports root with one project, and a media root beside it (not inside). */
+async function world({ withOut = true } = {}) {
+ const base = await mkdtemp(path.join(tmpdir(), "umtool-storage-"));
+ const reportsRoot = path.join(base, "reports");
+ const mediaRoot = path.join(base, "media");
+ const projectDir = path.join(reportsRoot, "folder", "proj");
+ await mkdir(projectDir, { recursive: true });
+ await mkdir(mediaRoot);
+ await writeFile(path.join(projectDir, "video.manifest.json"), "{}\n");
+ if (withOut) {
+ await mkdir(path.join(projectDir, "out", "clips-raw"), { recursive: true });
+ await writeFile(path.join(projectDir, "out", "proj.mp4"), Buffer.alloc(4096, 1));
+ await writeFile(path.join(projectDir, "out", "clips-raw", "v_0-9.mp4"), Buffer.alloc(2048, 2));
+ }
+ const roots = { reportsRoot, mediaRoot };
+ const mirror = path.join(mediaRoot, "folder", "proj");
+ return { base, reportsRoot, mediaRoot, projectDir, roots, mirror, done: () => rm(base, { recursive: true, force: true }) };
+}
+
+const kind = async (p) => {
+ const st = await lstat(p).catch(() => null);
+ if (!st) return "missing";
+ return st.isSymbolicLink() ? "link" : st.isDirectory() ? "dir" : "file";
+};
+
+// ---------------------------------------------------------------------------
+// The roots
+// ---------------------------------------------------------------------------
+
+/** lib/paths.mjs's values under a given environment, evaluated fresh. */
+function rootsUnder(envPatch) {
+ const env = { ...process.env };
+ for (const k of ["UMTOOL_MEDIA_DIR", "UMTOOL_CACHE_DIR", "UMTOOL_INDEX_DIR", "XDG_CACHE_HOME", "MIX_ROOTS", "MIX_WRITE_ROOTS"]) delete env[k];
+ Object.assign(env, { REPORTS_DIR: "/r/reports", SONG_DIR: "/r/song-data-that-is-not-there" }, envPatch);
+ const code =
+ "const m = await import(process.argv[1]);" +
+ "console.log(JSON.stringify({ media: m.MEDIA_ROOT, tiered: m.MEDIA_TIERED, reports: m.REPORTS_ROOT, cache: m.CACHE_DIR," +
+ " old: m.OLD_CACHE_DIR, index: m.INDEX_DIR, read: m.READ_ROOTS, write: m.WRITE_ROOTS }));";
+ const out = execFileSync(process.execPath, ["--input-type=module", "-e", code, path.join(HERE, "..", "paths.mjs")], {
+ env,
+ encoding: "utf8",
+ });
+ return JSON.parse(out);
+}
+
+test("MEDIA_ROOT unset is REPORTS_ROOT: nothing tiered, no new root", () => {
+ const r = rootsUnder({});
+ assert.equal(r.media, "/r/reports");
+ assert.equal(r.tiered, false);
+ assert.equal(r.read.filter((p) => p === "/r/reports").length, 1);
+});
+
+test("UMTOOL_MEDIA_DIR is a READ root and never a write root", () => {
+ const r = rootsUnder({ UMTOOL_MEDIA_DIR: "/m/umtool" });
+ assert.equal(r.media, "/m/umtool");
+ assert.equal(r.tiered, true);
+ assert.ok(r.read.includes("/m/umtool"));
+ assert.ok(!r.write.includes("/m/umtool"));
+});
+
+test("CACHE_DIR: UMTOOL_CACHE_DIR, else XDG_CACHE_HOME/archilyzer/umtool, else ~/.cache — never SONG_DATA", () => {
+ assert.equal(rootsUnder({ UMTOOL_CACHE_DIR: "/c/u" }).cache, "/c/u");
+ assert.equal(rootsUnder({ XDG_CACHE_HOME: "/x" }).cache, "/x/archilyzer/umtool");
+ const plain = rootsUnder({ XDG_CACHE_HOME: "" });
+ assert.equal(plain.cache, path.join(homedir(), ".cache", "archilyzer", "umtool"));
+ assert.equal(plain.index, path.join(plain.cache, "index"));
+ assert.equal(plain.old, "/r/song-data-that-is-not-there/.cache/umtool");
+ assert.ok(!plain.cache.startsWith("/r/song-data"));
+});
+
+test("mediaMirror: the project-relative path under the media root, or null outside the reports root", () => {
+ const roots = { reportsRoot: "/r", mediaRoot: "/m" };
+ assert.equal(mediaMirror("/r/a/b", roots), "/m/a/b");
+ assert.equal(mediaMirror("/r", roots), null);
+ assert.equal(mediaMirror("/elsewhere/p", roots), null);
+ assert.equal(mediaMirror("/r-sibling/p", roots), null);
+});
+
+// ---------------------------------------------------------------------------
+// ensureOutDir / ensureWriteDir
+// ---------------------------------------------------------------------------
+
+test("ensureOutDir, not tiered: a plain directory, as every writer made it", async () => {
+ const w = await world({ withOut: false });
+ try {
+ const out = await ensureOutDir(w.projectDir, { reportsRoot: w.reportsRoot, mediaRoot: w.reportsRoot });
+ assert.equal(await kind(out), "dir");
+ assert.deepEqual(await readdir(w.mediaRoot), []);
+ } finally {
+ await w.done();
+ }
+});
+
+test("ensureOutDir, tiered and absent: the mirror is made under the media root and linked", async () => {
+ const w = await world({ withOut: false });
+ try {
+ const out = await ensureOutDir(w.projectDir, w.roots);
+ assert.equal(await kind(out), "link");
+ assert.equal(await readlink(out), path.join(w.mirror, "out"));
+ assert.equal(await kind(path.join(w.mirror, "out")), "dir");
+ assert.deepEqual(await outDirState(w.projectDir), { state: "link", target: path.join(w.mirror, "out") });
+ // Again: the link is kept as it is.
+ assert.equal(await ensureOutDir(w.projectDir, w.roots), out);
+ } finally {
+ await w.done();
+ }
+});
+
+test("ensureOutDir keeps an existing real out/ even when tiered (move-out moves it, not the writer)", async () => {
+ const w = await world();
+ try {
+ await ensureOutDir(w.projectDir, w.roots);
+ assert.equal(await kind(path.join(w.projectDir, "out")), "dir");
+ assert.deepEqual(await readdir(w.mediaRoot), []);
+ } finally {
+ await w.done();
+ }
+});
+
+test("ensureOutDir never creates the media root: an unplugged drive refuses and nothing is made", async () => {
+ const w = await world({ withOut: false });
+ try {
+ await rm(w.mediaRoot, { recursive: true });
+ await assert.rejects(ensureOutDir(w.projectDir, w.roots), /is not there — is its drive mounted\? Nothing was created/);
+ assert.equal(existsSync(w.mediaRoot), false);
+ assert.equal(await kind(path.join(w.projectDir, "out")), "missing");
+ } finally {
+ await w.done();
+ }
+});
+
+test("a dangling out link refuses loudly; nothing is materialised in its place or under it", async () => {
+ const w = await world({ withOut: false });
+ try {
+ await ensureOutDir(w.projectDir, w.roots);
+ await rename(w.mediaRoot, `${w.mediaRoot}.unplugged`);
+ assert.equal((await outDirState(w.projectDir)).state, "dangling");
+ await assert.rejects(ensureOutDir(w.projectDir, w.roots), /is the media drive mounted\?/);
+ await assert.rejects(ensureWriteDir(path.join(w.projectDir, "out", "sourced", "segments")), /is the media drive mounted\?/);
+ // And a plain recursive mkdir through the link fails too (ENOTDIR), making nothing.
+ await assert.rejects(mkdir(path.join(w.projectDir, "out", "clips-raw"), { recursive: true }));
+ assert.equal(await kind(path.join(w.projectDir, "out")), "link");
+ assert.equal(existsSync(w.mediaRoot), false);
+ } finally {
+ await w.done();
+ }
+});
+
+test("a media root inside the reports root (or around it) is refused", async () => {
+ const w = await world({ withOut: false });
+ try {
+ const inner = { reportsRoot: w.reportsRoot, mediaRoot: path.join(w.reportsRoot, "media") };
+ await mkdir(inner.mediaRoot);
+ assert.match(await mediaRootProblem(inner), /must be outside the reports root/);
+ assert.match(await mediaRootProblem({ reportsRoot: w.reportsRoot, mediaRoot: w.base }), /must be outside/);
+ assert.equal(await mediaRootProblem(w.roots), null);
+ assert.equal(await mediaRootProblem({ reportsRoot: w.reportsRoot, mediaRoot: w.reportsRoot }), null);
+ } finally {
+ await w.done();
+ }
+});
+
+test("ensureWriteDir makes a project's out/ through ensureOutDir before anything under it", async () => {
+ const w = await world({ withOut: false });
+ try {
+ // ensureWriteDir uses the process's roots; under the default (no
+ // UMTOOL_MEDIA_DIR in the test environment) it is a plain directory, the
+ // old behaviour. The tiered case is ensureOutDir's, tested above, and the
+ // storage e2e drives the pipeline scripts through it.
+ const deep = path.join(w.projectDir, "out", "sourced", "chrome", "deck-stills");
+ assert.equal(await ensureWriteDir(deep), deep);
+ assert.equal(await kind(deep), "dir");
+ // A directory that is not under any out/ is made as it always was.
+ const other = path.join(w.base, "elsewhere", "x");
+ await ensureWriteDir(other);
+ assert.equal(await kind(other), "dir");
+ } finally {
+ await w.done();
+ }
+});
+
+// ---------------------------------------------------------------------------
+// The movers
+// ---------------------------------------------------------------------------
+
+test("moveDirToMedia: copy, verify, link; the bytes are the same and nothing is left parked", async () => {
+ const w = await world();
+ try {
+ const logs = [];
+ const r = await moveDirToMedia(w.projectDir, "out", { ...w.roots, log: (m) => logs.push(m) });
+ assert.equal(r.state, "moved");
+ assert.equal(r.files, 2);
+ assert.equal(r.bytes, 4096 + 2048);
+ const out = path.join(w.projectDir, "out");
+ assert.equal(await kind(out), "link");
+ assert.equal(await readlink(out), path.join(w.mirror, "out"));
+ assert.deepEqual(await readFile(path.join(out, "proj.mp4")), Buffer.alloc(4096, 1));
+ assert.deepEqual((await readdir(w.projectDir)).sort(), ["out", "video.manifest.json"]);
+ assert.ok(logs.some((l) => l.includes("--delete")), "the mirror pass ran");
+ // Again: nothing to do.
+ assert.equal((await moveDirToMedia(w.projectDir, "out", w.roots)).state, "already");
+ } finally {
+ await w.done();
+ }
+});
+
+test("moveDirToMedia --dry-run measures and changes nothing", async () => {
+ const w = await world();
+ try {
+ const r = await moveDirToMedia(w.projectDir, "out", { ...w.roots, dryRun: true });
+ assert.equal(r.state, "would-move");
+ assert.equal(r.bytes, 6144);
+ assert.equal(await kind(path.join(w.projectDir, "out")), "dir");
+ assert.deepEqual(await readdir(w.mediaRoot), []);
+ } finally {
+ await w.done();
+ }
+});
+
+test("moveDirToMedia refuses without a media root, and on a link that is not its own", async () => {
+ const w = await world();
+ try {
+ await assert.rejects(
+ moveDirToMedia(w.projectDir, "out", { reportsRoot: w.reportsRoot, mediaRoot: w.reportsRoot }),
+ /UMTOOL_MEDIA_DIR is not set/,
+ );
+ await assert.rejects(moveDirToMedia(w.projectDir, "../x", w.roots), /not a project directory name/);
+ const other = path.join(w.base, "other");
+ await mkdir(other);
+ await symlink(other, path.join(w.projectDir, "clips"));
+ await assert.rejects(moveDirToMedia(w.projectDir, "clips", w.roots), /already a link, to .* not to/);
+ assert.equal((await moveDirToMedia(w.projectDir, "share-none", w.roots)).state, "absent");
+ } finally {
+ await w.done();
+ }
+});
+
+test("moveDirToMedia: a failed copy leaves the source untouched", async () => {
+ const w = await world();
+ try {
+ await assert.rejects(moveDirToMedia(w.projectDir, "out", { ...w.roots, rsyncBin: "false" }), /rsync failed/);
+ assert.equal(await kind(path.join(w.projectDir, "out")), "dir");
+ assert.deepEqual(await readFile(path.join(w.projectDir, "out", "proj.mp4")), Buffer.alloc(4096, 1));
+ } finally {
+ await w.done();
+ }
+});
+
+test("moveDirToMedia finishes a run cut between the park and the link", async () => {
+ const w = await world();
+ try {
+ // What a cut leaves: the verified copy on the media root, the source parked.
+ await mkdir(path.join(w.mirror), { recursive: true });
+ execFileSync("cp", ["-a", path.join(w.projectDir, "out"), path.join(w.mirror, "out")]);
+ await rename(path.join(w.projectDir, "out"), path.join(w.projectDir, "out.moved-20261001T000000Z"));
+ const r = await moveDirToMedia(w.projectDir, "out", w.roots);
+ assert.equal(r.state, "moved");
+ assert.equal(r.resumed, true);
+ assert.equal(await kind(path.join(w.projectDir, "out")), "link");
+ assert.deepEqual((await readdir(w.projectDir)).sort(), ["out", "video.manifest.json"]);
+ } finally {
+ await w.done();
+ }
+});
+
+test("moveDirToMedia removes a parked copy left by a cut after the link", async () => {
+ const w = await world();
+ try {
+ await moveDirToMedia(w.projectDir, "out", w.roots);
+ await mkdir(path.join(w.projectDir, "out.moved-20261001T000000Z"));
+ assert.equal((await moveDirToMedia(w.projectDir, "out", w.roots)).state, "already");
+ assert.deepEqual((await readdir(w.projectDir)).sort(), ["out", "video.manifest.json"]);
+ } finally {
+ await w.done();
+ }
+});
+
+test("moveDirToLocal: a real directory again, the media copy and its empty parents gone, the root kept", async () => {
+ const w = await world();
+ try {
+ await moveDirToMedia(w.projectDir, "out", w.roots);
+ const r = await moveDirToLocal(w.projectDir, "out", w.roots);
+ assert.equal(r.state, "moved");
+ assert.equal(r.bytes, 6144);
+ const out = path.join(w.projectDir, "out");
+ assert.equal(await kind(out), "dir");
+ assert.deepEqual(await readFile(path.join(out, "clips-raw", "v_0-9.mp4")), Buffer.alloc(2048, 2));
+ assert.deepEqual((await readdir(w.projectDir)).sort(), ["out", "video.manifest.json"]);
+ assert.equal(existsSync(path.join(w.mediaRoot, "folder")), false);
+ assert.equal(existsSync(w.mediaRoot), true);
+ assert.equal((await moveDirToLocal(w.projectDir, "out", w.roots)).state, "already");
+ } finally {
+ await w.done();
+ }
+});
+
+test("moveDirToLocal refuses a dangling link and finishes a cut rename", async () => {
+ const w = await world();
+ try {
+ await moveDirToMedia(w.projectDir, "out", w.roots);
+ await rename(w.mediaRoot, `${w.mediaRoot}.unplugged`);
+ await assert.rejects(moveDirToLocal(w.projectDir, "out", w.roots), /is the media drive mounted\? Nothing moved/);
+ assert.equal(await kind(path.join(w.projectDir, "out")), "link");
+ await rename(`${w.mediaRoot}.unplugged`, w.mediaRoot);
+
+ // A cut between removing the link and renaming the verified copy. The link
+ // is gone, so what was copied cannot be told: the mirror is reported, kept.
+ execFileSync("cp", ["-a", path.join(w.mirror, "out"), path.join(w.projectDir, "out.incoming")]);
+ await rm(path.join(w.projectDir, "out"));
+ const r = await moveDirToLocal(w.projectDir, "out", w.roots);
+ assert.equal(r.state, "moved");
+ assert.equal(r.resumed, true);
+ assert.equal(await kind(path.join(w.projectDir, "out")), "dir");
+ assert.equal(r.mediaCopyLeft, path.join(w.mirror, "out"));
+ assert.equal(existsSync(path.join(w.mirror, "out")), true);
+ // And from then on "already" keeps saying so (review L4).
+ const again = await moveDirToLocal(w.projectDir, "out", w.roots);
+ assert.equal(again.state, "already");
+ assert.equal(again.mediaCopyLeft, path.join(w.mirror, "out"));
+ } finally {
+ await w.done();
+ }
+});
+
+// ---------------------------------------------------------------------------
+// The review's fixes (L3, L5, N5)
+// ---------------------------------------------------------------------------
+
+test("move-back without a media root never deletes what the link pointed at (L3)", async () => {
+ const w = await world({ withOut: false });
+ try {
+ // Another project's REAL out/, and a hand-made link to it.
+ const other = path.join(w.reportsRoot, "other", "out");
+ await mkdir(other, { recursive: true });
+ await writeFile(path.join(other, "keep.mp4"), Buffer.alloc(1024, 3));
+ await symlink(other, path.join(w.projectDir, "out"));
+ const untiered = { reportsRoot: w.reportsRoot, mediaRoot: w.reportsRoot };
+ const r = await moveDirToLocal(w.projectDir, "out", untiered);
+ assert.equal(r.state, "moved");
+ assert.equal(r.mediaCopyLeft, other);
+ assert.equal(await kind(path.join(w.projectDir, "out")), "dir");
+ assert.deepEqual(await readFile(path.join(other, "keep.mp4")), Buffer.alloc(1024, 3));
+ } finally {
+ await w.done();
+ }
+});
+
+test("a tiered move-back of a link that is not the project's own mirror leaves the target (L3)", async () => {
+ const w = await world({ withOut: false });
+ try {
+ const elsewhere = path.join(w.mediaRoot, "someone-else", "out");
+ await mkdir(elsewhere, { recursive: true });
+ await writeFile(path.join(elsewhere, "x.mp4"), Buffer.alloc(512, 4));
+ await symlink(elsewhere, path.join(w.projectDir, "out"));
+ const r = await moveDirToLocal(w.projectDir, "out", w.roots);
+ assert.equal(r.mediaCopyLeft, elsewhere);
+ assert.equal(existsSync(path.join(elsewhere, "x.mp4")), true);
+ } finally {
+ await w.done();
+ }
+});
+
+test("a cut move's leftovers are a guard: no fresh out/, no move over them (L5)", async () => {
+ const w = await world({ withOut: false });
+ try {
+ // A move-out cut after the park: out/ is missing, the parked copy is the data.
+ const parked = path.join(w.projectDir, "out.moved-20261001T000000Z");
+ await mkdir(parked);
+ await writeFile(path.join(parked, "proj.mp4"), Buffer.alloc(4096, 1));
+ for (const roots of [w.roots, { reportsRoot: w.reportsRoot, mediaRoot: w.reportsRoot }]) {
+ await assert.rejects(ensureOutDir(w.projectDir, roots), /was cut \(left: out\.moved-20261001T000000Z\).*move-out <project>.*nothing was created/);
+ }
+ assert.equal(await kind(path.join(w.projectDir, "out")), "missing");
+ // A writer that made a fresh out/ anyway (an older binary): move-out refuses to mirror it over.
+ await mkdir(path.join(w.projectDir, "out"));
+ // Both exist: neither move sends the person to the other, which would only
+ // refuse again (re-review R1); the sentence says what to do by hand.
+ for (const move of [moveDirToMedia, moveDirToLocal]) {
+ await assert.rejects(
+ move(w.projectDir, "out", w.roots),
+ /out\/ and out\.moved-20261001T000000Z both exist .*The leftover holds the moved data\. Keep one and remove the other by hand, then run the move; nothing moved/,
+ );
+ }
+ assert.deepEqual(await readdir(w.mediaRoot), []);
+ await rm(path.join(w.projectDir, "out"), { recursive: true });
+ await rm(parked, { recursive: true });
+
+ // A move-back cut after the unlink: move-out refuses; move-back finishes it.
+ await mkdir(path.join(w.projectDir, "out.incoming"));
+ await assert.rejects(moveDirToMedia(w.projectDir, "out", w.roots), /move-back <project>.*nothing moved/);
+ await assert.rejects(ensureOutDir(w.projectDir, w.roots), /left: out\.incoming/);
+ assert.equal((await moveDirToLocal(w.projectDir, "out", w.roots)).state, "moved");
+ assert.equal(await kind(path.join(w.projectDir, "out")), "dir");
+ } finally {
+ await w.done();
+ }
+});
+
+test("ensureOutDir for a project that does not exist leaves no empty mirror (N5)", async () => {
+ const w = await world({ withOut: false });
+ try {
+ const ghost = path.join(w.reportsRoot, "ghost");
+ await assert.rejects(ensureOutDir(ghost, w.roots), /is not a directory/);
+ assert.deepEqual(await readdir(w.mediaRoot), []);
+ // A project directory that is itself a link is a directory (re-review R2).
+ const real = path.join(w.base, "elsewhere-proj");
+ await mkdir(real);
+ const linked = path.join(w.reportsRoot, "linked");
+ await symlink(real, linked);
+ assert.equal(await kind(await ensureOutDir(linked, w.roots)), "link");
+ } finally {
+ await w.done();
+ }
+});
+
+// ---------------------------------------------------------------------------
+// The walk never descends what may sit on the media drive
+// ---------------------------------------------------------------------------
+
+test("the project walk skips out, clips and share-* (a link into an unplugged drive is never stat'd)", async () => {
+ assert.ok(SKIP_DIRS.has("out") && SKIP_DIRS.has("clips"));
+ assert.ok(skipsDir("share-emancipation") && skipsDir("clips") && !skipsDir("shares") && !skipsDir("project"));
+ assert.ok(skipsDir("out.moved-20261001T000000Z") && skipsDir("out.incoming") && !skipsDir("incoming"));
+ assert.ok(skipsDir("share-x.incoming") && !skipsDir("drafts.incoming") && !skipsDir("old.moved-2026"));
+ const w = await world();
+ try {
+ for (const hidden of ["clips", "share-x"]) {
+ const d = path.join(w.reportsRoot, hidden, "inner");
+ await mkdir(d, { recursive: true });
+ await writeFile(path.join(d, "video.manifest.json"), "{}\n");
+ }
+ const ids = (await walkProjects(w.reportsRoot)).map((p) => p.id);
+ assert.deepEqual(ids, ["folder/proj"]);
+ } finally {
+ await w.done();
+ }
+});
diff --git a/umtool/next.config.ts b/umtool/next.config.ts
@@ -19,7 +19,8 @@ const nextConfig: NextConfig = {
// with "Can't resolve 'cbor-x'", which names a package nothing here uses.
serverExternalPackages: ["lmdb"],
// No route's trace may list the e2e fixture (.e2e-song, where
- // e2e/fixtures/make-fixture.mjs links the song data), the e2e server's own
+ // e2e/fixtures/make-fixture.mjs links the song data; .e2e-song-media, the
+ // storage spec's media root), the e2e server's own
// build directory (.next-e2e) or an env file: none is a run-time input. The
// clip-audio route's trace listed 1,704 such files (plans/release-15.md, slice
// UT). That was fixed at the call (lib/paths.mjs `cacheFile`); this is the
@@ -27,7 +28,7 @@ const nextConfig: NextConfig = {
// back to what the sibling routes list. scripts/next-build-trace.test.mjs
// reads the last build's traces back.
outputFileTracingExcludes: {
- "/*": ["./.e2e-song/**/*", "./.next-e2e/**/*", "./.env*"],
+ "/*": ["./.e2e-song/**/*", "./.e2e-song-media/**/*", "./.next-e2e/**/*", "./.env*"],
},
turbopack: {
// Same reasoning as editor/next.config.ts: Turbopack infers the workspace
diff --git a/umtool/playwright.config.ts b/umtool/playwright.config.ts
@@ -76,6 +76,15 @@ export default defineConfig({
// var. CHANNELS_DIR has to be said explicitly: it is where a report
// video's cue files live, and its default is the real 3 GB corpus.
`CHANNELS_DIR=${FIXTURE}/channels ` +
+ // The cache (the project index, posters, analyses) is no longer under
+ // SONG_DIR (release 17): its default is the user's ~/.cache, which a
+ // suite must never write. The fixture's own, rebuilt with it every run.
+ `UMTOOL_CACHE_DIR=${FIXTURE}/cache ` +
+ // And never the media root: Playwright hands the app this shell's whole
+ // environment, so a shell that exports UMTOOL_MEDIA_DIR would put every
+ // fixture build's out/ on the real media drive. Empty is unset
+ // (lib/paths.mjs reads it with ||); storage.spec.ts gives its CLI its own.
+ `UMTOOL_MEDIA_DIR= ` +
// Stub binaries, so a build spec is offline and deterministic. The
// pipeline already reads both as overrides; the fixture writes them.
`YTDLP_BIN=${FIXTURE}/bin/yt-dlp QRENCODE_BIN=${FIXTURE}/bin/qrencode ` +
diff --git a/umtool/report-to-video/build-video.mjs b/umtool/report-to-video/build-video.mjs
@@ -77,6 +77,7 @@ import {
cardWidth, contentWidth, reservedFooterHeight,
} from "./render-cards.mjs";
import { createCueSource, siteOriginFromManifest } from "./cues.mjs";
+import { ensureWriteDir } from "../lib/report/storage.mjs";
// The deck (`render.chrome`): its geometry, validation and schedule are pure
// and live in deck.mjs. This file only frames segments into its box and writes
// the schedule down -- it never has a copy of the arithmetic.
@@ -3371,6 +3372,11 @@ export async function buildVideo({ manifestPath, opts = {}, out, only, fetchOnly
const dirs = variantPaths(outRoot, manifest.slug, variant);
const outDir = dirs.dir;
+ // A project's out/ first, through ensureOutDir: with UMTOOL_MEDIA_DIR set it
+ // is a link to the media root, and the recursive mkdirs below would
+ // otherwise make it a real directory here. A dangling link refuses here,
+ // before a byte is fetched.
+ await ensureWriteDir(outRoot);
await mkdir(dirs.rawDir, { recursive: true });
for (const d of ["cards", "segments", "qr"]) {
await mkdir(path.join(outDir, d), { recursive: true });
diff --git a/umtool/report-to-video/check-availability.mjs b/umtool/report-to-video/check-availability.mjs
@@ -22,10 +22,11 @@
import { execFile } from "node:child_process";
import { promisify } from "node:util";
-import { mkdir, readFile, writeFile } from "node:fs/promises";
+import { readFile, writeFile } from "node:fs/promises";
import path from "node:path";
import { DEFAULT_CHANNELS_DIR } from "./cues.mjs";
+import { ensureWriteDir } from "../lib/report/storage.mjs";
// The per-platform yt-dlp args (Rumble's `--impersonate chrome`): the ONE table,
// in common, plain JS so bare `node` can load it.
import { platformArgsForUrl } from "yt-dlp-transcript-common/ytdlp/platformArgs.mjs";
@@ -149,7 +150,8 @@ export async function checkAvailability(manifestPath, { outDir, maxAgeDays = 0 }
}
const report = { manifest: path.resolve(manifestPath), checkedAt: new Date().toISOString(), sources };
- await mkdir(dir, { recursive: true });
+ // ensureWriteDir, not mkdir: a project's out/ may belong on the media root.
+ await ensureWriteDir(dir);
await writeFile(file, JSON.stringify(report, null, 2) + "\n", "utf8");
return { ...report, file };
}
diff --git a/umtool/report-to-video/compose-chrome.mjs b/umtool/report-to-video/compose-chrome.mjs
@@ -38,6 +38,7 @@ import path from "node:path";
import { ledgerTotals, dateKey } from "./ledger-totals.mjs";
import { selectVariant } from "./build-video.mjs";
+import { ensureWriteDir } from "../lib/report/storage.mjs";
import {
chromeCacheKey, deckLayout, frameCount, hyperframesCommand, postWindows, resolveDeck, sha256, teaserSeconds, transitionOf,
validateTeaser,
@@ -757,6 +758,10 @@ export async function composeChrome({
const manifest = selectVariant(JSON.parse(await readFile(manifestPath, "utf8")), variant);
// Absolute: the still is a file:// URL, and a relative one is no page at all.
const base = path.resolve(outDir ?? path.join(path.dirname(path.resolve(manifestPath)), "out", variant));
+ // The project's out/ through ensureOutDir before anything lands under it: a
+ // link to the media root when UMTOOL_MEDIA_DIR is set, and a loud refusal
+ // when that link dangles.
+ await ensureWriteDir(base);
from = Number(from ?? 0);
// The regions keyed by the render cache: the two drawn from the deck's
diff --git a/umtool/report-to-video/render-cards.mjs b/umtool/report-to-video/render-cards.mjs
@@ -38,6 +38,7 @@ import { brandFaces, brandManifest, brandSvgFace, childOpts } from "./brand.mjs"
import { BRAND_CARD_STYLES, renderBrandCard } from "./brand-cards.mjs";
import { FIRA_SANS, textWidth } from "./svg-faces.mjs";
import { deckOn, resolveDeck } from "./deck.mjs";
+import { ensureWriteDir } from "../lib/report/storage.mjs";
const execFileP = promisify(execFile);
@@ -1607,6 +1608,7 @@ async function main() {
const outDir = flag("--out") ?? path.join(path.dirname(path.resolve(manifestPath)), "out");
const only = flag("--only");
+ await ensureWriteDir(outDir); // a project's out/ may be a link to the media root
await mkdir(path.join(outDir, "cards"), { recursive: true });
const cards = manifest.timeline.filter(