commit 8b3e6c1ed5583b4d5220a78873030255ec2949ff
parent b0699fcd3b9f798fde1c270ef5daa838555fe045
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 24 Sep 2026 20:03:52 -0400
Merge branch 'main' into one-core/phase-3-s3b
# Conflicts:
# editor/CHANGELOG.md
# plans/one-core-phase-3.md
Diffstat:
48 files changed, 2022 insertions(+), 261 deletions(-)
diff --git a/common/bin/migrate-channel-priority.ts b/common/bin/migrate-channel-priority.ts
@@ -1,6 +1,7 @@
#!/usr/bin/env tsx
import path from "node:path";
-import { readdir, readFile, rename, writeFile } from "node:fs/promises";
+import { readdir, readFile, writeFile } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import { getPaths } from "../lib/paths";
import { getSettings, writeSettings } from "../lib/settings";
import { siteChannelIndex } from "../lib/site";
@@ -203,9 +204,7 @@ async function clearExcludeFromSync(slug: string): Promise<boolean> {
}
if (!("excludeFromSync" in parsed)) return false;
delete parsed.excludeFromSync;
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(parsed, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(file, parsed);
return true;
}
diff --git a/common/controller/backupSavedVideos.ts b/common/controller/backupSavedVideos.ts
@@ -1,7 +1,8 @@
import path from "node:path";
import { createHash } from "node:crypto";
import { createReadStream } from "node:fs";
-import { mkdir, readFile, readdir, rename, stat, writeFile } from "node:fs/promises";
+import { mkdir, readFile, readdir, stat } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import { execa } from "execa";
import type { Paths } from "../lib/paths";
import {
@@ -115,9 +116,7 @@ export async function backupSavedVideos({
entries: manifestEntries,
};
const manifestPath = path.join(destRoot, BACKUP_MANIFEST_FILENAME);
- const tmp = `${manifestPath}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(manifest, null, 2) + "\n");
- await rename(tmp, manifestPath);
+ await writeJsonAtomic(manifestPath, manifest);
log(
`Backup complete: ${backedUp}/${saved.length} container(s), ${bytes} bytes. Manifest at ${manifestPath}`,
);
diff --git a/common/controller/compactJsonWriters.test.ts b/common/controller/compactJsonWriters.test.ts
@@ -0,0 +1,98 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import path from "node:path";
+import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import type { Paths } from "../lib/paths";
+import {
+ CUES_JSON_FILENAME,
+ LIVE_CHAT_CUES_FILENAME,
+ LIVE_CHAT_FILENAME,
+ META_FILENAME,
+ VTT_FILENAME,
+} from "../lib/videoStatus";
+import { DUPLICATES_FILENAME } from "../lib/duplicates";
+import { MEDIA_SCAN_FILENAME } from "../lib/mediaScan";
+import { normalizeTranscript } from "./normalizeTranscript";
+import { normalizeLiveChat } from "./normalizeLiveChat";
+import { detectDuplicateShorts } from "./duplicateShorts";
+import { scanCorruptMedia } from "./scanCorruptMedia";
+
+// THE FOUR COMPACT WRITERS keep their historical bytes after the fold onto
+// writeJsonAtomic (release 4 slice W): each wrote `JSON.stringify(x)` — no
+// indent, NO trailing newline — and a reader-side diff (the phase-3 numbers
+// tools, a site's published bytes) would move if that changed. Each test pins
+// that the file on disk is exactly `JSON.stringify` of its own parse.
+
+async function withDir(fn: (dir: string) => Promise<void>): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "compact-writers-"));
+ try {
+ await fn(dir);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+function assertCompact(text: string, what: string): void {
+ assert.ok(!text.endsWith("\n"), `${what}: no trailing newline`);
+ assert.equal(text, JSON.stringify(JSON.parse(text)), `${what}: compact`);
+}
+
+async function videoDir(dir: string): Promise<string> {
+ const v = path.join(dir, "vid1");
+ await mkdir(v, { recursive: true });
+ await writeFile(
+ path.join(v, META_FILENAME),
+ JSON.stringify({ id: "vid1", title: "A video", duration: 60 }),
+ );
+ return v;
+}
+
+test("normalizeTranscript writes transcript.cues.json compact, no newline", async () => {
+ await withDir(async (dir) => {
+ const v = await videoDir(dir);
+ await writeFile(
+ path.join(v, VTT_FILENAME),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nhello\n",
+ );
+ const out = await normalizeTranscript({ videoDir: v, channelSlug: "c" });
+ assert.equal(out.status, "wrote");
+ assertCompact(await readFile(path.join(v, CUES_JSON_FILENAME), "utf8"), "cues");
+ });
+});
+
+test("normalizeLiveChat writes the live-chat cues compact, no newline", async () => {
+ await withDir(async (dir) => {
+ const v = await videoDir(dir);
+ await writeFile(path.join(v, LIVE_CHAT_FILENAME), "");
+ const out = await normalizeLiveChat({ videoDir: v, channelSlug: "c" });
+ assert.equal(out.status, "wrote");
+ assertCompact(
+ await readFile(path.join(v, LIVE_CHAT_CUES_FILENAME), "utf8"),
+ "live chat cues",
+ );
+ });
+});
+
+function emptyCorpus(dir: string): Paths {
+ return {
+ transcriptsDir: dir,
+ channelsDir: path.join(dir, "channels"),
+ } as Paths;
+}
+
+test("detectDuplicateShorts writes duplicates.json compact, no newline", async () => {
+ await withDir(async (dir) => {
+ await mkdir(path.join(dir, "channels"), { recursive: true });
+ await detectDuplicateShorts({ paths: emptyCorpus(dir), onLog: () => {} });
+ assertCompact(await readFile(path.join(dir, DUPLICATES_FILENAME), "utf8"), "duplicates");
+ });
+});
+
+test("scanCorruptMedia writes media-scan.json compact, no newline", async () => {
+ await withDir(async (dir) => {
+ await mkdir(path.join(dir, "channels"), { recursive: true });
+ await scanCorruptMedia({ paths: emptyCorpus(dir), onLog: () => {} });
+ assertCompact(await readFile(path.join(dir, MEDIA_SCAN_FILENAME), "utf8"), "media scan");
+ });
+});
diff --git a/common/controller/duplicateShorts.ts b/common/controller/duplicateShorts.ts
@@ -38,7 +38,8 @@
// global transcripts/duplicates.json. Flag-only: nothing is merged or deleted.
import path from "node:path";
-import { mkdir, rename, writeFile, readFile } from "node:fs/promises";
+import { readFile } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import { createHash } from "node:crypto";
import { open } from "lmdb";
import pLimit from "p-limit";
@@ -636,10 +637,8 @@ export async function detectDuplicateShorts(
};
const outPath = path.join(opts.paths.transcriptsDir, DUPLICATES_FILENAME);
- await mkdir(path.dirname(outPath), { recursive: true });
- const tmp = `${outPath}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(report));
- await rename(tmp, outPath);
+ // Compact, no trailing newline: the report's historical bytes.
+ await writeJsonAtomic(outPath, report, { indent: 0, newline: false, mkdir: true });
noteRss();
const suspectClusters = clusters.filter((c) => c.needsReview).length;
log(
@@ -730,10 +729,7 @@ export async function updateDuplicateOverride(
}
const out = sanitizeDuplicateOverrides(current);
const file = duplicateOverridesPath(paths);
- await mkdir(path.dirname(file), { recursive: true });
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(out, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(file, out, { mkdir: true });
return out;
}
diff --git a/common/controller/failedTranscriptions.ts b/common/controller/failedTranscriptions.ts
@@ -1,5 +1,6 @@
import path from "node:path";
-import { readFile, rename, writeFile } from "node:fs/promises";
+import { readFile } from "node:fs/promises";
+import { writeFileAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
export function failedTranscriptionsFile(paths: Paths, slug: string): string {
@@ -30,9 +31,10 @@ export async function pruneFailedTranscriptions(
}
const original = raw.split("\n").filter(Boolean);
const filtered = original.filter((id) => !idsToRemove.has(id));
- const tmp = `${failureListFile}.tmp-${process.pid}`;
- await writeFile(tmp, filtered.length ? filtered.join("\n") + "\n" : "");
- await rename(tmp, failureListFile);
+ await writeFileAtomic(
+ failureListFile,
+ filtered.length ? filtered.join("\n") + "\n" : "",
+ );
return {
remaining: filtered.length,
pruned: original.length - filtered.length,
@@ -51,8 +53,6 @@ export async function clearFailedTranscriptions(
} catch {
return { cleared: 0 };
}
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, "");
- await rename(tmp, file);
+ await writeFileAtomic(file, "");
return { cleared };
}
diff --git a/common/controller/maybeMissingStore.ts b/common/controller/maybeMissingStore.ts
@@ -1,5 +1,6 @@
import path from "node:path";
-import { readFile, rename, writeFile } from "node:fs/promises";
+import { readFile } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
// Leaf module for the maybe-missing record: the set of videos we have on disk
@@ -50,10 +51,9 @@ export async function writeMaybeMissing(
slug: string,
record: MaybeMissingRecord,
): Promise<void> {
- const file = maybeMissingPath(paths, slug);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(record, null, 2) + "\n");
- await rename(tmp, file);
+ // No mkdir: the channel dir exists. Chained with the roster/scan writes of
+ // every other actor in this process (lib/jsonFile-server.ts).
+ await writeJsonAtomic(maybeMissingPath(paths, slug), record);
}
// Diff a fresh listing's canonical ids against the videos we already have on
diff --git a/common/controller/metadataScanStore.ts b/common/controller/metadataScanStore.ts
@@ -24,10 +24,11 @@
//
// Modelled on ./rosterStore.ts: versioned, normalize-on-read (unknown fields
// dropped), additive merge that returns the same object when nothing changed,
-// tmp+rename write.
+// tmp+rename write (lib/jsonFile-server.ts).
import path from "node:path";
-import { readFile, rename, writeFile } from "node:fs/promises";
+import { readFile } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
import type { ChannelConfig } from "../lib/channelConfig";
import {
@@ -185,27 +186,18 @@ export async function loadMetadataScan(
}
}
-let writeSeq = 0;
-function nextWriteSeq(): string {
- writeSeq = (writeSeq + 1) % Number.MAX_SAFE_INTEGER;
- return `${Date.now().toString(36)}-${writeSeq}`;
-}
-
async function writeMetadataScan(
paths: Paths,
slug: string,
scan: MetadataScan,
): Promise<void> {
- const file = metadataScanPath(paths, slug);
- // UNIQUE PER WRITE, not per process. A pid-named tmp file is fine only while
- // one write is in flight at a time: two overlapping writers in the SAME
- // process wrote to the same path, and the second rename found it already
- // moved and threw ENOENT, failing the job. The scan now serializes its own
- // flushes, but the download path writes here too, so the name has to be safe
- // on its own.
- const tmp = `${file}.tmp-${process.pid}-${nextWriteSeq()}`;
- await writeFile(tmp, JSON.stringify(scan, null, 2) + "\n");
- await rename(tmp, file);
+ // UNIQUE PER WRITE, not per process, and chained per path: two overlapping
+ // writers in the SAME process once shared a pid-named tmp and the second
+ // rename threw ENOENT, failing the job. The scan serializes its own flushes,
+ // but the download path writes here too. The shared writer's temp name and
+ // globalThis chain replace this module's own counter, which was per module
+ // COPY.
+ await writeJsonAtomic(metadataScanPath(paths, slug), scan);
}
export type MetadataScanUpsert = {
diff --git a/common/controller/normalizeLiveChat.ts b/common/controller/normalizeLiveChat.ts
@@ -4,7 +4,8 @@
// matches NormalizedTranscript with source: "live_chat".
import path from "node:path";
-import { readFile, rename, stat, writeFile } from "node:fs/promises";
+import { readFile, stat } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import { parseLiveChat } from "../lib/liveChat";
import type { Cue } from "../lib/vtt";
import { summarize, type RawMetadata } from "../lib/transcripts-server";
@@ -95,9 +96,8 @@ export async function normalizeLiveChat(
cues,
};
- const tmp = `${cuesPath}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(out));
- await rename(tmp, cuesPath);
+ // Compact, no trailing newline: live_chat.cues.json's historical bytes.
+ await writeJsonAtomic(cuesPath, out, { indent: 0, newline: false });
opts.log?.(
`Normalized live chat ${opts.channelSlug}/${path.basename(opts.videoDir)} (${cues.length} cues)`,
);
diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts
@@ -4,7 +4,8 @@
// shape that buildIndex would emit, plus a `source` marker.
import path from "node:path";
-import { readdir, readFile, rename, stat, writeFile } from "node:fs/promises";
+import { readdir, readFile, stat } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import { parseVtt, type Cue } from "../lib/vtt";
import {
detectTranscriptFormat,
@@ -136,9 +137,8 @@ export async function normalizeTranscript(
cues,
};
- const tmp = `${cuesPath}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(out));
- await rename(tmp, cuesPath);
+ // Compact, no trailing newline: transcript.cues.json's historical bytes.
+ await writeJsonAtomic(cuesPath, out, { indent: 0, newline: false });
opts.log?.(
`Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${transcriptFormat}, ${cues.length} cues)`,
);
diff --git a/common/controller/relocateDir.ts b/common/controller/relocateDir.ts
@@ -4,11 +4,10 @@ import {
readdir,
readFile,
readlink,
- rename,
rm,
stat,
- writeFile,
} from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import { execa } from "execa";
import {
formatRsyncProgressDetail,
@@ -136,9 +135,7 @@ export async function writeDirMarker(
file: string,
marker: DirRelocationMarker,
): Promise<void> {
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(marker, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(file, marker);
}
export async function clearDirMarker(file: string): Promise<void> {
diff --git a/common/controller/rosterStore.ts b/common/controller/rosterStore.ts
@@ -1,6 +1,7 @@
import path from "node:path";
-import { mkdir, open, readdir, readFile, rename, writeFile } from "node:fs/promises";
+import { open, readdir, readFile } from "node:fs/promises";
import pLimit from "p-limit";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
import { extractVideoId } from "../lib/videoId";
@@ -224,17 +225,15 @@ export async function loadRoster(paths: Paths, slug: string): Promise<Roster> {
// Atomic tmp+rename, like maybe-missing.json: a crashed write must never leave a
// truncated roster behind, because a truncated roster is exactly the data loss
-// this file exists to prevent.
+// this file exists to prevent. Several actors write it (the sync, the quick
+// availability check, the pipeline's server action — a different bundle), so
+// it goes through the one per-path chain on globalThis.
export async function writeRoster(
paths: Paths,
slug: string,
roster: Roster,
): Promise<void> {
- const file = rosterPath(paths, slug);
- await mkdir(path.dirname(file), { recursive: true });
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(roster, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(rosterPath(paths, slug), roster, { mkdir: true });
}
// Load, merge, write in one step — the shape every additive writer wants. Skips
diff --git a/common/controller/scanCorruptMedia.ts b/common/controller/scanCorruptMedia.ts
@@ -29,7 +29,8 @@
// first, delete second — not a dry-run flag on a deleting command.
import path from "node:path";
-import { mkdir, readdir, readFile, rename, stat, writeFile } from "node:fs/promises";
+import { readdir, readFile, stat } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import { execa } from "execa";
import type { Paths } from "../lib/paths";
import {
@@ -389,10 +390,8 @@ async function mergeAndWriteReport(
),
};
const outPath = path.join(paths.transcriptsDir, MEDIA_SCAN_FILENAME);
- await mkdir(path.dirname(outPath), { recursive: true });
- const tmp = `${outPath}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(merged));
- await rename(tmp, outPath);
+ // Compact, no trailing newline: the report's historical bytes.
+ await writeJsonAtomic(outPath, merged, { indent: 0, newline: false, mkdir: true });
return merged;
}
@@ -450,9 +449,6 @@ export async function updateMediaScanOverride(
reviewed,
};
const file = mediaScanOverridesPath(paths);
- await mkdir(path.dirname(file), { recursive: true });
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(out, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(file, out, { mkdir: true });
return out;
}
diff --git a/common/controller/shard.ts b/common/controller/shard.ts
@@ -1,5 +1,6 @@
import path from "node:path";
-import { readFile, rename, unlink, writeFile } from "node:fs/promises";
+import { readFile, unlink } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
export const SHARD_OPS = [
@@ -52,10 +53,7 @@ export async function saveShardConfig(
op: ShardOp,
cfg: ShardConfig,
): Promise<void> {
- const file = shardFile(paths, slug, op);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(cfg, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(shardFile(paths, slug, op), cfg);
}
export async function clearShardConfig(
diff --git a/common/controller/storageWatch.ts b/common/controller/storageWatch.ts
@@ -318,6 +318,10 @@ export async function runStorageWatchPass(
// of the process.
export const STORAGE_WATCH_INTERVAL_MS = 5 * 60_000;
+// A per-module-copy singleton, deliberately left so: it is not a temp-file
+// name (slice W folded every tmp + rename onto lib/jsonFile-server.ts, whose
+// state is on globalThis), and its one caller is editor/instrumentation.ts,
+// so only one copy ever arms it.
let timer: ReturnType<typeof setInterval> | null = null;
// ARMED ONCE PER PROCESS. `unref()` so it never holds the event loop open — a
diff --git a/common/jobs/autoQueueState.test.ts b/common/jobs/autoQueueState.test.ts
@@ -1,6 +1,6 @@
import { test } from "node:test";
import assert from "node:assert/strict";
-import { mkdtemp, rm, writeFile } from "node:fs/promises";
+import { mkdtemp, readdir, rm, writeFile } from "node:fs/promises";
import { tmpdir } from "node:os";
import path from "node:path";
import type { Paths } from "../lib/paths";
@@ -126,3 +126,29 @@ test("a lane missing from an in-memory state object still writes", async () => {
]);
});
});
+
+// --- write safety ------------------------------------------------------------
+
+test("overlapping writes do not collide on the tmp file", async () => {
+ await withPaths(async (paths) => {
+ // persist() is fire-and-forget and two lanes' runners can persist within
+ // milliseconds of each other. With one pid-named tmp the second write
+ // truncated the first's temp and one rename found it gone (ENOENT).
+ await Promise.all(
+ Array.from({ length: 12 }, (_, i) => {
+ const state = emptyAutoQueueState();
+ state.download.platformBackoff = { youtube: { until: i, fails: i } };
+ return writeAutoQueueState(paths, state);
+ }),
+ );
+ // Every write landed and nothing threw; the last ISSUED is on disk (the
+ // writes are chained per path) and no temp is left behind.
+ const back = await readAutoQueueState(paths);
+ assert.deepEqual(back.download.platformBackoff, {
+ youtube: { until: 11, fails: 11 },
+ });
+ assert.deepEqual(await readdir(path.dirname(paths.autoQueueStateFile)), [
+ "state.json",
+ ]);
+ });
+});
diff --git a/common/jobs/autoQueueState.ts b/common/jobs/autoQueueState.ts
@@ -1,5 +1,5 @@
-import { mkdir, readFile, rename, writeFile } from "node:fs/promises";
-import path from "node:path";
+import { readFile } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
import {
type AutoQueueRuntime,
@@ -136,9 +136,9 @@ export async function readAutoQueueState(paths: Paths): Promise<AutoQueueState>
// tmp name per process means the second write truncates the file the first is
// still writing and then renames it over the real one — leaving a state.json
// that reads as empty (tolerated) and loses a persisted platform cooldown
-// across a restart (not). One counter is enough: this is one process, and the
-// pid still separates two of them.
-let writeSeq = 0;
+// across a restart (not). `writeJsonAtomic` gives every write its own temp
+// name and chains writes to one path on globalThis — which this module's own
+// counter, one per module COPY, did not.
export async function writeAutoQueueState(
paths: Paths,
@@ -152,10 +152,7 @@ export async function writeAutoQueueState(
const out = Object.fromEntries(
LANES.map((lane) => [lane, trim(state[lane] ?? emptyAutoQueueKindState())]),
) as AutoQueueState;
- await mkdir(path.dirname(paths.autoQueueStateFile), { recursive: true });
- const tmp = `${paths.autoQueueStateFile}.tmp-${process.pid}-${++writeSeq}`;
- await writeFile(tmp, JSON.stringify(out, null, 2) + "\n");
- await rename(tmp, paths.autoQueueStateFile);
+ await writeJsonAtomic(paths.autoQueueStateFile, out, { mkdir: true });
}
export function recordPick(state: AutoQueueKindState, pick: AutoQueuePick): void {
diff --git a/common/jobs/syncSchedulerState.ts b/common/jobs/syncSchedulerState.ts
@@ -1,5 +1,5 @@
-import { mkdir, readFile, rename, writeFile } from "node:fs/promises";
-import path from "node:path";
+import { readFile } from "node:fs/promises";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
// Persistent state for the cron-driven sync scheduler. Unlike the in-memory job
@@ -118,10 +118,7 @@ export async function writeSchedulerState(
lastSavedVideoBackupAt: state.lastSavedVideoBackupAt ?? null,
lastSyncAllAt: state.lastSyncAllAt ?? null,
};
- await mkdir(path.dirname(paths.schedulerStateFile), { recursive: true });
- const tmp = `${paths.schedulerStateFile}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(out, null, 2) + "\n");
- await rename(tmp, paths.schedulerStateFile);
+ await writeJsonAtomic(paths.schedulerStateFile, out, { mkdir: true });
}
// Get (or lazily create) the mutable state entry for a channel.
diff --git a/common/jobs/workerDefaults.ts b/common/jobs/workerDefaults.ts
@@ -1,5 +1,5 @@
import fs from "node:fs";
-import path from "node:path";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
// Persisted "default" worker arrangement. Unlike the in-memory worker pool
@@ -55,10 +55,5 @@ export async function writeWorkerDefaults(
if (typeof id === "string" && id) seen.add(id);
}
const out: WorkerDefaults = { enabledWorkerIds: Array.from(seen) };
- await fs.promises.mkdir(path.dirname(paths.workerDefaultsFile), {
- recursive: true,
- });
- const tmp = `${paths.workerDefaultsFile}.tmp-${process.pid}`;
- await fs.promises.writeFile(tmp, JSON.stringify(out, null, 2) + "\n");
- await fs.promises.rename(tmp, paths.workerDefaultsFile);
+ await writeJsonAtomic(paths.workerDefaultsFile, out, { mkdir: true });
}
diff --git a/common/lib/homepage.ts b/common/lib/homepage.ts
@@ -1,4 +1,5 @@
import fs from "node:fs";
+import { writeJsonAtomic } from "./jsonFile-server";
import { getPaths, type Paths } from "./paths";
import { PROJECT_NAME, PROJECT_TAGLINE } from "./project";
import {
@@ -124,9 +125,6 @@ export async function writeHomepageConfig(
? { cloudflareProject: config.cloudflareProject.trim() }
: {}),
};
- await fs.promises.mkdir(paths.homepageDir, { recursive: true });
- const file = paths.homepageConfigFile;
- const tmp = `${file}.tmp-${process.pid}`;
- await fs.promises.writeFile(tmp, JSON.stringify(merged, null, 2) + "\n");
- await fs.promises.rename(tmp, file);
+ // mkdir: the parent of homepageConfigFile IS homepageDir (lib/paths.ts).
+ await writeJsonAtomic(paths.homepageConfigFile, merged, { mkdir: true });
}
diff --git a/common/lib/jsonFile-server.test.ts b/common/lib/jsonFile-server.test.ts
@@ -2,9 +2,11 @@ import { test } from "node:test";
import assert from "node:assert/strict";
import fs from "node:fs";
import { mkdtemp, readFile, readdir, writeFile } from "node:fs/promises";
+import { createRequire, syncBuiltinESMExports } from "node:module";
import os from "node:os";
import path from "node:path";
import {
+ copyFileAtomic,
jsonFileState,
jsonText,
pendingJsonWrites,
@@ -12,9 +14,12 @@ import {
readJsonFileSync,
tmpPathFor,
writeJsonAtomic,
+ writeFileAtomic,
writeJsonAtomicSync,
} from "./jsonFile-server";
+const require = createRequire(import.meta.url);
+
async function scratch(): Promise<string> {
return mkdtemp(path.join(os.tmpdir(), "jsonfile-"));
}
@@ -123,3 +128,79 @@ test("writeJsonAtomicSync: same bytes, mkdir, no temp left", async () => {
assert.equal(fs.readFileSync(file, "utf8"), '{\n "k": 1\n}');
assert.deepEqual(await readdir(path.dirname(file)), ["b.json"]);
});
+
+test("writeFileAtomic: exact bytes for a string and a Buffer, mkdir, no temp left", async () => {
+ const dir = await scratch();
+ const txt = path.join(dir, "d", "playlist");
+ await writeFileAtomic(txt, "a\nb\n", { mkdir: true });
+ assert.equal(await readFile(txt, "utf8"), "a\nb\n");
+ const bin = path.join(dir, "d", "blob");
+ const bytes = Buffer.from([0, 255, 10, 13, 128]);
+ await writeFileAtomic(bin, bytes);
+ assert.deepEqual(await readFile(bin), bytes);
+ // An empty string is a real write (clearFailedTranscriptions relies on it).
+ await writeFileAtomic(txt, "");
+ assert.equal(await readFile(txt, "utf8"), "");
+ assert.deepEqual((await readdir(path.join(dir, "d"))).sort(), ["blob", "playlist"]);
+});
+
+test("writeFileAtomic: mode is on the file from creation (0o600 cookie jar)", async () => {
+ const dir = await scratch();
+ const file = path.join(dir, "cookies.txt");
+ await writeFileAtomic(file, "secret\n", { mode: 0o600 });
+ assert.equal(fs.statSync(file).mode & 0o777, 0o600);
+ // The temp itself carries the mode BEFORE the rename: intercept the rename
+ // (the module calls fs/promises' `rename`, so patch it and sync the builtin
+ // ESM exports) and stat its source — the temp — at that moment.
+ const fsp = require("node:fs/promises") as typeof import("node:fs/promises");
+ const realRename = fsp.rename;
+ const seen: number[] = [];
+ fsp.rename = (async (from: fs.PathLike, to: fs.PathLike) => {
+ seen.push(fs.statSync(from).mode & 0o777);
+ return realRename(from, to);
+ }) as typeof fsp.rename;
+ syncBuiltinESMExports();
+ try {
+ await writeFileAtomic(file, "second\n", { mode: 0o600 });
+ } finally {
+ fsp.rename = realRename;
+ syncBuiltinESMExports();
+ }
+ assert.deepEqual(seen, [0o600], "the rename saw exactly one temp, already 0o600");
+ assert.equal(await readFile(file, "utf8"), "second\n");
+});
+
+test("writeFileAtomic and writeJsonAtomic share one chain on one path", async () => {
+ const dir = await scratch();
+ const file = path.join(dir, "shared.json");
+ const writes: Promise<void>[] = [];
+ for (let i = 0; i < 40; i++) {
+ writes.push(
+ i % 2 === 0
+ ? writeJsonAtomic(file, { i })
+ : writeFileAtomic(file, `{"i":${i}}`),
+ );
+ }
+ assert.equal(pendingJsonWrites(), 1, "one chain entry for the path, not two");
+ await Promise.all(writes);
+ assert.deepEqual(JSON.parse(await readFile(file, "utf8")), { i: 39 });
+ assert.deepEqual(await readdir(dir), ["shared.json"]);
+});
+
+test("copyFileAtomic: dest gets src's bytes, src stays, no temp left", async () => {
+ const dir = await scratch();
+ const src = path.join(dir, "src.bin");
+ await writeFile(src, Buffer.from([1, 2, 3]));
+ const dest = path.join(dir, "out", "dest.bin");
+ await copyFileAtomic(src, dest, { mkdir: true });
+ assert.deepEqual(await readFile(dest), Buffer.from([1, 2, 3]));
+ assert.deepEqual(await readdir(path.join(dir, "out")), ["dest.bin"]);
+ await assert.rejects(copyFileAtomic(path.join(dir, "nope"), dest));
+ assert.deepEqual(await readdir(path.join(dir, "out")), ["dest.bin"]);
+});
+
+test("the remark-empty transcript is the literal it replaced (videoActions)", () => {
+ // editor/app/channels/[slug]/videos/[id]/videoActions.ts wrote this literal
+ // by hand before slice W; it now writes the object with { indent: 0 }.
+ assert.equal(jsonText({ transcription: [] }, { indent: 0 }), '{"transcription":[]}\n');
+});
diff --git a/common/lib/jsonFile-server.ts b/common/lib/jsonFile-server.ts
@@ -1,6 +1,8 @@
-// ONE JSON READER AND ONE ATOMIC JSON WRITER for the config files and sidecars.
-// (Not yet every JSON writer: the slice 4b record lists the ones still on the
-// per-pid temp name.)
+// ONE JSON READER AND ONE ATOMIC WRITER — for the config files and sidecars,
+// and (since release 4 slice W) for every tmp + rename write in common/ and
+// editor/: JSON through `writeJsonAtomic`, text and binary through
+// `writeFileAtomic`, copies through `copyFileAtomic`. What is left on its own
+// temp name is listed in plans/one-core-phase-3.md, "Slice W, as shipped".
//
// one-core phase 3 slice 4b. Before this module the repo had seven private
// copies of `writeJsonAtomic` and some twenty inline `tmp + rename` writes, all
@@ -43,7 +45,7 @@
import { randomBytes } from "node:crypto";
import fs from "node:fs";
-import { mkdir, readFile, rename, rm, writeFile } from "node:fs/promises";
+import { copyFile, mkdir, readFile, rename, rm, writeFile } from "node:fs/promises";
import path from "node:path";
export type ReadJsonResult =
@@ -134,15 +136,42 @@ export function tmpPathFor(file: string): string {
return `${file}.tmp-${process.pid}-${seq}-${randomBytes(4).toString("hex")}`;
}
-async function writeNow(
+export type WriteFileOptions = {
+ // Create the parent directory first (`mkdir -p`).
+ mkdir?: boolean;
+ // Permission bits for the NEW file, applied as the temp file is CREATED
+ // (`writeFile(tmp, data, { mode })`, under the umask), so the bytes are never
+ // readable more widely than `mode` — not even for the instant before the
+ // rename. The X cookie jar writes 0o600 through this.
+ mode?: number;
+};
+
+// Run `op` for `file` after every earlier chained operation on the same
+// absolute path this process issued has settled. A failed earlier op does not
+// block a later one; each caller sees only its own op's error.
+function chained(key: string, op: () => Promise<void>): Promise<void> {
+ const { chains } = jsonFileState();
+ const prev = chains.get(key) ?? Promise.resolve();
+ const next = prev.then(op, op);
+ chains.set(key, next);
+ const release = () => {
+ if (chains.get(key) === next) chains.delete(key);
+ };
+ next.then(release, release);
+ return next;
+}
+
+// Put the temp file in place with `fill`, then rename it over `file`; on any
+// failure remove the temp and rethrow.
+async function replaceNow(
file: string,
- text: string,
makeDir: boolean,
+ fill: (tmp: string) => Promise<void>,
): Promise<void> {
if (makeDir) await mkdir(path.dirname(file), { recursive: true });
const tmp = tmpPathFor(file);
try {
- await writeFile(tmp, text);
+ await fill(tmp);
await rename(tmp, file);
} catch (err) {
await rm(tmp, { force: true }).catch(() => {});
@@ -150,30 +179,47 @@ async function writeNow(
}
}
-// Write `value` as JSON to `file` atomically (tmp + rename), after every write
-// to the same absolute path this process issued earlier has settled. The value
-// is serialised NOW, at the call, so a caller that goes on mutating its object
-// cannot change what lands. A failed earlier write does not block a later one;
-// each caller sees only its own write's error.
+// THE ONE ATOMIC WRITE (tmp + rename) for text and binary files: the playlist,
+// failed-transcriptions, the cookie jar, a CHANGELOG, a VTT — and every JSON
+// file, through `writeJsonAtomic` below. Same chain, same unique temp name.
+// `data` is not copied: a Buffer must not change before the promise settles.
+export function writeFileAtomic(
+ file: string,
+ data: string | Buffer,
+ opts: WriteFileOptions = {},
+): Promise<void> {
+ const key = path.resolve(file);
+ const { mode } = opts;
+ return chained(key, () =>
+ replaceNow(key, opts.mkdir === true, (tmp) =>
+ mode === undefined ? writeFile(tmp, data) : writeFile(tmp, data, { mode }),
+ ),
+ );
+}
+
+// Copy `src` to `dest` atomically: copy into a temp beside `dest`, then rename,
+// so a crash mid-copy never leaves a partial file under the final name. On the
+// same chain as the writers above (keyed on `dest`).
+export function copyFileAtomic(
+ src: string,
+ dest: string,
+ opts: { mkdir?: boolean } = {},
+): Promise<void> {
+ const key = path.resolve(dest);
+ return chained(key, () =>
+ replaceNow(key, opts.mkdir === true, (tmp) => copyFile(src, tmp)),
+ );
+}
+
+// Write `value` as JSON to `file` atomically: `jsonText` → `writeFileAtomic`.
+// The value is serialised NOW, at the call, so a caller that goes on mutating
+// its object cannot change what lands.
export function writeJsonAtomic(
file: string,
value: unknown,
opts: WriteJsonOptions = {},
): Promise<void> {
- const key = path.resolve(file);
- const text = jsonText(value, opts);
- const { chains } = jsonFileState();
- const prev = chains.get(key) ?? Promise.resolve();
- const next = prev.then(
- () => writeNow(key, text, opts.mkdir === true),
- () => writeNow(key, text, opts.mkdir === true),
- );
- chains.set(key, next);
- const release = () => {
- if (chains.get(key) === next) chains.delete(key);
- };
- next.then(release, release);
- return next;
+ return writeFileAtomic(file, jsonText(value, opts), { mkdir: opts.mkdir });
}
// The synchronous twin, for the three stores whose API is synchronous
@@ -215,7 +261,8 @@ export function withJsonFileLock<T>(file: string, fn: () => Promise<T>): Promise
return next;
}
-// For tests: how many paths currently have a write in flight.
+// For tests: how many paths currently have a write (of any of the three
+// kinds) in flight.
export function pendingJsonWrites(): number {
return jsonFileState().chains.size;
}
diff --git a/common/lib/savedVideo-server.ts b/common/lib/savedVideo-server.ts
@@ -1,7 +1,6 @@
-import { writeJsonAtomic } from "./jsonFile-server";
+import { copyFileAtomic, writeJsonAtomic } from "./jsonFile-server";
import path from "node:path";
import {
- copyFile,
mkdir,
readFile,
rename,
@@ -76,9 +75,7 @@ async function moveFileCrossDevice(src: string, dest: string): Promise<void> {
} catch (err) {
if ((err as NodeJS.ErrnoException).code !== "EXDEV") throw err;
}
- const tmp = `${dest}.tmp-${process.pid}`;
- await copyFile(src, tmp);
- await rename(tmp, dest);
+ await copyFileAtomic(src, dest);
await rm(src, { force: true });
}
diff --git a/common/lib/widgetPresets.ts b/common/lib/widgetPresets.ts
@@ -1,5 +1,5 @@
import fs from "node:fs";
-import path from "node:path";
+import { writeJsonAtomic } from "./jsonFile-server";
import type { Paths } from "./paths";
// Persisted monitor-widget presets: named arrangements the /widget/builder board
@@ -77,12 +77,7 @@ async function writePresets(
presets: WidgetPreset[],
): Promise<void> {
const out: PresetsFile = { v: 1, presets };
- await fs.promises.mkdir(path.dirname(paths.widgetPresetsFile), {
- recursive: true,
- });
- const tmp = `${paths.widgetPresetsFile}.tmp-${process.pid}`;
- await fs.promises.writeFile(tmp, JSON.stringify(out, null, 2) + "\n");
- await fs.promises.rename(tmp, paths.widgetPresetsFile);
+ await writeJsonAtomic(paths.widgetPresetsFile, out, { mkdir: true });
}
// Save a preset under `name`. Saving over an existing name OVERWRITES its query
diff --git a/common/social/xSessionBroker.ts b/common/social/xSessionBroker.ts
@@ -20,7 +20,8 @@
// build-time one.
import path from "node:path";
-import { mkdir, readFile, rename, writeFile, rm, stat } from "node:fs/promises";
+import { mkdir, readFile, rm, stat } from "node:fs/promises";
+import { writeFileAtomic } from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
import { importPlaywright } from "./playwrightRuntime";
@@ -212,11 +213,12 @@ async function writeCookieJar(
log: (line: string) => void,
): Promise<void> {
const file = xCookieFile(paths);
- await mkdir(path.dirname(file), { recursive: true });
- const body = toNetscapeCookieFile(cookies);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, body, { mode: 0o600 });
- await rename(tmp, file);
+ // mode is applied as the temp is CREATED, before the rename: the jar is
+ // never world-readable, not even for an instant.
+ await writeFileAtomic(file, toNetscapeCookieFile(cookies), {
+ mkdir: true,
+ mode: 0o600,
+ });
const authed = cookies.some((c) => c.name === AUTH_COOKIE);
log(
`Exported ${cookies.length} X cookie(s) to ${file}` +
diff --git a/common/views/channelGroupSections.test.ts b/common/views/channelGroupSections.test.ts
@@ -13,11 +13,14 @@ import { normalizeBuckets } from "./pipeline/stageStatus";
import {
buildChannelGroupSections,
slugsInGroup,
+ stationWorkFor,
+ transcribeStationIds,
type ChannelGroupSection,
} from "./channelGroupSections";
-// Run from this directory:
-// cd editor/app/channels/lib && ../../../../node_modules/.bin/tsx --test channelGroupSections.test.ts
+// Run with the rest of the common suite (`pnpm --filter yt-dlp-transcript-common
+// test`), or alone from common/:
+// pnpm exec tsx --test views/channelGroupSections.test.ts
// Snapshot shape as WRITTEN TO DISK — same trick as channelFlow.test.ts.
function snapshotOf(patch: Partial<ChannelSnapshot> = {}): ChannelSnapshot {
@@ -224,7 +227,7 @@ test("download excludes members-only/deleted/private ids", () => {
assert.equal(sections[0].download.total, 1);
});
-test("a youtube-handling channel is not transcribe-eligible but is download-eligible", () => {
+test("a youtube-handling channel IS transcribe-eligible: missing auto-subs is whisper work", () => {
const sections = build(
siteOf({ channels: [{ slug: "yt" }, { slug: "whisper" }] }),
[
@@ -241,10 +244,87 @@ test("a youtube-handling channel is not transcribe-eligible but is download-elig
const { download, transcribe } = sections[0];
assert.deepEqual(download.eligible.sort(), ["whisper", "yt"]);
assert.equal(download.total, 2);
- assert.deepEqual(transcribe.eligible, ["whisper"]);
- // The youtube channel's two awaiting-whisper videos are NOT in the figure —
- // whisper will never run on them.
- assert.equal(transcribe.total, 1);
+ // Handling does not decide it — the files do. The youtube channel's two
+ // downloaded videos with no transcript at all are what its button queues.
+ assert.deepEqual(transcribe.eligible.sort(), ["whisper", "yt"]);
+ assert.equal(transcribe.total, 3);
+});
+
+test("the transcribe figure counts downloaded auto-caption-only videos too", () => {
+ const sections = build(
+ siteOf({ channels: [{ slug: "yt" }] }),
+ [
+ channel("yt", { handling: "youtube" }, snapshotOf({
+ buckets: normalizeBuckets({ downloadedAutoSubsOnly: ["a"] }),
+ })),
+ ],
+ );
+ assert.deepEqual(sections[0].transcribe.eligible, ["yt"]);
+ assert.equal(sections[0].transcribe.total, 1);
+});
+
+test("a youtube channel with nothing to transcribe is eligible at 0 (skipped at click time)", () => {
+ const sections = build(
+ siteOf({ channels: [{ slug: "yt" }] }),
+ [channel("yt", { handling: "youtube" })],
+ );
+ assert.deepEqual(sections[0].transcribe.eligible, ["yt"]);
+ assert.equal(sections[0].transcribe.total, 0);
+ assert.deepEqual(sections[0].transcribe.unknown, []);
+});
+
+test("transcribe drops excluded ids from BOTH buckets", () => {
+ const snapshot = snapshotOf({
+ buckets: normalizeBuckets({
+ downloadedNoTranscript: ["keep-1", "gone-1"],
+ downloadedAutoSubsOnly: ["keep-2", "gone-2"],
+ }),
+ excludedFromDownload: {
+ membersOnly: ["gone-1"],
+ deleted: [],
+ private: ["gone-2"],
+ },
+ });
+ const sections = build(
+ siteOf({ channels: [{ slug: "yt" }] }),
+ [channel("yt", { handling: "youtube" }, snapshot)],
+ );
+ assert.equal(sections[0].transcribe.total, 2);
+ assert.deepEqual(transcribeStationIds(snapshot), {
+ missing: ["keep-1"],
+ autoSubs: ["keep-2"],
+ });
+});
+
+test("an excluded download is neither counted nor named for the batch (the review's walk)", () => {
+ // x, y downloaded with no transcript; a downloaded with auto-captions only;
+ // y went private after it was downloaded. The figure is 2 and the two id
+ // lists the station queues BY ID are exactly [x] and [a] — never y.
+ const snapshot = snapshotOf({
+ buckets: normalizeBuckets({
+ downloadedNoTranscript: ["x", "y"],
+ downloadedAutoSubsOnly: ["a"],
+ }),
+ excludedFromDownload: { membersOnly: [], deleted: [], private: ["y"] },
+ });
+ const sections = build(
+ siteOf({ channels: [{ slug: "yt" }] }),
+ [channel("yt", { handling: "youtube" }, snapshot)],
+ );
+ assert.equal(sections[0].transcribe.total, 2);
+ assert.deepEqual(transcribeStationIds(snapshot), {
+ missing: ["x"],
+ autoSubs: ["a"],
+ });
+});
+
+test("a social channel is not transcribe-eligible, and says why", () => {
+ const { brief } = channel("posts", { sourceKind: "social", handling: "transcribe" });
+ assert.deepEqual(stationWorkFor("transcribe", brief, LANE_ON), {
+ eligible: false,
+ work: 0,
+ reason: "social account",
+ });
});
test("a social channel is eligible for sync only", () => {
diff --git a/common/views/channelGroupSections.ts b/common/views/channelGroupSections.ts
@@ -105,6 +105,23 @@ export function laneOffFor(
return false;
}
+// The two id lists the transcribe station counts and queues, after the same
+// download exclusions every other station honours. Disjoint by construction:
+// an ASR VTT makes a video "transcribed" for `downloadedNoTranscript`, so a
+// video is in at most one of them. They are two lists because they are two
+// batches — `runWhisperBatch`'s default scan drains the first, its
+// replace-auto-captions mode needs an ASR track per id for the second.
+export function transcribeStationIds(
+ snapshot: NonNullable<ChannelBrief["snapshot"]>,
+): { missing: string[]; autoSubs: string[] } {
+ const excluded = excludedDownloadIdSet(snapshot);
+ const buckets = normalizeBuckets(snapshot.buckets);
+ return {
+ missing: buckets.downloadedNoTranscript.filter((id) => !excluded.has(id)),
+ autoSubs: buckets.downloadedAutoSubsOnly.filter((id) => !excluded.has(id)),
+ };
+}
+
export type StationChannelWork = {
// Whether the operation applies to this channel at all.
eligible: boolean;
@@ -168,19 +185,17 @@ export function stationWorkFor(
}
if (station === "transcribe") {
- // A `youtube`-handling channel never runs whisper, so counting it would
- // inflate the figure on a button that would skip it anyway.
- if (config.handling !== "transcribe") {
- return { eligible: false, work: 0, reason: "not set to transcribe" };
- }
+ // HANDLING DOES NOT DECIDE THIS. Buckets are decided by FILES
+ // (channelSnapshot.ts), and a `youtube`-handling channel whose video came
+ // down with no captions at all is in `downloadedNoTranscript` exactly like a
+ // `transcribe` one — the runner drains that bucket for every channel. What
+ // the station counts is what its button queues: those, plus the videos whose
+ // only transcript is YouTube's auto-captions and whose audio is on disk
+ // (`downloadedAutoSubsOnly`, the channel page's replace-auto-captions half).
+ // `transcribeStationIds` is the one fold; the group action runs off it too.
if (!snapshot) return { eligible: true, work: null };
- const excluded = excludedDownloadIdSet(snapshot);
- return {
- eligible: true,
- work: normalizeBuckets(snapshot.buckets).downloadedNoTranscript.filter(
- (id) => !excluded.has(id),
- ).length,
- };
+ const { missing, autoSubs } = transcribeStationIds(snapshot);
+ return { eligible: true, work: missing.length + autoSubs.length };
}
if (station === "digest") {
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -1,5 +1,6 @@
import path from "node:path";
-import { mkdir, readdir, readFile, rename, writeFile } from "node:fs/promises";
+import { mkdir, readdir, readFile } from "node:fs/promises";
+import { writeFileAtomic } from "../lib/jsonFile-server";
import { execa } from "execa";
import pLimit from "p-limit";
import { readArchive } from "../lib/archive";
@@ -457,9 +458,7 @@ async function writePlaylistFile(
onLog: (s: string) => void,
): Promise<void> {
const playlistPath = path.join(root, "playlist");
- const tmpPath = `${playlistPath}.tmp-${process.pid}`;
- await writeFile(tmpPath, urls.join("\n") + (urls.length ? "\n" : ""));
- await rename(tmpPath, playlistPath);
+ await writeFileAtomic(playlistPath, urls.join("\n") + (urls.length ? "\n" : ""));
onLog(`Wrote ${urls.length} URLs to ${playlistPath}\n`);
}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,9 @@
# Changelog
## [Unreleased]
+- **Channel rows no longer scroll over a group's controls on `/channels`.** Scrolled down and to the right, the pinned Slug column of every row painted over the pinned group header and its five station buttons (Sync, Download, Transcribe, Digest and the speaker lane), and took the clicks. The pinned Slug cell and the group header sat at the same stacking level, and the later rows won. The rack now has one named layer order, kept in one file: the Advanced panel, then the column header, then the group header, then the pinned checkbox and Slug cells. Nothing ties any more. The screenshot audit found four more problems, fixed as well. A group header's name and buttons now stay on screen however far the columns scroll across (they used to scroll off to the left). An Advanced panel opened near the bottom or the right edge scrolls itself into view instead of being cut off. The rule above a pinned group header moves with it instead of leaving a gap the rows showed through. On a phone, the column header no longer paints over the selection bar pinned to the bottom of the screen.
+- **A group's Transcribe works for YouTube channels, and it counts what it queues.** The station used to be disabled for every `youtube`-handling channel with the message "a youtube-handling channel never runs whisper". That was wrong. A YouTube video that came down with no captions is transcription work like any other, and the automatic runner already treats it that way. Transcribe now counts two kinds of video, after the usual members-only, deleted and private exclusions: downloaded videos with no transcript at all, and downloaded videos whose only transcript is YouTube's auto-captions. Pressing it queues exactly those videos, by id, as the channel page does: up to two jobs per channel on the transcription queue. A video downloaded before it went private, members-only or deleted is no longer transcribed by the group button, because it was never in the figure. Pressing it again while either job runs says *already running*. The wording names no method ("…has downloaded audio to transcribe", "…each takes minutes"). **This figure can now be higher than the Transcription band in the same rack on channels with many auto-caption-only videos.** The band counts videos with no transcript at all, while the station counts everything its button would queue. That is intended.
+- **The editor's atomic JSON, text and binary writes now go one way, and a failed write no longer leaves a temp file behind.** Nineteen JSON write sites and seven text and binary ones each wrote `<file>.tmp-<pid>` and renamed it over the original — the channel roster, maybe-missing and metadata-scan records, the scheduler and auto-queue state, worker defaults, widget presets, the homepage config, relocation markers, shard configs, the duplicate and media-scan reports and their review decisions, the saved-video backup manifest, both cue normalizers, the playlist, the failed-transcriptions list, the X cookie jar, a site's CHANGELOG cut, a saved video copied into its store across drives, and the video page's VTT promote and remark. They now all go through one writer (`common/lib/jsonFile-server.ts`), which gives every write its own temp name and queues writes to the same file one behind another, so two jobs touching one channel's roster at once cannot trip over each other's temp file. A write that fails now removes its temp: the live `.auto-queue/` holds 175 `state.json.tmp-…` files (173 of them empty) from the day `/home` filled up (2026-09-11), each one a failed write the old code left behind; nothing deletes those old ones for you — `find transcripts -name '*.tmp-*'` lists them. Four temp names stay, on purpose: the export build's two page writers `buildIndex.ts` (a streaming page writer) and `buildStats.ts` (a hand-joined array) — folding them is a restructuring, not a swap — and `transcode.ts` / `transcribeOne.ts` name the output file ffmpeg or the transcription app writes, which is not our write to fold. No file's contents change — every writer puts the same bytes on disk it did before, measured over the live corpus. The cookie jar is still created readable only by you.
- **The video page's chore cards are one file each.** Redownload, Transcode, Transcribe, Transcript source, Mark untranscribable, Source video, Fetched windows, Archive media, Truncated check, Files, the Danger zone, the availability history, the download-outcome badge and the two truncation banners each moved out of one 1,839-line `VideoPanel.tsx` into their own module under `videos/[id]/components/cards/`; `VideoPanel` is now just their order and when each shows, and `lib/videoChoreCards.ts` lists them — with why each is a chore and not an operation — beside the registry's operation panels. Nothing on the page changed: same cards, same order, same labels.
- **Every channel table and every job-in-flight line is now drawn one way.** The /channels rack, the dashboard's Channels table and the work tables on the operation pages and /cleanup are one table with a column set per page, over one channel row built on the server (which no longer ships a channel's config to the browser); the dashboard's "Needs work" seed is computed by the same code the widget endpoint serves. On the jobs side, /jobs rows, the "Active jobs" cards on channel/video/operation pages, the monitor widget's Active jobs strip and the operations board's "In flight" list are one job row in three sizes, with one rule for which buttons (Retry / Reorder / Drain / Cancel / Force-release) a job gets. **What you might notice:** a work table's report column reads "stale"/"missing" like the rack's instead of a date; the dashboard's Sync button is the rack's; a lane line on /jobs offers Force-release while its runner is running; widget job lines show who asked for the job; an in-flight download on the operations board links to its job page. Nothing a count says moved.
- **`site.json`, each channel's `config.json` and the per-video sidecars now have one schema each, and the two config files have generated key tables.** **`SITE.md`** and **`CHANNEL.md`** (new, repo root) list every key with its default and meaning, generated by `common/bin/file-schemas-docs.ts` and checked by a test. Nothing an operator has configured reads or saves differently: every live `site.json` and `config.json`, and a 1,763-file sample of sidecars, read and write back byte-for-byte as before. **Fixed:** a social-channel fetch no longer undoes Configure-form edits made while it was running (it used to write back the whole config it read when it started). Every change to a channel's config now re-reads the file at the moment it saves and changes only its own fields, so a sync stamping its time and a form save made at the same moment both land. Two writes to the same file from the editor no longer share one temporary file.
diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts
@@ -1,7 +1,7 @@
"use server";
import path from "node:path";
-import { readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises";
+import { readdir, readFile, rm, stat } from "node:fs/promises";
import { revalidatePath } from "next/cache";
import { redirect } from "next/navigation";
import type {
@@ -71,6 +71,7 @@ import {
import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
import { makeTaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks";
import { fixIncompleteTranscriptOne } from "../../lib/fixIncompleteTranscript";
+import { writeFileAtomic, writeJsonAtomic } from "yt-dlp-transcript-common/lib/jsonFile-server";
function videoQueueKey(config: ChannelConfig, override: string | undefined): string {
return resolveQueueKey(downloadQueueKey(config), override);
@@ -495,10 +496,7 @@ export async function setPrimaryTranscriptAction(
return { ok: false, error: `File not found: ${filename}` };
}
if (filename !== VTT_FILENAME) {
- const dest = path.join(videoDir, VTT_FILENAME);
- const tmp = path.join(videoDir, `${VTT_FILENAME}.tmp-${process.pid}`);
- await writeFile(tmp, raw);
- await rename(tmp, dest);
+ await writeFileAtomic(path.join(videoDir, VTT_FILENAME), raw);
}
// The video page reads the dir directly, so it reflects the new primary right
// away. The channel list + diagnostics bucket read the cached snapshot and
@@ -607,9 +605,8 @@ export async function markVideoUntranscribableAction(
} catch {
// good — file does not exist
}
- const tmp = path.join(videoDir, `transcript.tmp-${process.pid}.json`);
- await writeFile(tmp, '{"transcription":[]}\n');
- await rename(tmp, transcriptPath);
+ // Compact + "\n": exactly the literal `{"transcription":[]}\n` this wrote.
+ await writeJsonAtomic(transcriptPath, { transcription: [] }, { indent: 0 });
const failureListFile = path.join(
getPaths().channelsDir,
slug,
diff --git a/editor/app/channels/components/ChannelGroupHeaderRow.tsx b/editor/app/channels/components/ChannelGroupHeaderRow.tsx
@@ -2,6 +2,7 @@
import type { ChannelGroupSectionView } from "yt-dlp-transcript-common/views/channelGroupSections";
import { ChannelGroupLine } from "./ChannelGroupLine";
+import { RACK_LAYERS } from "./rackLayout";
// A group's section header: one full-colspan row above its channels' rows, so
// the columns stay locked across every group (comparing transcript counts
@@ -28,7 +29,12 @@ export function ChannelGroupHeaderRow({
const { group, channels } = section;
const name = group.name || group.id;
return (
- <tr className="border-t-2 border-border">
+ // THE SECTION RULE IS THE CELL'S SHADOW, NOT THE ROW'S BORDER. Tailwind's
+ // preflight collapses table borders, and a collapsed border belongs to the
+ // table grid, not to the sticky cell — so pinned under the thead, the th
+ // moved and its 2px top rule stayed behind as a gap the rows showed
+ // through. An inset shadow is painted by the th and travels with it.
+ <tr>
<th
colSpan={colSpan}
scope="rowgroup"
@@ -38,38 +44,52 @@ export function ChannelGroupHeaderRow({
className={
// The section name stays on screen while its rows scroll past, pinned
// just under the column header — whose height is measured into
- // `--thead-h` by ChannelsTable rather than guessed. Opaque, and above
- // the sticky identity cells it passes over, but below the thead.
+ // `--thead-h` by ChannelsRack rather than guessed. Opaque, and above
+ // the sticky identity cells it passes over, but below the thead
+ // (RACK_LAYERS — a tie with the identity cells is what let the Slug
+ // band paint over the station buttons).
"px-2 py-1.5 text-left font-normal align-top bg-muted " +
- "md:sticky md:top-[var(--thead-h,2.25rem)] md:z-20"
+ "shadow-[inset_0_2px_0_var(--color-border)] " +
+ `md:sticky md:top-[var(--thead-h,2.25rem)] ${RACK_LAYERS.groupHeader}`
}
>
- <div className="flex flex-wrap items-baseline gap-x-3 gap-y-1">
- <span
- data-testid="group-name"
- className="text-xs font-semibold uppercase tracking-wider"
- >
- {name}
- </span>
- {/* Rendered because it is true and currently invisible: a visitor to
- the public site does not get this group preselected. */}
- {!group.selectedByDefault && (
- <span className="text-[10px] uppercase tracking-wide text-muted-foreground border border-border rounded px-1">
- off by default
+ {/* THE CONTENT PINS LEFT. The th spans every column, so its name and
+ five stations used to scroll off to the left with the table — the
+ group's controls out of reach exactly when the operator had
+ scrolled across to read a column. Capped at the region's visible
+ width (`--rack-w`, measured by ChannelsRack; 100% before the first
+ measure) less the th's padding, so the count on the right stays
+ in view too. */}
+ <div className="sticky left-2 max-w-[calc(var(--rack-w,100%)-1rem)]">
+ <div className="flex flex-wrap items-baseline gap-x-3 gap-y-1">
+ <span
+ data-testid="group-name"
+ className="text-xs font-semibold uppercase tracking-wider"
+ >
+ {name}
</span>
- )}
- {/* The authored description. Written by SiteForm, parsed by
+ {/* Rendered because it is true and currently invisible: a visitor to
+ the public site does not get this group preselected. */}
+ {!group.selectedByDefault && (
+ <span className="text-[10px] uppercase tracking-wide text-muted-foreground border border-border rounded px-1">
+ off by default
+ </span>
+ )}
+ {/* The authored description. Written by SiteForm, parsed by
parseChannelGroup, and until now rendered nowhere in the editor. */}
- {group.description && (
- <span className="text-xs text-muted-foreground">
- {group.description}
+ {group.description && (
+ <span className="text-xs text-muted-foreground">
+ {group.description}
+ </span>
+ )}
+ <span className="ml-auto text-xs text-muted-foreground whitespace-nowrap">
+ {channels.length === 1
+ ? "1 channel"
+ : `${channels.length} channels`}
</span>
- )}
- <span className="ml-auto text-xs text-muted-foreground whitespace-nowrap">
- {channels.length === 1 ? "1 channel" : `${channels.length} channels`}
- </span>
+ </div>
+ <ChannelGroupLine section={section} siteId={siteId} />
</div>
- <ChannelGroupLine section={section} siteId={siteId} />
</th>
</tr>
);
diff --git a/editor/app/channels/components/ChannelGroupLine.tsx b/editor/app/channels/components/ChannelGroupLine.tsx
@@ -60,10 +60,12 @@ const STATIONS: Station[] = [
id: "transcribe",
label: "Transcribe",
done: "Transcribed",
- notEligible:
- "No channel in this group is set to transcribe — a youtube-handling channel never runs whisper.",
+ // Reachable only when every member is a social account: handling does not
+ // decide eligibility (channelGroupSections.ts), the files do. No method is
+ // named — which engine runs is the transcription settings' business.
+ notEligible: "No channel in this group has downloaded audio to transcribe.",
confirm: (group, count) =>
- `Transcribe ${count} downloaded video(s) across every channel in "${group}"? They run strictly one at a time on the transcription queue, and whisper is slow.`,
+ `Transcribe ${count} downloaded video(s) across every channel in "${group}"? They run one at a time on the transcription queue, and each takes minutes.`,
run: transcribeChannelGroupAction,
},
{
diff --git a/editor/app/channels/components/ChannelSelectionDeck.tsx b/editor/app/channels/components/ChannelSelectionDeck.tsx
@@ -44,6 +44,7 @@ import {
} from "../bulkStorageActions";
import type { MoveDestination } from "../lib/moveDestination";
import { useBarAction } from "./ChannelFocusBar";
+import { RACK_LAYERS } from "./rackLayout";
const TIER_LABEL: Record<StoredChannelTier, string> = {
normal: "Normal",
@@ -126,7 +127,7 @@ export function ChannelSelectionDeck({
className={
// Below md the document scrolls and the deck pins to the screen; on md+
// it is the last item of the page's flex column and is already docked.
- "sticky bottom-0 z-20 md:static md:shrink-0 " +
+ `sticky bottom-0 ${RACK_LAYERS.deck} md:static md:shrink-0 ` +
"border-t md:border md:rounded-md border-border bg-card/95 backdrop-blur " +
"px-3 py-2 text-sm shadow-lg " +
"motion-safe:animate-in motion-safe:fade-in motion-safe:slide-in-from-bottom-2"
diff --git a/editor/app/channels/components/ChannelTierSelect.tsx b/editor/app/channels/components/ChannelTierSelect.tsx
@@ -38,6 +38,7 @@ import {
setChannelTierAction,
type ActionResult,
} from "../actions";
+import { RACK_LAYERS } from "./rackLayout";
export type ChannelTierSelectProps = {
slug: string;
@@ -198,15 +199,37 @@ export default function ChannelTierSelect({
{/* OPENING THIS MUST NOT MOVE THE RACK. In a 40px row an inline panel
would push every row below it down by 150px, so the panel is absolute
- and overlays them instead. */}
- <details className="text-[11px]">
+ and overlays them instead.
+ ...WHICH MEANS THE SCROLL REGION CLIPS IT. Opened on a row near the
+ rack's bottom (or, below md, its right edge) the panel hung off the
+ region with its selects out of reach, and nothing said it was there.
+ Opening scrolls the region — and the document, below md — just far
+ enough to show the whole panel; `nearest` leaves an already-visible
+ one exactly where it is. Below md the selection deck pins to the
+ bottom of the SCREEN, outside the (isolated) rack, and paints over
+ it — so the panel keeps a 12rem scroll margin there, about the
+ deck's height, and "nearest" stops it above the deck instead of
+ under it. */}
+ <details
+ className="text-[11px]"
+ onToggle={(e) => {
+ const details = e.currentTarget;
+ if (!details.open) return;
+ requestAnimationFrame(() =>
+ details.lastElementChild?.scrollIntoView({
+ block: "nearest",
+ inline: "nearest",
+ }),
+ );
+ }}
+ >
<summary
aria-label={`advanced priority for ${slug}`}
className="cursor-pointer text-muted-foreground hover:text-foreground"
>
Advanced
</summary>
- <div className="absolute z-30 mt-1 flex w-64 flex-col gap-1 rounded-md border border-border bg-popover p-2 shadow-md">
+ <div className={`absolute ${RACK_LAYERS.popover} max-md:scroll-mb-48 mt-1 flex w-64 flex-col gap-1 rounded-md border border-border bg-popover p-2 shadow-md`}>
{PRIORITY_OPERATIONS.map((op) => (
<label key={op} className="flex items-center justify-between gap-2">
<span className="text-muted-foreground">
diff --git a/editor/app/channels/components/ChannelsRack.tsx b/editor/app/channels/components/ChannelsRack.tsx
@@ -134,11 +134,18 @@ export function ChannelsRack({
const region = regionRef.current;
const thead = theadRef.current;
if (!region || !thead || typeof ResizeObserver === "undefined") return;
- const measure = () =>
+ // `--rack-w` is the region's visible width: a group header's content pins
+ // to the region's left edge and is capped at this, so its name and
+ // stations stay on screen however far the columns scroll across
+ // (ChannelGroupHeaderRow).
+ const measure = () => {
region.style.setProperty("--thead-h", `${thead.offsetHeight}px`);
+ region.style.setProperty("--rack-w", `${region.clientWidth}px`);
+ };
measure();
const observer = new ResizeObserver(measure);
observer.observe(thead);
+ observer.observe(region);
return () => observer.disconnect();
}, []);
@@ -202,10 +209,16 @@ export function ChannelsRack({
Slug cells pin to its left, and the sixteen columns move underneath
them. The table itself must NOT clip (`overflow-hidden` would make it
the sticky ancestor and nothing would pin) — the rounded corners are
- the region's. */}
+ the region's.
+ `isolate` makes the region its own stacking context, so the whole
+ ladder (rackLayout.ts) is ordered INSIDE it: without it the pinned
+ cells and the thead competed with the page itself, and below md
+ the thead scrolling under the screen-pinned selection deck
+ (RACK_LAYERS.deck) painted over it. */}
<div
ref={regionRef}
- className="relative -mx-4 min-h-0 flex-1 overflow-auto border-y border-border md:mx-0 md:rounded-md md:border"
+ data-testid="channels-rack"
+ className="relative isolate -mx-4 min-h-0 flex-1 overflow-auto border-y border-border md:mx-0 md:rounded-md md:border"
>
<ChannelsTable
rows={channels}
diff --git a/editor/app/channels/components/ChannelsTable.tsx b/editor/app/channels/components/ChannelsTable.tsx
@@ -19,6 +19,7 @@ import {
type ChannelSortKey,
type PipelineColumn,
} from "./channelColumns";
+import { RACK_BRIDGE, RACK_IDENTITY, RACK_LAYERS } from "./rackLayout";
// ONE CHANNEL TABLE — the /channels rack, the dashboard's channels table and
// every operation page's work section draw their rows here, off the one
@@ -273,7 +274,7 @@ export function ChannelsTable({
head.push(
<th
key="select"
- className={`${sticky ? "sticky left-0 z-10 bg-muted " : ""}w-9 px-2 py-1.5 align-bottom`}
+ className={`${sticky ? `sticky left-0 ${RACK_LAYERS.identity} bg-muted ` : ""}${RACK_IDENTITY.checkboxWidth} px-2 py-1.5 align-bottom`}
>
<input
type="checkbox"
@@ -294,7 +295,7 @@ export function ChannelsTable({
// close the block run the full height of the rack.
pipelineColumns.forEach((col, i) => {
const className =
- "w-24 min-w-20" +
+ RACK_BRIDGE +
(i === 0 ? " border-l border-border" : "") +
(i === pipelineColumns.length - 1 ? " border-r border-border" : "");
head.push(
@@ -373,6 +374,11 @@ export function ChannelsTable({
const table = (
<table
className={
+ // THE FLAT PATH CLIPS, THE STICKY PATH MUST NOT. `md:overflow-hidden`
+ // keeps the flat table's bg-muted head inside its rounded border, and
+ // nothing pins there. The sticky branch must never carry an overflow
+ // class: it would make the table the sticky ancestor and nothing would
+ // pin to the rack's scroll region (ChannelsRack.tsx, "THE RACK").
sticky
? "w-full text-sm"
: "text-sm border-y md:border border-border md:rounded-md md:overflow-hidden w-full"
@@ -380,7 +386,7 @@ export function ChannelsTable({
>
<thead
ref={theadRef}
- className={sticky ? "sticky top-0 z-30 bg-muted" : "bg-muted"}
+ className={sticky ? `sticky top-0 ${RACK_LAYERS.thead} bg-muted` : "bg-muted"}
>
<tr>{head}</tr>
</thead>
@@ -503,7 +509,7 @@ function ChannelTableRow({
// background — it has to carry the same one explicitly or the rows would show
// through the pinned identity column while the rest scrolls.
const stickyBg = selected ? "bg-accent" : "bg-background";
- const bridge = "bg-surface w-24 min-w-20";
+ const bridge = `bg-surface ${RACK_BRIDGE}`;
const pad = sticky ? "px-2 py-1.5" : "px-3 py-2";
// Dimmed for the two things that take the row out of a pipeline: it is
// excluded from the export build, or its base tier is Paused. (The sync
@@ -513,7 +519,7 @@ function ChannelTableRow({
// THE DIM IS PER CELL, NEVER ON THE `<tr>`. `opacity` below 1 creates a
// STACKING CONTEXT, and a stacking context confines every positioned
// descendant to it: put `opacity-60` on the row and the Tier cell's
- // `absolute z-30` Advanced popover (ChannelTierSelect) can no longer paint
+ // `absolute` Advanced popover (ChannelTierSelect, `RACK_LAYERS.popover`) can no longer paint
// above the rows that follow, however high its z-index — every later row
// draws over it and swallows the clicks. So the Tier cell — the one that
// hosts the popover — is the one cell that is NOT dimmed (its registry entry
@@ -544,7 +550,7 @@ function ChannelTableRow({
<Td
key="select"
pad="px-2 py-1.5"
- className={`${sticky ? `sticky left-0 z-10 ${stickyBg} ` : ""}w-9${dim}`}
+ className={`${sticky ? `sticky left-0 ${RACK_LAYERS.identity} ${stickyBg} ` : ""}${RACK_IDENTITY.checkboxWidth}${dim}`}
>
<input
type="checkbox"
diff --git a/editor/app/channels/components/channelColumns.tsx b/editor/app/channels/components/channelColumns.tsx
@@ -11,6 +11,7 @@ import { ChannelAvailabilityButton } from "./ChannelAvailabilityButton";
import { ChannelBuildToggle } from "./ChannelBuildToggle";
import { ChannelSyncButton } from "./ChannelSyncButton";
import ChannelTierSelect from "./ChannelTierSelect";
+import { RACK_IDENTITY, RACK_LAYERS } from "./rackLayout";
import type {
ChannelColumnId,
ChannelSortKey,
@@ -119,12 +120,11 @@ export const CHANNEL_COLUMNS: Record<
label: "Slug",
sortKey: "slug",
th: {
- stickyClassName:
- "sticky left-8 z-20 bg-muted shadow-[1px_0_0_var(--color-border)]",
+ stickyClassName: `sticky ${RACK_IDENTITY.slugLeft} ${RACK_LAYERS.identity} bg-muted shadow-[1px_0_0_var(--color-border)]`,
},
cell: (c, ctx) => ({
className: ctx.sticky
- ? `sticky left-8 z-20 whitespace-nowrap font-mono shadow-[1px_0_0_var(--color-border)] ${ctx.stickyBg}`
+ ? `sticky ${RACK_IDENTITY.slugLeft} ${RACK_LAYERS.identity} whitespace-nowrap font-mono shadow-[1px_0_0_var(--color-border)] ${ctx.stickyBg}`
: "whitespace-nowrap font-mono",
content: (
<span className="inline-flex items-center gap-1.5">
diff --git a/editor/app/channels/components/rackLayout.test.ts b/editor/app/channels/components/rackLayout.test.ts
@@ -0,0 +1,23 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { RACK_IDENTITY, RACK_LAYERS } from "./rackLayout";
+
+// Tailwind's default spacing scale: one unit = 0.25rem = 4 px.
+const px = (token: string, prefix: string) => {
+ const m = new RegExp(`^${prefix}-(\\d+)$`).exec(token);
+ assert.ok(m, `${token} is not a ${prefix}-<n> token`);
+ return Number(m[1]) * 4;
+};
+const z = (token: string) => Number(/z-(\d+)$/.exec(token)?.[1]);
+
+test("the pinned Slug cell starts inside the checkbox cell — the overlap closes the seam", () => {
+ assert.ok(
+ px(RACK_IDENTITY.slugLeft, "left") < px(RACK_IDENTITY.checkboxWidth, "w"),
+ );
+});
+
+test("the layer ladder is strict: popover > thead > group header > identity", () => {
+ assert.ok(z(RACK_LAYERS.popover) > z(RACK_LAYERS.thead));
+ assert.ok(z(RACK_LAYERS.thead) > z(RACK_LAYERS.groupHeader));
+ assert.ok(z(RACK_LAYERS.groupHeader) > z(RACK_LAYERS.identity));
+});
diff --git a/editor/app/channels/components/rackLayout.ts b/editor/app/channels/components/rackLayout.ts
@@ -0,0 +1,59 @@
+// THE RACK'S LAYOUT TOKENS — plain data, deliberately NOT "use client" (see
+// ./channelColumnPresets.ts for why: a server component importing a value from
+// a client module gets a client reference, not the value).
+//
+// Tailwind v4 (`app/globals.css`, no config) finds classes by scanning source
+// for complete literals, so every entry is a whole class string. Never build
+// one (`z-${n}`): it would not be generated.
+
+// THE LAYER LADDER, top to bottom:
+//
+// popover > thead > groupHeader > identity
+//
+// - popover: the Tier cell's Advanced panel (ChannelTierSelect). It opens over
+// the rows below AND over the thead when its row sits just under it; it used
+// to TIE the thead at z-30 and win only by coming later in the DOM.
+// - thead: the column header pins to the top of the scroll region, over
+// everything that scrolls under it, group headers included.
+// - groupHeader: a section's header row pins under the thead (md+ only — below
+// md the region does not scroll vertically) and carries the five station
+// buttons, so it must paint over the pinned identity cells of the rows
+// scrolling up beneath it.
+// - identity: the checkbox and Slug cells pinned to the left. BOTH at the same
+// level. Two sibling cells at one z-index paint in DOM order, and the Slug
+// cell comes later, so it covers the 4 px the two overlap (RACK_IDENTITY) —
+// which is what closes the seam between them. `20ee34db` closed that seam by
+// lifting the Slug cell to z-20 instead, which TIED it with the group header:
+// scrolled down and right, the slug band of every later row painted over the
+// group header and swallowed clicks on its stations (operator report
+// 2026-09-24). A higher Slug z re-creates that tie; do not.
+//
+// - deck (off the ladder's order, beside it): the selection deck pins to the
+// bottom of the SCREEN below md (ChannelSelectionDeck), over whatever part of
+// the page scrolls under it; on md+ it is static and z does nothing.
+//
+// `groupHeader` carries its `md:` prefix because the sticky it orders is md+
+// only (ChannelGroupHeaderRow).
+export const RACK_LAYERS = {
+ popover: "z-40",
+ thead: "z-30",
+ groupHeader: "md:z-20",
+ identity: "z-10",
+ deck: "z-20",
+} as const;
+
+// THE PINNED IDENTITY COLUMN'S GEOMETRY.
+//
+// Invariant: `slugLeft` (32 px) < `checkboxWidth` (36 px). The Slug cell pins
+// 4 px INSIDE the checkbox cell, and that overlap is the seam closer (see
+// `identity` above). A cell's width is a floor under auto table layout, so the
+// checkbox cell never renders narrower and the gap cannot reopen. Equal values
+// reopen the sub-pixel seam `20ee34db` found — the scrolled columns showing
+// through a hairline between the two pins. rackLayout.test.ts holds this.
+export const RACK_IDENTITY = {
+ checkboxWidth: "w-9",
+ slugLeft: "left-8",
+} as const;
+
+// One pipeline band column's width, header and cell alike — the meter bridge.
+export const RACK_BRIDGE = "w-24 min-w-20";
diff --git a/editor/app/channels/groupActions.ts b/editor/app/channels/groupActions.ts
@@ -1,7 +1,10 @@
"use server";
import { revalidatePath } from "next/cache";
-import { listChannelBriefs } from "yt-dlp-transcript-common/controller/channels";
+import {
+ listChannelBriefs,
+ type ChannelBrief,
+} from "yt-dlp-transcript-common/controller/channels";
import type { StreamActionResult } from "yt-dlp-transcript-common/jobs/streamCommand";
import { activeSlugsForKinds } from "yt-dlp-transcript-common/jobs/syncJobs";
import { isValidGroupId } from "yt-dlp-transcript-common/lib/channelGroups";
@@ -13,11 +16,16 @@ import {
laneOffFor,
slugsInGroup,
stationWorkFor,
+ transcribeStationIds,
type StationId,
} from "yt-dlp-transcript-common/views/channelGroupSections";
import { queueForSlugs } from "./lib/queueForSlugs";
import { downloadMissingAction, syncAction } from "./[slug]/pipelineActions";
-import { transcribeMissingAction } from "./[slug]/whisperActions";
+import {
+ transcribeAutoSubsBucketAction,
+ transcribeBucketAction,
+ transcribeMissingAction,
+} from "./[slug]/whisperActions";
import { backfillChannelAction } from "./[slug]/backfillActions";
import { runOperationChannelJob } from "yt-dlp-transcript-common/controller/operationJobs";
import { DIGEST_OPERATION_ID } from "yt-dlp-transcript-common/lib/operations";
@@ -42,26 +50,92 @@ export type GroupOpResult = {
skipped: { slug: string; reason: string }[];
};
-// The job kind each station enqueues, for the "already running" dedupe. The job
-// registry has no dedupe of its own.
-const KIND_FOR: Record<StationId, string> = {
- sync: "sync",
- download: "download-missing",
- transcribe: "whisper-all",
- digest: "digest-channel-local",
- speakers: "backfill-channel",
+// The job kinds each station enqueues, for the "already running" dedupe. The job
+// registry has no dedupe of its own. Transcribe names THREE: its button queues
+// one or both bucket jobs (or the scan, for a channel with no report), and the
+// dedupe spans them all — pressing it again while any one runs, including the
+// channel page's own bucket job, reads "already running", not a second batch.
+const KIND_FOR: Record<StationId, readonly string[]> = {
+ sync: ["sync"],
+ download: ["download-missing"],
+ transcribe: [
+ "whisper-all",
+ "whisper-bucket-downloaded-no-transcript",
+ "whisper-bucket-auto-subs",
+ ],
+ digest: ["digest-channel-local"],
+ speakers: ["backfill-channel"],
};
+// THE TRANSCRIBE STATION QUEUES WHAT ITS FIGURE COUNTS — `transcribeStationIds`,
+// the fold `stationWorkFor` sums — BY ID. Two batches, because one cannot cover
+// both: the no-transcript half is `transcribeBucketAction` over its ids (the
+// channel page's own bucket control, replayable as `downloadedNoTranscript`),
+// and the auto-captions half is `transcribeAutoSubsBucketAction`, whose
+// replace mode needs an ASR track per id. By id matters: a video downloaded
+// before it went private, members-only or deleted is still on disk and still
+// in `downloadedNoTranscript`, and a directory SCAN would transcribe it — one
+// more than the figure said. The channel page runs the same jobs; a combined
+// job kind would need its own replay spec and /jobs label and mirror nothing.
+// Handling is not consulted — a youtube channel's captionless download is
+// whisper work too.
+//
+// The SCAN (`transcribeMissingAction`, a whisper-all with no ids) is used ONLY
+// when the channel has never reported: with no snapshot there are no ids to
+// name, and the scan finds what there is, as the station always did.
+//
+// All three kinds resolve to TRANSCRIPTION_QUEUE, so the "strictly one at a
+// time" note above stays true. `queued` counts CHANNELS; `jobIds` carries one
+// id per channel — the no-transcript job's when both halves were queued, and
+// the other stream is cancelled here exactly as queueForSlugs cancels the one
+// returned. A half that fails while the other queues cannot ride on an ok
+// result (StreamActionResult's ok arm has no message), so it is logged.
+async function transcribeChannel(
+ slug: string,
+ brief: ChannelBrief | undefined,
+): Promise<StreamActionResult> {
+ if (!brief?.snapshot) return transcribeMissingAction(slug);
+ const ids = transcribeStationIds(brief.snapshot);
+ const halves = [
+ ids.missing.length > 0
+ ? await transcribeBucketAction(
+ slug,
+ ids.missing,
+ undefined,
+ undefined,
+ undefined,
+ "downloadedNoTranscript",
+ )
+ : null,
+ ids.autoSubs.length > 0
+ ? await transcribeAutoSubsBucketAction(slug, ids.autoSubs)
+ : null,
+ ].filter((r): r is StreamActionResult => r !== null);
+ // Both empty is skipped upstream as "nothing to do"; say so if reached.
+ if (halves.length === 0) return { ok: false, error: "nothing to do", info: true };
+ const ok = halves.find((r) => r.ok);
+ if (!ok) return halves[0];
+ for (const r of halves) {
+ if (r === ok) continue;
+ if (r.ok) void r.stream.cancel();
+ else
+ console.warn(
+ `[group transcribe] ${slug}: one half queued, the other refused: ${r.error}`,
+ );
+ }
+ return ok;
+}
+
const RUN_FOR: Record<
StationId,
- (slug: string) => Promise<StreamActionResult>
+ (slug: string, brief: ChannelBrief | undefined) => Promise<StreamActionResult>
> = {
sync: (slug) => syncAction(slug, undefined),
// runPipelineAction already refuses with {ok:false, error} for a paused
// install, a per-platform 429 cooldown and low disk, so all three land in
// `skipped` for free.
download: (slug) => downloadMissingAction(slug),
- transcribe: (slug) => transcribeMissingAction(slug),
+ transcribe: (slug, brief) => transcribeChannel(slug, brief),
// THE LOCAL LANE, PINNED. The metered lane is behind a settings gate and a
// spend cap, so it is never what a group button starts — which is why this
// names the lane explicitly instead of letting the dispatcher pick the
@@ -132,7 +206,7 @@ async function runGroupStation(
members.has(b.slug),
);
const briefBySlug = new Map(briefs.map((b) => [b.slug, b]));
- const active = activeSlugsForKinds([KIND_FOR[station]]);
+ const active = activeSlugsForKinds(KIND_FOR[station]);
const outcome = await queueForSlugs(
briefs.map((b) => b.slug),
@@ -153,7 +227,7 @@ async function runGroupStation(
if (station !== "sync" && work.work === 0) return "nothing to do";
return null;
},
- run: RUN_FOR[station],
+ run: (slug) => RUN_FOR[station](slug, briefBySlug.get(slug)),
},
);
return { group, ...outcome };
diff --git a/editor/app/sites/lib/cutReleaseAction.ts b/editor/app/sites/lib/cutReleaseAction.ts
@@ -9,6 +9,7 @@ import {
CutReleaseError,
} from "yt-dlp-transcript-common/lib/changelog";
import { commitPath, listDirtyPaths } from "yt-dlp-transcript-common/lib/git";
+import { writeFileAtomic } from "yt-dlp-transcript-common/lib/jsonFile-server";
export type CutReleaseState =
| { ok: true; version: string; committed: boolean }
@@ -93,9 +94,7 @@ export async function cutReleaseAction(
}
throw err;
}
- const tmp = `${filePath}.tmp-${process.pid}`;
- await fs.promises.writeFile(tmp, next);
- await fs.promises.rename(tmp, filePath);
+ await writeFileAtomic(filePath, next);
if (shouldCommit) {
const result = await commitPath(
paths.monorepoRoot,
diff --git a/editor/e2e/channel-groups.spec.ts b/editor/e2e/channel-groups.spec.ts
@@ -1,9 +1,12 @@
import { test, expect, type Page } from "@playwright/test";
+import { mkdir, writeFile } from "node:fs/promises";
import {
channelStage,
generateReport,
jobRowByKind,
+ pathExists,
resetData,
+ resolvePath,
writeChannelConfig,
writeSite,
} from "./helpers";
@@ -147,18 +150,19 @@ test("a group's Sync queues that group's channels and nothing else", async ({
await expect(syncRows).toContainText("slow-b");
});
-test("a station with no eligible channel is disabled and says why", async ({
+test("transcribe is open to a youtube channel; a station with no eligible channel is disabled and says why", async ({
page,
}) => {
await seed();
await page.goto("/channels?site=alpha");
- // Both fixture channels are handling: "youtube", and a youtube channel never
- // runs whisper — so counting it would inflate the figure on a button that
- // would skip it anyway.
+ // Both fixture channels are handling: "youtube" — and that no longer shuts
+ // the station: a youtube video that came down with no captions is whisper
+ // work like any other. Neither channel has reported, so the figure is the
+ // honest unknown, as the download station's is below.
const transcribe = page.getByLabel("transcribe group News");
- await expect(transcribe).toBeDisabled();
- await expect(transcribe).toHaveAttribute("title", /never runs whisper/);
+ await expect(transcribe).toBeEnabled();
+ await expect(transcribe).toHaveText("Transcribe —");
// The default test settings enable no operation on the backfill lane, so the
// speakers station reads "off" — NOT 0 (which reads as finished) and not —
@@ -206,3 +210,79 @@ test("a group figure reads — until a channel reports, then a real number", asy
)
.toMatch(/Download 5$/);
});
+
+// An auto-caption-only video, shaped like real yt-dlp --write-auto-subs output
+// (the provenance sniff reads cue settings and inline word timings), with its
+// audio on disk — i.e. the `downloadedAutoSubsOnly` bucket.
+const ASR_VTT = `WEBVTT
+Kind: captions
+Language: en
+
+00:00:00.030 --> 00:00:03.919 align:start position:0%
+so<00:00:00.719> today<00:00:01.199> we're<00:00:01.439> going<00:00:01.680> to
+
+00:00:03.919 --> 00:00:03.929 align:start position:0%
+so today we're going to
+
+00:00:03.929 --> 00:00:07.070 align:start position:0%
+so today we're going to
+talk<00:00:04.320> about<00:00:04.639> the<00:00:04.879> whole<00:00:05.199> thing
+`;
+
+test("a youtube group's Transcribe counts and queues its auto-caption-only videos", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await seed();
+ const id = "asrgrp0001";
+ const dataRel = `test-transcripts/channels/slow-b/data/${id}`;
+ const dir = resolvePath(dataRel);
+ await mkdir(dir, { recursive: true });
+ await writeFile(`${dir}/transcript.en.vtt`, ASR_VTT);
+ await writeFile(
+ `${dir}/metadata.info.json`,
+ JSON.stringify({
+ id,
+ title: `Synthetic ${id}`,
+ upload_date: "20240101",
+ duration: 60,
+ extractor_key: "Youtube",
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ subtitles: {},
+ automatic_captions: { en: [{ ext: "vtt", url: "fake://subs" }] },
+ }),
+ );
+ await writeFile(`${dir}/audio.mp3`, `fake audio ${id}\n`);
+ await writeFile(
+ resolvePath("test-transcripts/channels/slow-b/playlist"),
+ `https://www.youtube.com/watch?v=${id}\n`,
+ );
+ await generateReport(page, "slow-b");
+
+ await page.goto("/channels?site=alpha");
+ const transcribe = page.getByLabel("transcribe group News");
+ await expect(transcribe).toHaveText("Transcribe 1");
+ await expect(transcribe).toBeEnabled();
+ page.once("dialog", (d) => void d.accept());
+ await transcribe.click();
+ await expect(page.getByLabel("transcribe group News result")).toContainText(
+ /Queued 1 . skipped 0/,
+ { timeout: 15_000 },
+ );
+
+ await page.goto("/jobs");
+ const rows = jobRowByKind(page, "whisper-bucket-auto-subs");
+ await expect(rows).toHaveCount(1);
+ await expect(rows).toContainText("slow-b");
+ // Nothing was captionless, so neither the by-id no-transcript batch nor the
+ // scan rode along.
+ await expect(
+ jobRowByKind(page, "whisper-bucket-downloaded-no-transcript"),
+ ).toHaveCount(0);
+ await expect(jobRowByKind(page, "whisper-all")).toHaveCount(0);
+
+ // Let the fake whisper finish before the next spec's resetData.
+ await expect
+ .poll(() => pathExists(`${dataRel}/transcript.json`), { timeout: 60_000 })
+ .toBe(true);
+});
diff --git a/editor/e2e/channels-rack-audit.spec.ts b/editor/e2e/channels-rack-audit.spec.ts
@@ -0,0 +1,143 @@
+import { test, expect, type Page } from "@playwright/test";
+import {
+ resetData,
+ resolvePath,
+ writeChannelConfig,
+ writeSite,
+} from "./helpers";
+
+// THE /channels RACK AUDIT — screenshots, not assertions. A person (and the
+// reviewer) LOOKS at every shot for layering, overlap, clipping and
+// misalignment; channels-rack-layers.spec.ts is where a finding becomes a
+// hit-test that fails. Skipped unless RACK_SHOTS is set, so the ordinary suite
+// never spends time here:
+//
+// RACK_SHOTS=1 pnpm e2e channels-rack-audit.spec.ts (from the repo root)
+//
+// Shots land in editor/test-results/rack-shots/ (gitignored; Playwright clears
+// test-results/ at the start of every run, so copy a set out before the next).
+
+test.skip(!process.env.RACK_SHOTS, "audit only");
+
+const EXTRA = Array.from(
+ { length: 16 },
+ (_, i) => `rack-${String(i + 1).padStart(2, "0")}`,
+);
+
+// Two groups so a section boundary is on screen, and enough rows to scroll on
+// both axes.
+async function seed() {
+ await resetData("two-slow-channels");
+ for (const slug of EXTRA) await writeChannelConfig(slug);
+ await writeSite("alpha", {
+ siteTitle: "Alpha",
+ groups: [
+ { id: "default", name: "All channels", selectedByDefault: true, order: 2 },
+ {
+ id: "news",
+ name: "News",
+ description: "Shows where the host is the guest",
+ selectedByDefault: false,
+ order: 1,
+ },
+ ],
+ channels: [
+ { slug: "slow-b", groupId: "news" },
+ { slug: "rack-01", groupId: "news" },
+ { slug: "slow-a" },
+ ...EXTRA.slice(1).map((slug) => ({ slug })),
+ ],
+ });
+}
+
+const VIEWPORTS = [
+ { name: "desktop", width: 1440, height: 900 },
+ { name: "mobile", width: 390, height: 844 },
+] as const;
+
+async function shot(page: Page, name: string) {
+ await page.screenshot({ path: resolvePath(`test-results/rack-shots/${name}.png`) });
+}
+
+// Bottom-right: the region's own scroll (both axes on md+, the horizontal one
+// below md) and the document's (below md the document is what scrolls down).
+async function scrollBottomRight(page: Page) {
+ await page.getByTestId("channels-rack").evaluate((el) => {
+ el.scrollTop = el.scrollHeight;
+ el.scrollLeft = el.scrollWidth;
+ });
+ await page.evaluate(() => window.scrollTo(0, document.body.scrollHeight));
+ // Let sticky / ResizeObserver settle before the shutter.
+ await page.waitForTimeout(250);
+}
+
+async function open(page: Page) {
+ await seed();
+ await page.goto("/channels?site=alpha");
+ await expect(page.getByRole("link", { name: "rack-16" })).toBeVisible();
+ // HYDRATED, not just painted: the stations are disabled until mount, and the
+ // rack's measured custom properties (--thead-h, --rack-w) and every
+ // onToggle exist only from then on. A click before it lands on server HTML.
+ await expect(page.getByLabel("sync group News")).toBeEnabled();
+}
+
+for (const vp of VIEWPORTS) {
+ test.describe(`rack audit @ ${vp.name}`, () => {
+ test.use({ viewport: { width: vp.width, height: vp.height } });
+
+ test(`${vp.name} grouped, top`, async ({ page }) => {
+ await open(page);
+ await shot(page, `${vp.name}-grouped-top`);
+ });
+
+ test(`${vp.name} grouped, scrolled bottom-right`, async ({ page }) => {
+ await open(page);
+ await scrollBottomRight(page);
+ await shot(page, `${vp.name}-grouped-scrolled`);
+ });
+
+ test(`${vp.name} flat, scrolled bottom-right`, async ({ page }) => {
+ await open(page);
+ await page.getByLabel("Group by section").uncheck();
+ await expect(page.getByRole("rowheader")).toHaveCount(0);
+ await scrollBottomRight(page);
+ await shot(page, `${vp.name}-flat-scrolled`);
+ });
+
+ test(`${vp.name} deck open, scrolled bottom`, async ({ page }) => {
+ await open(page);
+ await page.getByLabel("select all channels").check();
+ await expect(page.getByLabel("channel priority bulk")).toBeVisible();
+ await scrollBottomRight(page);
+ await shot(page, `${vp.name}-deck-open`);
+ });
+
+ // Below md the deck pins to the SCREEN while the document scrolls: park
+ // the column header under the deck's top edge to see which paints on top.
+ // A shorter window, because at 844 px the header sits above where the
+ // deck starts before the page has scrolled at all.
+ test(`${vp.name} deck open over the column header`, async ({ page }) => {
+ await page.setViewportSize({ width: vp.width, height: 640 });
+ await open(page);
+ await page.getByLabel("select all channels").check();
+ const deck = page.getByLabel("channel priority bulk");
+ await expect(deck).toBeVisible();
+ await page.evaluate(() => {
+ const deckTop = document
+ .querySelector('[aria-label="channel priority bulk"]')!
+ .getBoundingClientRect().top;
+ const theadTop = document.querySelector("thead")!.getBoundingClientRect().top;
+ window.scrollBy(0, theadTop - (deckTop + 12));
+ });
+ await page.waitForTimeout(250);
+ await shot(page, `${vp.name}-deck-over-thead`);
+ });
+
+ test(`${vp.name} Advanced popover open`, async ({ page }) => {
+ await open(page);
+ await page.getByLabel("advanced priority for slow-a").click();
+ await expect(page.getByLabel("download override for slow-a")).toBeVisible();
+ await shot(page, `${vp.name}-popover-open`);
+ });
+ });
+}
diff --git a/editor/e2e/channels-rack-layers.spec.ts b/editor/e2e/channels-rack-layers.spec.ts
@@ -0,0 +1,191 @@
+import { test, expect, type Locator, type Page } from "@playwright/test";
+import { resetData, writeChannelConfig, writeSite } from "./helpers";
+
+// THE RACK'S LAYER LADDER, asserted by hit-testing rather than by reading
+// classes: `document.elementFromPoint` answers "what would a click here land
+// on", which is the operator's complaint ("rows scroll OVER the group
+// controls") stated as a test.
+//
+// The order is popover > thead > group header > identity cells
+// (app/channels/components/rackLayout.ts). The regression this pins down was a
+// TIE: the pinned Slug cell and the group header both at z-20, broken by DOM
+// order — so scrolled down AND right, the slug band of every later row painted
+// over the group header and swallowed clicks on its station buttons.
+
+// Enough channels that the default group's rows outgrow the 1280x720 region,
+// so it scrolls vertically; the sixteen columns already make it scroll across.
+const EXTRA = Array.from(
+ { length: 16 },
+ (_, i) => `rack-${String(i + 1).padStart(2, "0")}`,
+);
+
+// Seeded, loaded and HYDRATED: the stations are disabled until mount, and the
+// measured --thead-h / --rack-w and the popover's onToggle exist only after it.
+async function open(page: Page) {
+ await seed();
+ await page.goto("/channels?site=alpha");
+ await expect(page.getByLabel("sync group News")).toBeEnabled();
+}
+
+async function seed() {
+ await resetData("two-slow-channels");
+ for (const slug of EXTRA) await writeChannelConfig(slug);
+ await writeSite("alpha", {
+ siteTitle: "Alpha",
+ groups: [
+ { id: "default", name: "All channels", selectedByDefault: true, order: 2 },
+ { id: "news", name: "News", selectedByDefault: false, order: 1 },
+ ],
+ channels: [
+ { slug: "slow-b", groupId: "news" },
+ { slug: "slow-a" },
+ ...EXTRA.map((slug) => ({ slug })),
+ ],
+ });
+}
+
+// Does a hit-test at (x, y) land inside `target`?
+async function hitsInside(
+ page: Page,
+ target: Locator,
+ x: number,
+ y: number,
+): Promise<boolean> {
+ const handle = await target.elementHandle();
+ return page.evaluate(
+ ([el, px, py]) => {
+ const hit = document.elementFromPoint(px as number, py as number);
+ return !!hit && (el as Element).contains(hit);
+ },
+ [handle, x, y] as const,
+ );
+}
+
+test.use({ viewport: { width: 1280, height: 720 } });
+
+test("scrolled to the bottom-right, the pinned group header paints over the pinned identity cells", async ({
+ page,
+}) => {
+ await open(page);
+
+ const region = page.getByTestId("channels-rack");
+ await expect(region).toBeVisible();
+ await expect(page.getByRole("link", { name: "rack-16" })).toBeVisible();
+ // The premise: the region really scrolls on both axes, or nothing below
+ // tests anything.
+ const extent = await region.evaluate((el) => ({
+ y: el.scrollHeight - el.clientHeight,
+ x: el.scrollWidth - el.clientWidth,
+ }));
+ expect(extent.y).toBeGreaterThan(0);
+ expect(extent.x).toBeGreaterThan(0);
+ await region.evaluate((el) => {
+ el.scrollTop = el.scrollHeight;
+ el.scrollLeft = el.scrollWidth;
+ });
+
+ const regionBox = (await region.boundingBox())!;
+ const header = page.getByRole("rowheader", { name: "All channels" });
+ const headerBox = (await header.boundingBox())!;
+ // The header pinned under the thead, inside the region's viewport.
+ expect(headerBox.y).toBeGreaterThan(regionBox.y);
+ expect(headerBox.y + headerBox.height).toBeLessThan(
+ regionBox.y + regionBox.height,
+ );
+
+ // (1) Inside the pinned Slug band (it starts at left-8 = 32px of the
+ // region) and on the header's line: the header owns that point. The header
+ // itself scrolled left with the table, so its own box.x is off to the left —
+ // the band is measured from the region, which is what stays put.
+ const x = regionBox.x + 40;
+ const y = headerBox.y + headerBox.height / 2;
+ expect(await hitsInside(page, header, x, y)).toBe(true);
+
+ // (2) The same fact as numbers: the header's z sits strictly above the slug
+ // cell's. A tie is what broke it.
+ const slugCell = page.locator("td", {
+ has: page.getByRole("link", { name: "rack-16", exact: true }),
+ });
+ const zOf = (l: Locator) =>
+ l.evaluate((el) => Number(getComputedStyle(el).zIndex) || 0);
+ expect(await zOf(header)).toBeGreaterThan(await zOf(slugCell));
+
+ // (2b) The group's controls stayed on screen: the header's content pins to
+ // the region's left edge, so scrolled all the way across, its Sync station
+ // is inside the region and a click on it lands on it.
+ const sync = page.getByLabel("sync group All channels");
+ const syncBox = (await sync.boundingBox())!;
+ expect(syncBox.x).toBeGreaterThanOrEqual(regionBox.x);
+ expect(syncBox.x + syncBox.width).toBeLessThanOrEqual(
+ regionBox.x + regionBox.width,
+ );
+ expect(
+ await hitsInside(
+ page,
+ sync,
+ syncBox.x + syncBox.width / 2,
+ syncBox.y + syncBox.height / 2,
+ ),
+ ).toBe(true);
+
+ // (3) The column header still owns its own centre over everything scrolling
+ // under it.
+ const slugTh = page.locator("th", {
+ has: page.getByRole("button", { name: "sort by Slug" }),
+ });
+ const thBox = (await slugTh.boundingBox())!;
+ expect(
+ await hitsInside(
+ page,
+ slugTh,
+ thBox.x + thBox.width / 2,
+ thBox.y + thBox.height / 2,
+ ),
+ ).toBe(true);
+});
+
+test("the Advanced priority popover paints over the rows, pins and headers around it", async ({
+ page,
+}) => {
+ await open(page);
+
+ // A row with rows under it: the panel overlays them, and their pinned
+ // identity cells, which is the overlap the ladder orders.
+ await page.getByLabel("advanced priority for rack-03").click();
+ const popover = page
+ .getByLabel("advanced priority for rack-03")
+ .locator("xpath=following-sibling::div[1]");
+ await expect(popover).toBeVisible();
+ const box = (await popover.boundingBox())!;
+ expect(
+ await hitsInside(page, popover, box.x + box.width / 2, box.y + box.height / 2),
+ ).toBe(true);
+ // And its bottom edge, which overlays the rows that follow it.
+ expect(
+ await hitsInside(page, popover, box.x + box.width / 2, box.y + box.height - 4),
+ ).toBe(true);
+});
+
+test("an Advanced popover opened on the last row scrolls itself into view", async ({
+ page,
+}) => {
+ await open(page);
+ const region = page.getByTestId("channels-rack");
+ // slow-a sorts last: its panel opens past the region's bottom edge.
+ await page.getByLabel("advanced priority for slow-a").click();
+ const popover = page
+ .getByLabel("advanced priority for slow-a")
+ .locator("xpath=following-sibling::div[1]");
+ await expect(popover).toBeVisible();
+ await expect
+ .poll(async () => {
+ const r = (await region.boundingBox())!;
+ const p = (await popover.boundingBox())!;
+ return p.y + p.height <= r.y + r.height + 1;
+ })
+ .toBe(true);
+ const box = (await popover.boundingBox())!;
+ expect(
+ await hitsInside(page, popover, box.x + box.width / 2, box.y + box.height - 4),
+ ).toBe(true);
+});
diff --git a/plans/STATE.md b/plans/STATE.md
@@ -3,7 +3,7 @@
The working memory for the local-AI derived-corpus work. Rewritten at the end of every
session, before context is cleared. See [`README.md`](README.md) for the protocol.
-**Live on :3001 (2026-09-24, 18:42):** `9ab10d77` (release 3), `BUILD_ID` `XsKaA_drqdAbTguxUGVsn`; smoke clean, boot and a form save rewrote nothing but the saved channel's key order; the one-sync md5 sweep is still owed (YouTube 429 cooldown refused it) — [rollout record](one-core-phase-3.md#rollout-2026-09-24--9ab10d77-live-on-3001).
+**Live on :3001 (2026-09-24, 18:42):** `9ab10d77` (release 3), `BUILD_ID` `XsKaA_drqdAbTguxUGVsn`; smoke clean, boot and a form save rewrote nothing but the saved channel's key order; the one-sync md5 sweep is owed — YouTube held a 429 cooldown across three attempts (18:43, 19:13, 19:41) — [rollout record](one-core-phase-3.md#rollout-2026-09-24--9ab10d77-live-on-3001).
**Last updated:** 2026-09-24 — **release 3: one-core Phase 3 slices 3a and 4b are on `main` @
`ef88ac4d`.** Rollout prelude first (`c0a90a37`, `1e27f7c3`: the backfill lane form's
diff --git a/plans/one-core-phase-3.md b/plans/one-core-phase-3.md
@@ -688,13 +688,19 @@ with no field changed on `legal-mindset`, through `pnpm ops channel-config` with
`channels/legal-mindset/config.json` changed, `87ee6758…` → `7dd35dac…` — exactly the md5
the offline write-back predicted; `diff <(jq -S) <(jq -S)` empty, the raw diff moves `name`
after `platform` and `cookieMode` after `downloadFilter` (key order only). `settings.json` and
-every `site.json` unchanged. (b) **One sync was not run.** `pnpm ops sync
-{"slug":"FearAnd"}` was refused by the editor — "youtube is in a rate-limit cooldown (1634s
-remaining)", the 429 cooldown the autoQueueStatus pair caught opening — so no sync job ran;
-the md5 sweep after the refusal is unchanged. A non-YouTube channel was not substituted.
-Syncs stamp through `patchChannelConfig`, so the expected effect of a sync is a
-`lastSyncedAt` (and, on a full sweep, `lastFullSweepAt`) change on the synced channel's file
-only; that remains unmeasured for this build.
+every `site.json` unchanged. (b) **Owed; YouTube held a 429 cooldown across three attempts (18:43, 19:13, 19:41).**
+`pnpm ops sync {"slug":"FearAnd"} --wait` was refused by the editor each time — "youtube is
+in a rate-limit cooldown" with 1634 s, 1534 s and 1680 s remaining: auto-download kept hitting
+HTTP 429 and re-arming it, so no sync job ran. A non-YouTube channel was not substituted. The
+md5 sweep after the third refusal, against the post-form-save sweep, moved one file:
+`channels/the-quartering-rumble/config.json` (`ac111558…` → `fed252e8…`, written 19:36:06),
+which gained `fullSweepIntervalMinutes: 0` and was re-emitted in schema key order
+(`lastSyncedAt` unchanged). That is not this rollout: it is the operator's step 1 of
+[`rumble-sweep-pacing.md`](rumble-sweep-pacing.md), applied through the editor; the file is
+`jq -S` equal to its release-3 frozen copy once that key is removed. Syncs stamp through
+`patchChannelConfig`, so the expected effect of a sync is a `lastSyncedAt` (and, on a full
+sweep, `lastFullSweepAt`) change on the synced channel's file only; that remains unmeasured
+for this build.
**Boot cost, as observed:** small. The API views and every page but one answered in 0–3 s from
8 s after boot; `/operations/diarization` took 62 s and the first `/api/auto-queue/status`
@@ -703,6 +709,251 @@ undated candidates`, twice — once per Next module graph). IO pressure at boot
`full avg10 7.26` (58–66 % at the last rollout, when a remux was saturating the platter), which
is the difference from the 25 minutes release 2 paid.
+### Slice P, as shipped — /channels rack polish (2026-09-24)
+
+Branch `one-core/phase-3-p` off `4130aca1`, eleven commits (the ten below and this record, amended after review),
+not merged — the parent merges; slice W merges after it (W folded nothing under
+`editor/app/channels/components`). The operator's ask: fix every table and z-index problem on
+/channels ("channel rows scroll OVER the group-based controls"). Also, the group Transcribe
+station must stop refusing youtube-handling channels with a whisper-specific sentence.
+
+| sha | what |
+|---|---|
+| `5ac3e8ca` | `rackLayout.ts` — one named layer ladder (popover z-40 > thead z-30 > group header md:z-20 > identity z-10, plus the deck), `RACK_IDENTITY`, `RACK_BRIDGE`; every class site reads it; unit test for the width invariant and the order; `data-testid="channels-rack"`; flat-path overflow comment; `channels-rack-layers.spec.ts` |
+| `7a0b3d75` | the transcribe station counts what its button queues: handling branch deleted, `transcribeStationIds` (one fold, both buckets, exclusions), the group action runs `transcribeAutoSubsBucketAction` + `transcribeMissingAction`, `KIND_FOR` lists both kinds, method-free strings; unit + e2e |
+| `564f767d` | `channels-rack-audit.spec.ts` — 12 screenshots behind `RACK_SHOTS` |
+| `99b06b40` | audit fix A: a group header's name and stations pin left (`sticky left-2`, capped at the measured `--rack-w`) |
+| `53bbd482` | audit fix B: an opened Advanced panel scrolls itself into view (`nearest`) |
+| `476c8470` | audit fix D: the section rule is the th's inset shadow, not the `<tr>`'s collapsed border |
+| `6ff63cbf` | audit fix C: the scroll region is `isolate` (its own stacking context); both rack specs wait for hydration |
+| `e0f731fa` | review F1/F2: the no-transcript half is queued BY ID (`transcribeBucketAction`, `downloadedNoTranscript`); the scan only with no snapshot; `KIND_FOR.transcribe` gains `whisper-bucket-downloaded-no-transcript`; a refused half is logged |
+| `4f951ecc` | review F3/F5: below md an opened Advanced panel keeps a `scroll-mb-48` so it stops above the screen-pinned deck; the region comment names layers by key |
+
+**Root cause 1 — rows over the group controls.** It was a z-index TIE, broken by DOM order.
+The group header (`ChannelGroupHeaderRow`, `md:sticky … md:z-20`) holds the five station
+buttons. The pinned Slug cell had been raised to `z-20` by **`20ee34db`** (2026-09-13) to
+close a sub-pixel seam against the checkbox cell. Scrolled down and right, every later row's
+slug band painted over the header, because it comes later in the DOM. The z bump was never
+needed. `left-8` (32 px) pins the Slug cell 4 px inside the `w-9` (36 px) checkbox cell. Two
+sibling cells at one z-index paint in DOM order, so the Slug cell already covers the overlap at
+z-10. The rack plan's order (`plans/editor-channels-rack.md:198-204`) was right. `278d4463`
+shipped it as 30/20/10, and `20ee34db` put the slug cell on the header's level. A second tie
+went unstated: the popover and the thead were both z-30, and the popover won on DOM order only.
+Both ties are gone. After the fix, grep finds no `z-<n>` literal in
+`editor/app/channels/components/` outside `rackLayout.ts`.
+
+**Root cause 2 — "a youtube-handling channel never runs whisper".** `stationWorkFor`
+refused `handling !== "transcribe"`. But buckets are decided by files, never by handling:
+`downloadedNoTranscript` is whisper work for every channel, and the runner drains it for
+every channel. The channel page already replaces auto-captions for any handling. So the
+station never counted what the runner would do. It now counts `|downloadedNoTranscript| +
+|downloadedAutoSubsOnly|` after the download exclusions. The two buckets are disjoint, and one
+batch cannot cover both, so the button queues two jobs:
+- `whisper-bucket-auto-subs` over the auto-caption ids;
+- `whisper-all` when there are captionless videos, or when the channel has no report.
+
+Both jobs run on `TRANSCRIPTION_QUEUE`. The dedupe spans both kinds. `queued` counts channels,
+and `jobIds` carries the whisper-all id when both jobs were queued. **By id, after review:** with a snapshot, the no-transcript half is
+`transcribeBucketAction` over exactly `transcribeStationIds(…).missing`
+(`whisper-bucket-downloaded-no-transcript`, replayable as `downloadedNoTranscript`), not the
+`whisper-all` scan. `downloadedNoTranscript` does not filter excluded ids, so a scan would also
+transcribe a video downloaded before it went private: 2 on the label, 3 transcribed. The scan
+remains only for a channel with no report. The dedupe now spans all three kinds, including the
+channel page's own bucket job. If one half is refused while the other queues, the refusal is
+logged (`console.warn`): the ok arm of `StreamActionResult` has no message field to carry it. A combined job kind was
+rejected: it would need a replay spec and a /jobs label, and it would mirror nothing, since the
+channel page runs two jobs. A social account is still ineligible (`"social account"`), and
+that is now the only way `notEligible` can be reached.
+
+**The divergence, on purpose.** The rack's transcription band still counts
+`downloadedNoTranscript` alone (`channelSnapshot.ts:583-585`), and the stage title lists
+auto-captions as informational (`stageStatus.ts:344-352`). On a youtube channel with many
+auto-caption-only videos, the station's figure is now higher than the band's. That is the ask:
+the station counts exactly what its button queues. It is not a bug.
+
+**The audit.** Viewports 1440×900 and 390×844. Shots: grouped top; grouped and flat scrolled
+bottom-right; deck open scrolled to the bottom; deck parked over the column header (390×640 /
+1440×640); Advanced popover open. Baseline shots are in `$T/p-shots-before/` (pre-fix, plus
+the testid only). Fix C's before-shot is in `$T/p-shots-c-before/`. The final set is in
+`$T/p-shots-after/` (12 PNGs). `$T` = `/home/user/.claude/jobs/c0baff27/tmp`.
+
+| shot | finding | fix |
+|---|---|---|
+| desktop-grouped-scrolled, layers spec | the pinned slug band paints over the pinned group header and its stations (root cause 1) | `5ac3e8ca` |
+| desktop/mobile-grouped-scrolled | the group header's name and five stations scroll off to the left with the table; only "16 channels" stays in view | `99b06b40` |
+| desktop/mobile-popover-open, layers spec at 1280×720 | the Advanced panel on a row near the bottom (below md: near the right edge too) is clipped by the scroll region, with its selects out of reach | `53bbd482` |
+| desktop-grouped-scrolled (after `5ac3e8ca`) | a gap under the pinned group header: the `<tr>`'s collapsed `border-t-2` belongs to the table grid, so it stays behind when the th pins | `476c8470` |
+| mobile-deck-over-thead | the z-30 thead paints over the screen-pinned z-20 deck ("18 selected", tier select), because the region formed no stacking context | `6ff63cbf` |
+| mobile-* | below md the thead never pins. The region scrolls on both axes, so `sticky top-0` pins to the region and not to the document | recorded, not fixed — the documented trade-off (`ChannelsRack.tsx:152-155`, `channels/page.tsx:332-336`) |
+| mobile-deck-open | the deck at the end of the scroll sits in flow after the last row and covers nothing; while scrolling, it covers what passes under it, as a pinned bar does | recorded, no defect — no padding needed |
+| every shot | the Next dev-tools badge at the bottom left | recorded, dev-only |
+| desktop/mobile-popover-open (after) | while a panel is open, the region's scroll extent grows and a blank band shows under the last row | recorded, not fixed (see "Left") |
+| mobile-deck-over-thead (after) | the region's `-mx-4` edge shows the thead checkbox and the row edge in the 16 px gutter beside the deck | recorded, not fixed — cosmetic, predates the slice |
+| mobile-grouped-* | "Derived data off" wraps under the station line | recorded, not fixed — a wrap at 390 px, not a defect |
+| mobile popover under the deck (review) | with the rack isolated, the deck paints over an overlapping panel, and the scroll-into-view stopped the panel under it | `4f951ecc` |
+| desktop/mobile-*-scrolled | the Build/Tier columns show as a sliver under the pinned slug band | recorded, no defect — columns scroll under a pinned identity column |
+
+**The layers spec fails on the pre-fix code.** The run was the spec's first revision, with two
+tests: the header case and a popover case on `slow-a`. It ran on the `4130aca1` components plus
+the testid only. The header hit-test failed at step (1): the point inside the slug band landed
+on the slug cell. The popover case failed because the last row's panel was clipped. That became
+finding B. The final spec has three tests:
+- the header case, with the Sync station check added by fix A;
+- the popover case, moved to `rack-03`;
+- a `slow-a` scroll-into-view case, added by fix B.
+
+After the fixes: **3/3**. The spec does NOT exercise the popover-vs-thead tie: `rack-03`'s panel
+opens below the thead. That order is held by `rackLayout.test.ts`'s token check only.
+
+**Gates.**
+- tsc: clean after every commit.
+- common: **1727** (1723 − 1 rewritten + 5 new in `channelGroupSections.test.ts`); **1728** after review (the reviewer's walk).
+- editor unit (`tsx --test "app/**/*.test.ts"`): **69** (67 + 2 `rackLayout.test.ts`).
+- `test:scripts`: 156 + 1 skip.
+
+**e2e.** Every run was detached, from the worktree root:
+- baseline (audit + layers, pre-fix): 10 passed, 2 failed (the expected pair above), 1.0 min.
+- mid 1 (audit + layers + channel-groups): 20 passed, 1 failed (finding B), 1.3 min.
+- mid 2 (audit + layers): 13 passed, 2 failed. Both were clicks and measures before hydration. Fixed in the specs (`6ff63cbf`).
+- fix-C before-shot (isolate removed): 2/2, 0.3 min.
+- after (audit + layers): **15 passed, 0 failed**, 0.9 min.
+- after review (`channel-groups`, `channels-rack-layers`, `channel-priority`): **19 passed, 0 failed**, 1.1 min.
+- full list (`$T/p-specs.txt`, 24 files, every named file present, none dropped): **146 passed, 0 failed, 0 flaky, exit 0, 11.8 min**.
+
+**Numbers.** `phase3-view-numbers.ts`, primary's `transcripts/` and `settings.json`,
+read-only. The first before/after pair, hours apart, differed: nuxanor-kick went from 4 to 3
+untranscribed, hasanabi dropped out, digest eligibility went from 76,519 to 76,521. That is
+the live editor transcribing, not code. None of P's files is in the tool's import graph. Run
+back to back, main (primary checkout `77f63356` = `4130aca1` + one plan file) and the branch
+gave **diff empty, 5,119 bytes each**.
+
+**Builds** (at `6ff63cbf`): editor `next build` exit 0, 82 s, route table lists `ƒ /api/view/[name]`. Export `next build` exit 0, 54 s.
+
+**Left.**
+- The mobile thead does not pin (a documented trade-off).
+- `20ee34db`'s `min-w-52` on the tier cell is unchanged.
+- An opened Advanced panel lengthens the region's scroll extent while it is open, because it
+ is absolute inside the scroll box. Closing it restores the extent.
+
+### Slice W, as shipped — one write idiom (2026-09-24)
+
+Branch `one-core/phase-3-w` off `main` `4130aca1`, five commits, unmerged. Slice 4b left
+**16 JSON write sites in 13 files** on the per-pid temp name `${file}.tmp-${process.pid}`,
+plus the text and binary tmp + rename writers, plus two modules (`metadataScanStore`,
+`autoQueueState`) that had grown their own per-module-copy write counters to dodge the
+collision. All of them now go through `common/lib/jsonFile-server.ts`, byte-for-byte.
+
+| sha | what |
+|---|---|
+| `db9e3f44` | `writeFileAtomic(file, data: string \| Buffer, {mkdir, mode})` under `writeJsonAtomic` (now `jsonText` → `writeFileAtomic`): the same per-path chain on `globalThis`, the same `${file}.tmp-${pid}-${seq}-${random}` name. `mode` goes to `writeFile(tmp, data, {mode})`, i.e. onto the temp at creation, before the rename — the ordering `xSessionBroker` had. `copyFileAtomic(src, dest, {mkdir})` on the same chain (decided: the saved-video move folds rather than stays). No new sync twin. Tests: string / Buffer / empty bytes, mode 0o600 on the file and on every temp seen, one chain shared by the two writers on one path (40 interleaved writes, last issued lands), copy |
+| `9bfd15cd` | Race class: `rosterStore.writeRoster` (mkdir), `maybeMissingStore.writeMaybeMissing` (no mkdir), `metadataScanStore.writeMetadataScan` (no mkdir) on `writeJsonAtomic`. The counters `metadataScanStore.ts:188-191` and `autoQueueState.ts:141` deleted; `autoQueueState` on `writeJsonAtomic` (mkdir). New `autoQueueState.test.ts` "overlapping writes do not collide on the tmp file" (12 concurrent writes, last issued lands, no temp left); `metadataScanStore.test.ts:253` and `rosterStore.test.ts:184-186` unchanged and green |
+| `ae6ed9d2` | Process-global + per-video JSON: `syncSchedulerState`, `workerDefaults`, `widgetPresets`, `homepage` (mkdir `homepageDir` = the file's parent), `migrate-channel-priority`, `relocateDir.writeDirMarker`, `shard.saveShardConfig`, `duplicateShorts` ×2, `scanCorruptMedia` ×2, `backupSavedVideos` manifest, `normalizeLiveChat`, `normalizeTranscript`, `videoActions.ts` remark (`'{"transcription":[]}\n'` → `writeJsonAtomic(file, {transcription: []}, {indent: 0})`). Compact without newline (`{indent: 0, newline: false}`) for the two reports and the two cue files. mkdir exactly where a site had one. New `controller/compactJsonWriters.test.ts` (the four compact writers' bytes = `JSON.stringify` of their parse, no newline; passes on the parent too) and the literal pinned in `jsonFile-server.test.ts` |
+| `588fd7e9` | Text / binary: `failedTranscriptions` prune + clear, `runYtdlp.writePlaylistFile`, `xSessionBroker.writeCookieJar` (`{mkdir: true, mode: 0o600}`), `savedVideo-server.moveFileCrossDevice` (`copyFileAtomic`), `sites/lib/cutReleaseAction.ts`, `videoActions.ts` VTT promote. One comment on `storageWatch.ts`'s `let timer` (a per-copy singleton, not a temp name, one caller) |
+| `be4769de` | `plans/tools/phase3-writers-numbers.ts`, this record, the changelog bullet |
+| (review fixes) | record wording (188, writes-not-a-lock, `transcribeOne.ts:173`), changelog scope, the mode test made non-vacuous — see below |
+
+`videoActions.ts` changed at its two write sites and one added import line only — no export
+renamed, no signature changed (slice 3b owns its import list).
+
+**After commit 4,** `git grep -n 'tmp-${process.pid}' -- common editor`:
+
+```
+common/controller/buildIndex.ts:971: tmpPath = `${outPath}.tmp-${process.pid}`;
+common/controller/buildStats.ts:220: const tmp = `${outPath}.tmp-${process.pid}`;
+common/controller/transcode.ts:31: `audio.tmp-${process.pid}.${opts.targetFormat}`,
+common/controller/transcribeOne.ts:142: const tmpBase = `transcript.tmp-${process.pid}`;
+common/lib/jsonFile-server.test.ts:65: assert.ok(a.startsWith(`/x/config.json.tmp-${process.pid}-`));
+common/lib/jsonFile-server.ts:9:// of them spelling the temp file `${file}.tmp-${process.pid}`. That name is the
+common/lib/jsonFile-server.ts:136: return `${file}.tmp-${process.pid}-${seq}-${randomBytes(4).toString("hex")}`;
+```
+
+The two export page writers, as planned, **plus two the plan did not list**: `transcode.ts:31`
+and `transcribeOne.ts:142` are not writers of ours — they name the output file an external
+process (ffmpeg; whisper / chough) writes, which the code then renames. Nothing to fold; left
+and named. The last three hits are the shared writer itself, its history comment and its test.
+`git grep 'rename(tmp'` over `common editor` finds only `buildIndex`, `buildStats`, `transcode`.
+
+**Out of scope by name:** `buildIndex.ts:971` (the streaming `createWriteStream` page writer)
+and `buildStats.ts:220` (a hand-joined array) — restructuring, not a fold;
+`scripts/diarize.mjs`; everything under `umtool/`. And one the grep cannot see:
+`transcribeOne.ts:173` writes the remote transcript straight to `transcript.json`
+(`writeFile(transcriptPath, bytes)` — no temp, no rename), so a crash mid-write can leave it
+truncated; it never had the tmp idiom. A candidate for `writeFileAtomic` in a later slice. The module-level `storageWatch` timer and
+the TTL caches in `autoRunner.ts` / `recencyIndex.ts` are not temp names.
+
+**Behaviour changes (intended).** (1) Every folded write is chained per absolute path with every
+other `writeFileAtomic`/`writeJsonAtomic`/`copyFileAtomic` in the process, across module
+copies — the roster's writers in `runYtdlp`, `quickAvailabilityCheck` and the
+`pipelineActions` server action now have their WRITES serialised. That is not a lock: a
+`load → merge → writeRoster` from two actors can still lose one merge, exactly as before
+(the read-modify-write lock is `withJsonFileLock`, which these callers do not take). (2) A failed write removes its temp; the old
+code left it. (3) Temp names changed shape (`<file>.tmp-<pid>-<seq>-<hex>`); the remark's temp
+was `transcript.tmp-<pid>.json` and is now `transcript.json.tmp-…`. Nothing reads temp names.
+
+**Found and left: 188 orphan temps in the live corpus.** `transcripts/.auto-queue/` holds
+**175** `state.json.tmp-2514131-NNNN` files, all dated 2026-09-11 — the day `/home` hit 100 % —
+**173 of them 0 bytes** (two are partial, 16–20 KB): each a failed `writeFile` (ENOSPC) the old
+code never cleaned. Thirteen more elsewhere: seven `snapshot.json.tmp-<pid>`, three sidecar
+temps (`availability.json`, `download-outcome.json`) — all from pre-4b writers — and three
+`transcript.tmp-<pid>` whisper output bases (`transcribeOne`, an external process's file). The
+new writer cannot leave more of the first kind; the existing ones are corpus files and were
+**not touched** — the operator's to delete (`find transcripts -name '*.tmp-*'` lists all 188).
+
+**Numbers** (`plans/tools/phase3-writers-numbers.ts`, one process, never writes the corpus,
+never boots a server, clock frozen, only names present on both sides). Inputs frozen once
+(`FREEZE_TO`, 1,787 files) and both runs read the frozen tree: the parent `4130aca1` from a
+detached scratch worktree, the branch from `588fd7e9` + the tool. Per writer and sample it
+loads through the module's reader, writes through the module's writer into scratch, and prints
+the md5 of the bytes plus whether they equal the OLD idiom's bytes (`JSON.stringify(v, null,
+2) + "\n"`, or `JSON.stringify(v)` for the compact four). Coverage: 53 rosters, 52
+maybe-missing, 1 metadata-scan, scheduler, auto-queue (33,004 B), worker defaults, widget
+presets, homepage, the duplicates report (empty scan), the media-scan report (a merge of the
+live 27,756 B report), a relocation marker (synthetic — none live), the backup manifest, 100
+`transcript.cues.json` and 100 `live_chat.cues.json` re-normalized. 323 lines each side,
+**315 `old-idiom=same`, 0 DIFF, 0 THREW, 0 leftover temps; `diff` before/after empty**
+(`w-numbers-{before,after}.txt`). Not measurable without a live sample: shard configs (none
+live), duplicate overrides (the live file has no clusters), media-scan overrides (no file),
+the priority migration (a CLI whose only write is the default-options `writeJsonAtomic`,
+pinned by `jsonText`'s tests) — and the remark literal, pinned by a unit test.
+
+**Gates** (worktree `one-core-phase-3-w`, ports 3301/3311/3310/3320):
+- `pnpm -r --no-bail --workspace-concurrency=1 exec tsc --noEmit` — clean before every commit.
+- common **1733/1733** (1723 + 4 `writeFileAtomic`/`copyFileAtomic` + 1 literal + 1
+ `autoQueueState` overlap + 4 compact-writer bytes); editor unit (`tsx --test
+ "app/**/*.test.ts"`) **67/67**; `test:scripts` **156 + 1 skip** (a first run while this
+ slice's own e2e held the machine lock failed the queue-lock banner test — the lock was
+ taken; re-run with it free, green); mcp **219/219**.
+- `pnpm --filter editor exec next build` exit 0, `ƒ /api/view/[name]` in the table;
+ `pnpm --filter export exec next build` exit 0; `ZodError|_zod` over both `.next/static`: 0
+ files each.
+- e2e, the prompt's 20 specs (all exist): **140 passed, 0 failed, 10.7 min**.
+
+**Review fixes** (review verdict: ship after fixes; four nits, none blocking). The heading's
+orphan count is corrected from 173 to 188, the body's total. The record now says the roster
+writers' writes are serialised but a load-merge-write is not locked, and lists
+`transcribeOne.ts:173` as out of scope. The changelog no longer says "every file": it names
+the four temp names that stay. The mode test (`jsonFile-server.test.ts`) could pass vacuously,
+because a `fs.watch` could see no temp. It now intercepts `rename` (patched on
+`node:fs/promises` + `syncBuiltinESMExports`) and stats the temp at that moment, asserting
+exactly one temp, already 0o600. With `mode` removed from `writeFileAtomic` it fails.
+Gates after: tsc clean, common **1733/1733**, editor unit **67/67**. No runtime code changed,
+so the builds, e2e and numbers were not re-run.
+
+**Merged with `main` `77a32de2` (slice P) as `44dd843f`**. The merge conflicted only in
+`editor/CHANGELOG.md` and this file, and both sides were kept, P then W. Gates on the merged
+tip:
+- tsc clean.
+- common **1738/1738**: W's 1733 plus P's 5.
+- editor unit **69/69**: 67 plus P's 2.
+- `test:scripts` **156 + 1 skip**, run once the e2e lock was free.
+- mcp **219/219**.
+- Editor `next build` exit 0, with `ƒ /api/view/[name]` in the route table. Export
+ `next build` exit 0. No `ZodError|_zod` in either `.next/static`.
+- Numbers tool: inputs re-frozen (1,787 files). The before run is `main` `77a32de2` (the old
+ writers), from a detached scratch worktree; the after run is `44dd843f`. 323 lines each,
+ 315 `old-idiom=same`, 0 DIFF/THREW/LEFT, **diff empty** (`w-m-numbers-{before,after}.txt`).
+- e2e, the same 20 specs, on ports 3411/3410: **140 passed, 0 failed, 8.8 min**.
+
### Slice 3b, as shipped — the video page's chore cards, one module each (2026-09-24)
Branch `one-core/phase-3-s3b` off `main` `4130aca1`, one code commit and this record.
diff --git a/plans/rumble-sweep-pacing.md b/plans/rumble-sweep-pacing.md
@@ -0,0 +1,79 @@
+# Plan — a full sweep of a large Rumble channel must be paced, or it never completes
+
+**Found 2026-09-24** (operator: "job `01M3AVGZC5EZ9GCD04R9QW6NWX` keeps trying and failing a full
+playlist fetch … no sync has happened in 44 days"). Facts verified on the live corpus and
+`4130aca1`:
+
+- `the-quartering-rumble` (`https://rumble.com/c/TheQuartering`, 7,866 listed / 7,871 archived)
+ has `lastFullSweepAt: null`, so `isFullSweepDue` (`common/jobs/deepSync.ts:33-40`) is true on
+ every sync and `sync()` (`common/ytdlp/runYtdlp.ts:1333-1336`) always takes `syncFullSweep`,
+ never `syncPaged`.
+- The sweep's one `--flat-playlist` spawn walks the channel's listing pages back to back; Rumble
+ answers **HTTP 429 on page 155** (job log line 173). The patched extractor
+ (`~/Projects/yt-dlp-patched`, the pipx editable install the editor runs) logs the error and
+ finishes with 3,846 entries, but yt-dlp exits 1; `enumeratePlaylistUrls`
+ (`runYtdlp.ts:352-361`) throws on any exit other than 0/101 and discards stdout. Nothing stamps
+ `lastSyncedAt` (2026-08-11) or `lastFullSweepAt`; the next sync is a sweep again. Three
+ attempts on 2026-09-24 (16:05, 20:20, 23:20 UTC), identical.
+- **Tolerating exit 1 is the wrong fix.** The 3,846-entry listing is half the channel. The shrink
+ guard (`common/controller/acceptListing.ts`, `fullSweepShrinkGuardPercent` 10 %, floor 25)
+ would reject it once, and a second identical partial read — the same 429 at the same page —
+ would *confirm* a mass deletion and truncate the stored `playlist`. The throw is what prevents
+ that today.
+
+**Step 1, applied by the operator 2026-09-24 (no code):** `fullSweepIntervalMinutes: 0` on the
+channel, through `pnpm ops channel-config` (the `patchChannelConfig` writer), so every sync is
+the paged walk (`--lazy-playlist -I start:end`, a few newest pages, stops at the first archived
+hit) and new videos arrive on the next cadence. Cost until step 2: no maybe-missing detection for
+this channel.
+
+**Step 2, a small slice after one-core Phase 3 release 4** (it edits `runYtdlp.ts`, in slice W's
+file set, so it lands after W merges):
+
+1. **Pace the enumeration for Rumble.** Add `--sleep-requests <s>` to the sweep spawn when the
+ channel's platform is Rumble (a per-platform table beside `downloadQueueKey`, not a per-channel
+ knob; the value is an operator setting under `syncScheduler`, default ~1 s). Pacing applies to
+ every page request, which is exactly where the 429 comes from.
+2. **A sweep that ends on a 429 is "incomplete", not "failed" and not "a listing".** Detect the
+ 429 in stderr (or the non-zero exit with a partial stdout) and record it on the roster as an
+ incomplete observation: it must not enter `acceptListing` as a reading (so two identical
+ partial walks can never confirm a deletion), and it must not stamp `lastFullSweepAt`. Log one
+ line naming the page it reached.
+3. **Do not re-sweep on a cooldown.** After an incomplete sweep, the next syncs are paged walks
+ until a per-platform cooldown (reuse the rate-limit cooldown the download runner already keeps
+ for YouTube 429s, keyed `platform:rumble`) has elapsed — today a failed sweep is retried at
+ every cadence, each one another 155-page walk that keeps the limit tripped.
+4. **Tests:** unit for the incomplete verdict (partial + 429 → no `acceptListing` call, no
+ stamp), the cooldown gate, and the pacing arg on a Rumble channel only; the `runYtdlp` e2e
+ fixtures with a fake yt-dlp that exits 1 after N pages.
+5. **Rollout:** after the slice is live, restore the channel's `fullSweepIntervalMinutes` (delete
+ the override) and watch one paced sweep complete; the first complete listing after 44 days may
+ legitimately shrink — the two-observation guard handles that as designed.
+
+## Found next, 2026-09-24 19:36 — Rumble's embed endpoint answers 403 to every download
+
+Step 1 worked: job `01M3AWDQ4RZD6CV93BJPREQVPX` ran the paged walk (`--lazy-playlist -I 1:50`),
+listed the newest page and found new videos. The first download died on
+`[RumbleEmbed] … Unable to download JSON metadata: HTTP Error 403: Forbidden`
+(`https://rumble.com/embedJS/u3/`, `rumble.py:168`), the managed download aborted as "network",
+and the sync failed on that one video. **It is not that video and not the patch:** the
+`leaflit-rumble` and `rekietalaw-rumble` syncs at 16:05 UTC hit the same 403, and no retained
+job log holds a successful Rumble archive line. Upstream: yt-dlp issue #17496 (opened 2026-08-20,
+same error, same `2026.08.19`, labelled *impersonation* + *site-bug*, no fix), after #15089 /
+#15148 / #15129 (Rumble behind Cloudflare, browser-only fingerprints); a June report says
+`--impersonate chrome` worked for a while. The editor never passes `--impersonate`
+(`grep -rn impersonate common editor` is empty) although the pipx venv has `curl_cffi` targets
+(Chrome-133/136, Safari-18).
+
+**Step 0 for the slice, ahead of pacing:** a per-platform yt-dlp args table (Rumble →
+`--impersonate chrome`, and the pacing from step 2) applied in `configArgs`
+(`common/ytdlp/runYtdlp.ts:221`) so every Rumble spawn — enumeration, metadata probe, download —
+carries it; `downloadQueueKey` (`common/lib/queueKeys.ts:69`) already resolves the platform.
+Verify first by hand, ONCE, metadata only (operator's rule: fetches go through Archilyzer;
+this is the ask-first exception):
+`yt-dlp --impersonate chrome --skip-download --print title https://rumble.com/v7fxzsc-lindsay-clancy-insane-twist-on-the-view-these-people-are-nuts.html`.
+If that also 403s, Rumble is closed to this client until upstream moves and the slice waits;
+the channels then need a "source unreachable" outcome rather than a failed sync per cadence.
+
+Out of scope: tolerating exit 1 generally; changing the shrink guard; anything in the patched
+extractor (the pagination fix is in and working; the 429 is Rumble's rate limit, not a bug).
diff --git a/plans/site-exports-off.md b/plans/site-exports-off.md
@@ -0,0 +1,51 @@
+# Plan — visitor exports off on the published sites, kept as an option for the OSS release
+
+**Asked 2026-09-24** (operator, mid release 4): turn off "exports" on every existing published
+site to strengthen the legal footing of the operator's own instances, while keeping the
+capability as a per-site option so anyone running the OSS release can leave it on. Scheduled as
+the release AFTER one-core Phase 3 release 4 (it needs a rebuild + deploy of the five published
+sites, and Phase 3's slices are already cut). Inventory below was taken on `4130aca1`
+(2026-09-24, one Explore pass); re-verify the path:line anchors before cutting the slice.
+
+## What a visitor can export from a published site today
+
+| Surface | Where | Switch today |
+|---|---|---|
+| Downloads page: whole-channel `.zip` of transcripts + live chat, its nav and footer links | `export/app/downloads/page.tsx`, `export/app/components/Header.tsx:39,49`, `Footer.tsx:38,44-50`; zips built by `common/bin/build-archives.ts` | `site.archives` (`common/lib/siteSchema.ts:86`, SITE.md), absent = **on** |
+| Offline / PWA whole-channel cache in the browser | `export/app/components/OfflineManager.tsx`, `export/app/offline/page.tsx` | `site.pwa`, default **off** |
+| Duplicates page (bulk cluster listing) | `export/app/duplicates/**` | `site.duplicates`, absent = on |
+| Per-video **Download** menu (txt / srt / json), **Copy MD**, **Copy download command** (a `yt-dlp --download-sections` string; the site never serves media) | `common/components/TranscriptModal.tsx:405-500`, impl `common/components/PlayerProvider.tsx:480-572` — client-side from cues already in the browser | **none** — unconditional on every build |
+| Machine contract: `/corpus.json`, `/llms.txt`, tree manifests, `page-<NNNN>.json` shards, `robots.txt`, `sitemap.xml` | written by `common/bin/compose-site.ts:158-185`; URL contract `common/lib/archive/contract.ts` | none, and must **stay unconditional**: the MCP server (`common/lib/archive/reader.ts` `RemoteSource`, `mcp/src/sourceRegistry.ts`) and `umtool/report-to-video/cues.mjs` walk it over HTTP — it is the "no local corpus" research mode `AGENTS.md` promises |
+
+None of the six `transcripts/sites/*/site.json` files sets `archives`, `pwa` or `duplicates`
+today, so every published site ships the zip Downloads page and the in-modal buttons.
+
+## The slice
+
+1. **One new `site.json` key**, `transcriptDownloads` (name to settle at planning; boolean,
+ absent = **on**, so the OSS default is unchanged and the operator opts out per site), declared
+ in `common/lib/siteSchema.ts` beside `archives` and regenerated into `SITE.md`
+ (`file-schemas-docs.ts --check` stays green). It gates the three in-modal actions (Download
+ menu, Copy MD, Copy download command) on the export site only. `TranscriptModal` is shared with
+ the editor, so the gate is a prop the export site derives from `currentSite()`
+ (`export/app/lib/site.ts`); the editor keeps every button.
+2. **`archives: false`** on the operator's sites — the existing switch; no code.
+3. Nothing changes for `/corpus.json`, `llms.txt`, manifests or shards. The `llms.txt` prose and
+ `use-with-ai` page (`export/app/use-with-ai/page.tsx:128-136`) already hide the Downloads link
+ when archives are off; check the prose does not still promise zips.
+4. **Site form + `pnpm ops`**: the two new/used keys are set through `writeSite` (the site's
+ Publish tab or `pnpm ops site-config`), never by editing `site.json` by hand.
+5. **Rollout**: for each of the five published sites (jeralyzer, rekietalyzer, hasanalyzer,
+ anilyzer, bonnellyzer; jasolyzer has no `siteUrl`), set `archives: false` +
+ `transcriptDownloads: false`, then `build-site` + `deploy-site` — five independent runs; the
+ archives volume can be pruned afterwards. Verify per site: no `/downloads` link, no Download menu
+ in a transcript modal, `/corpus.json` and one `page-0001.json` still 200.
+6. **Tests**: export e2e for the gated modal (buttons present with the key absent, absent with it
+ false), the existing archives-off coverage, a schema round-trip in `phase3-files-numbers.ts`
+ (unknown-key check must not flag the new key on either side).
+
+Out of scope, stated: gating the machine contract (breaks the MCP and report-to-video), removing
+"Copy share link" (not an export), umtool's clip fetching (operator tooling, not visitor-facing).
+
+Open for the operator: whether the Duplicates page counts as an export (it lists, it does not
+serve data) and whether the hub (`_homepage`) needs the same key.
diff --git a/plans/tools/phase3-writers-numbers.ts b/plans/tools/phase3-writers-numbers.ts
@@ -0,0 +1,439 @@
+#!/usr/bin/env tsx
+// The one-core Phase 3 release 4 slice W measurement: the bytes every folded
+// tmp + rename JSON writer puts on disk, over live samples — printed
+// deterministically so a run on the parent commit and a run on the branch can
+// be diffed. Slice W moved these writers onto lib/jsonFile-server.ts and
+// claims it changed no byte; an empty diff is that claim.
+//
+// Model: phase3-files-numbers.ts next door, and the same rules.
+//
+// NEVER WRITES THE CORPUS. Every input is COPIED into a scratch tree under
+// os.tmpdir() first; every read-for-measurement and every write happens on the
+// copy, and the scratch tree is deleted at the end. The live corpus is only
+// ever read (readdir, stat, readFile, copyFile source).
+//
+// NEVER BOOTS A SERVER. The writers are called in-process; instrumentation.ts
+// is not loaded, so nothing is armed.
+//
+// ONE PROCESS. `TRANSCRIPTS_DIR` is pointed at the scratch tree BEFORE any
+// module that memoises `getPaths()` is imported.
+//
+// THE CLOCK IS FROZEN. Several writers stamp `new Date()` into what they write
+// (a decision's `decidedAt`, a report's `generatedAt`); `Date` is replaced with
+// a subclass whose "now" is fixed, so two runs write the same bytes.
+//
+// USES ONLY NAMES PRESENT ON BOTH SIDES of slice W (the modules' exported
+// readers and writers, unchanged by the slice), so the SAME file runs on the
+// parent and on the branch.
+//
+// Per writer and sample it prints the md5 of the bytes written and whether they
+// equal the OLD idiom's bytes for the same value — `JSON.stringify(v, null, 2)
+// + "\n"` or, for the compact four, `JSON.stringify(v)` — where v is the parse
+// of what was written. The old-idiom column pins the FORMAT on each side; the
+// before/after diff of the md5 column pins the CONTENT.
+//
+// Usage, from the repo root:
+// LIVE_TRANSCRIPTS_DIR=/abs/transcripts node_modules/.bin/tsx plans/tools/phase3-writers-numbers.ts > out.txt
+// Default LIVE_TRANSCRIPTS_DIR: the primary checkout's, a sibling of this repo.
+// SAMPLE_N (default 100): video dirs sampled per cue file.
+//
+// FREEZING THE INPUTS. The live editor rewrites the scheduler and auto-queue
+// state, rosters and maybe-missing records all day, so a parent run and a
+// branch run minutes apart would differ for reasons that are not this code.
+// `FREEZE_TO=/abs/dir` copies exactly what a measurement reads, in the same
+// layout, into that directory and exits; both runs then take
+// `LIVE_TRANSCRIPTS_DIR=/abs/dir`, and a frozen tree samples itself.
+
+import { createHash } from "node:crypto";
+import fs from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const REPO = path.resolve(HERE, "..", "..");
+const LIVE =
+ process.env.LIVE_TRANSCRIPTS_DIR ??
+ path.join(path.dirname(REPO), "yt-dlp-transcript-browser", "transcripts");
+const SAMPLE_N = Number(process.env.SAMPLE_N ?? 100);
+// A live-chat file can be hundreds of MB; the sample takes small ones.
+const LIVE_CHAT_MAX_BYTES = 4 * 1024 * 1024;
+
+const SCRATCH = fs.mkdtempSync(path.join(os.tmpdir(), "phase3-writers-"));
+process.env.TRANSCRIPTS_DIR = SCRATCH;
+process.env.TZ = "UTC";
+
+const FIXED_NOW = Date.UTC(2026, 8, 24, 12, 0, 0);
+const RealDate = Date;
+class FixedDate extends RealDate {
+ constructor(...args: unknown[]) {
+ if (args.length === 0) super(FIXED_NOW);
+ else super(...(args as [number]));
+ }
+ static now(): number {
+ return FIXED_NOW;
+ }
+}
+globalThis.Date = FixedDate as DateConstructor;
+
+type Format = "pretty" | "compact";
+
+function md5(b: string | Buffer): string {
+ return createHash("md5").update(b).digest("hex");
+}
+
+function oldIdiom(v: unknown, format: Format): string {
+ return format === "compact" ? JSON.stringify(v) : JSON.stringify(v, null, 2) + "\n";
+}
+
+function sortedDir(dir: string): string[] {
+ try {
+ return fs.readdirSync(dir).sort();
+ } catch {
+ return [];
+ }
+}
+
+// Copy a live file (path relative to the corpus root) into the scratch tree.
+function stage(rel: string): boolean {
+ const src = path.join(LIVE, rel);
+ if (!fs.existsSync(src)) return false;
+ const dest = path.join(SCRATCH, rel);
+ fs.mkdirSync(path.dirname(dest), { recursive: true });
+ fs.copyFileSync(src, dest);
+ return true;
+}
+
+// Report the file a writer just wrote.
+function report(label: string, file: string, format: Format): void {
+ if (!fs.existsSync(file)) {
+ console.log(`${label} ABSENT`);
+ return;
+ }
+ const bytes = fs.readFileSync(file);
+ let idiom: string;
+ try {
+ idiom = oldIdiom(JSON.parse(bytes.toString("utf8")), format) === bytes.toString("utf8")
+ ? "old-idiom=same"
+ : "old-idiom=DIFF";
+ } catch {
+ idiom = "old-idiom=UNPARSEABLE";
+ }
+ const leftovers = sortedDir(path.dirname(file)).filter((n) =>
+ n.startsWith(`${path.basename(file)}.tmp-`),
+ );
+ console.log(
+ `${label} ${bytes.length}B md5=${md5(bytes)} ${idiom}` +
+ (leftovers.length ? ` LEFT ${leftovers.length} temp(s)` : ""),
+ );
+}
+
+async function run(label: string, fn: () => Promise<void>): Promise<void> {
+ try {
+ await fn();
+ } catch (e) {
+ console.log(`${label} THREW: ${(e as Error).message}`);
+ }
+}
+
+async function channelStores(slugs: string[]): Promise<void> {
+ const { getPaths } = await import("../../common/lib/paths");
+ const roster = await import("../../common/controller/rosterStore");
+ const mm = await import("../../common/controller/maybeMissingStore");
+ const ms = await import("../../common/controller/metadataScanStore");
+ const shard = await import("../../common/controller/shard");
+ const paths = getPaths();
+
+ console.log("# per-channel stores (every live file)");
+ for (const slug of slugs) {
+ const ch = path.join("channels", slug);
+ if (stage(path.join(ch, roster.ROSTER_FILENAME))) {
+ await run(`roster ${slug}`, async () => {
+ await roster.writeRoster(paths, slug, await roster.loadRoster(paths, slug));
+ report(`roster ${slug}`, roster.rosterPath(paths, slug), "pretty");
+ });
+ }
+ if (stage(path.join(ch, mm.MAYBE_MISSING_FILENAME))) {
+ await run(`maybe-missing ${slug}`, async () => {
+ const rec = await mm.loadMaybeMissing(paths, slug);
+ if (!rec) return void console.log(`maybe-missing ${slug} load=null`);
+ await mm.writeMaybeMissing(paths, slug, rec);
+ report(
+ `maybe-missing ${slug}`,
+ path.join(paths.channelsDir, slug, mm.MAYBE_MISSING_FILENAME),
+ "pretty",
+ );
+ });
+ }
+ if (stage(path.join(ch, ms.METADATA_SCAN_FILENAME))) {
+ await run(`metadata-scan ${slug}`, async () => {
+ // The writer is private; an upsert that changes one entry's title by a
+ // fixed suffix forces exactly one write of the whole store.
+ const scan = await ms.loadMetadataScan(paths, slug);
+ const id = Object.keys(scan.entries).sort()[0];
+ const upsert = id
+ ? { entries: { [id]: { ...scan.entries[id], title: `${scan.entries[id].title}·` } } }
+ : {};
+ await ms.upsertMetadataScan(paths, slug, upsert, "2026-09-24T12:00:00.000Z");
+ report(`metadata-scan ${slug}`, ms.metadataScanPath(paths, slug), "pretty");
+ });
+ }
+ for (const op of shard.SHARD_OPS) {
+ const rel = path.relative(SCRATCH, shard.shardFile(paths, slug, op));
+ if (!stage(rel)) continue;
+ await run(`shard-${op} ${slug}`, async () => {
+ const cfg = await shard.loadShardConfig(paths, slug, op);
+ if (!cfg) return void console.log(`shard-${op} ${slug} load=null`);
+ await shard.saveShardConfig(paths, slug, op, cfg);
+ report(`shard-${op} ${slug}`, shard.shardFile(paths, slug, op), "pretty");
+ });
+ }
+ }
+}
+
+async function globalStores(): Promise<void> {
+ const { getPaths } = await import("../../common/lib/paths");
+ const paths = getPaths();
+ const rel = (abs: string) => path.relative(SCRATCH, abs);
+ console.log("");
+ console.log("# process-global stores");
+
+ const sched = await import("../../common/jobs/syncSchedulerState");
+ if (stage(rel(paths.schedulerStateFile))) {
+ await run("scheduler", async () => {
+ await sched.writeSchedulerState(paths, await sched.readSchedulerState(paths));
+ report("scheduler", paths.schedulerStateFile, "pretty");
+ });
+ }
+ const aq = await import("../../common/jobs/autoQueueState");
+ if (stage(rel(paths.autoQueueStateFile))) {
+ await run("auto-queue", async () => {
+ await aq.writeAutoQueueState(paths, await aq.readAutoQueueState(paths));
+ report("auto-queue", paths.autoQueueStateFile, "pretty");
+ });
+ }
+ const wd = await import("../../common/jobs/workerDefaults");
+ if (stage(rel(paths.workerDefaultsFile))) {
+ await run("worker-defaults", async () => {
+ const d = wd.readWorkerDefaults(paths);
+ if (!d) return void console.log("worker-defaults load=null");
+ await wd.writeWorkerDefaults(paths, d.enabledWorkerIds);
+ report("worker-defaults", paths.workerDefaultsFile, "pretty");
+ });
+ }
+ const wp = await import("../../common/lib/widgetPresets");
+ if (stage(rel(paths.widgetPresetsFile))) {
+ await run("widget-presets", async () => {
+ const list = await wp.readWidgetPresets(paths);
+ if (list.length === 0) return void console.log("widget-presets empty");
+ // Saving over an existing name rewrites the whole list, unchanged.
+ await wp.addWidgetPreset(paths, list[0].name, list[0].query);
+ report("widget-presets", paths.widgetPresetsFile, "pretty");
+ });
+ }
+ const hp = await import("../../common/lib/homepage");
+ if (stage(rel(paths.homepageConfigFile))) {
+ await run("homepage", async () => {
+ await hp.writeHomepageConfig(hp.getHomepageConfig(paths), paths);
+ report("homepage", paths.homepageConfigFile, "pretty");
+ });
+ }
+
+ const dup = await import("../../common/controller/duplicateShorts");
+ if (stage(rel(dup.duplicateOverridesPath(paths)))) {
+ await run("duplicate-overrides", async () => {
+ const cur = await dup.readDuplicateOverrides(paths);
+ const id = Object.keys(cur.clusters).sort()[0];
+ if (!id) return void console.log("duplicate-overrides empty");
+ const c = cur.clusters[id] as Record<string, unknown>;
+ await dup.updateDuplicateOverride(paths, id, {
+ ...(typeof c.canonicalSlug === "string" ? { canonicalSlug: c.canonicalSlug } : {}),
+ ...(typeof c.notDuplicate === "boolean" ? { notDuplicate: c.notDuplicate } : {}),
+ ...(typeof c.confirmed === "boolean" ? { confirmed: c.confirmed } : {}),
+ ...(typeof c.note === "string" ? { note: c.note } : {}),
+ });
+ report("duplicate-overrides", dup.duplicateOverridesPath(paths), "pretty");
+ });
+ }
+ // The report writer runs at the end of a detection pass; over the scratch
+ // tree (configs and stores, no data/) that pass sees no videos.
+ await run("duplicates-report", async () => {
+ await dup.detectDuplicateShorts({ paths, onLog: () => {} });
+ report("duplicates-report (no videos)", path.join(paths.transcriptsDir, "duplicates.json"), "compact");
+ });
+
+ const scm = await import("../../common/controller/scanCorruptMedia");
+ if (stage(rel(scm.mediaScanOverridesPath(paths)))) {
+ await run("media-scan-overrides", async () => {
+ const cur = await scm.readMediaScanOverrides(paths);
+ const key = Object.keys(cur.reviewed).sort()[0];
+ if (!key) return void console.log("media-scan-overrides empty");
+ const note = (cur.reviewed[key] as { note?: string }).note;
+ await scm.updateMediaScanOverride(paths, key, { reviewed: true, ...(note ? { note } : {}) });
+ report("media-scan-overrides", scm.mediaScanOverridesPath(paths), "pretty");
+ });
+ }
+ // A scan of a channel that does not exist merges the live report's every
+ // finding back unchanged, through the report writer.
+ if (stage("media-scan.json")) {
+ await run("media-scan-report", async () => {
+ await scm.scanCorruptMedia({ paths, channels: ["__phase3-writers-none__"], onLog: () => {} });
+ report("media-scan-report (merge of live)", path.join(paths.transcriptsDir, "media-scan.json"), "compact");
+ });
+ }
+
+ const rd = await import("../../common/controller/relocateDir");
+ // No live marker exists outside a move; a fixed one exercises the writer.
+ await run("relocation-marker", async () => {
+ const file = path.join(SCRATCH, "channels", ".phase3-writers-marker.json");
+ fs.mkdirSync(path.dirname(file), { recursive: true });
+ await rd.writeDirMarker(file, {
+ target: "/mnt/x/slug/data",
+ direction: "out",
+ startedAt: "2026-09-24T12:00:00.000Z",
+ phase: "copy",
+ });
+ report("relocation-marker (synthetic)", file, "pretty");
+ });
+
+ const bk = await import("../../common/controller/backupSavedVideos");
+ await run("backup-manifest", async () => {
+ const dest = path.join(SCRATCH, ".phase3-writers-backup");
+ const r = await bk.backupSavedVideos({ paths, dest, onLog: () => {} });
+ report(`backup-manifest (${r.entries} saved)`, r.manifestPath, "pretty");
+ });
+}
+
+// The first SAMPLE_N video dirs (sorted slugs × sorted ids) holding `filename`.
+function sampleDirs(slugs: string[], filename: string, maxBytes?: number): string[] {
+ const out: string[] = [];
+ for (const slug of slugs) {
+ const data = path.join(LIVE, "channels", slug, "data");
+ for (const id of sortedDir(data)) {
+ if (out.length >= SAMPLE_N) return out;
+ const f = path.join(data, id, filename);
+ try {
+ const st = fs.statSync(f);
+ if (maxBytes !== undefined && st.size > maxBytes) continue;
+ out.push(path.join(data, id));
+ } catch {
+ // not in this dir
+ }
+ }
+ }
+ return out;
+}
+
+const MEDIA = /\.(opus|m4a|mp3|webm|mp4|mkv|wav|ogg|aac|flac|part|ytdl|jpg|jpeg|png|webp)$/i;
+
+// Copy a video dir's non-media files (metadata, raw transcripts, the prior cue
+// file whose format tag a re-normalize reads) into the scratch tree.
+function stageVideoDir(live: string, n: number): string {
+ const dir = path.join(SCRATCH, "videos", String(n));
+ fs.mkdirSync(dir, { recursive: true });
+ for (const name of sortedDir(live)) {
+ if (MEDIA.test(name)) continue;
+ const src = path.join(live, name);
+ if (!fs.statSync(src).isFile()) continue;
+ fs.copyFileSync(src, path.join(dir, name));
+ }
+ return dir;
+}
+
+async function cueWriters(slugs: string[]): Promise<void> {
+ const nt = await import("../../common/controller/normalizeTranscript");
+ const nl = await import("../../common/controller/normalizeLiveChat");
+ const vs = await import("../../common/lib/videoStatus");
+ let n = 0;
+ console.log("");
+ console.log(`# ${vs.CUES_JSON_FILENAME} (first ${SAMPLE_N} dirs holding one, force re-normalize)`);
+ for (const live of sampleDirs(slugs, vs.CUES_JSON_FILENAME)) {
+ const rel = path.relative(path.join(LIVE, "channels"), live);
+ const dir = stageVideoDir(live, n++);
+ await run(rel, async () => {
+ const out = await nt.normalizeTranscript({ videoDir: dir, channelSlug: "x", force: true } as never);
+ if ((out as { status: string }).status !== "wrote") {
+ return void console.log(`${rel} ${(out as { status: string }).status}`);
+ }
+ report(rel, path.join(dir, vs.CUES_JSON_FILENAME), "compact");
+ });
+ }
+ console.log("");
+ console.log(`# ${vs.LIVE_CHAT_CUES_FILENAME} (first ${SAMPLE_N} dirs with a live chat ≤ 4 MB)`);
+ for (const live of sampleDirs(slugs, vs.LIVE_CHAT_FILENAME, LIVE_CHAT_MAX_BYTES)) {
+ const rel = path.relative(path.join(LIVE, "channels"), live);
+ const dir = stageVideoDir(live, n++);
+ await run(rel, async () => {
+ const out = await nl.normalizeLiveChat({ videoDir: dir, channelSlug: "x", force: true });
+ if (out.status !== "wrote") return void console.log(`${rel} ${out.status}`);
+ report(rel, path.join(dir, vs.LIVE_CHAT_CUES_FILENAME), "compact");
+ });
+ }
+}
+
+// The files the measurement reads, relative to the corpus root.
+async function inputs(slugs: string[]): Promise<string[]> {
+ const { getPaths } = await import("../../common/lib/paths");
+ const shard = await import("../../common/controller/shard");
+ const dup = await import("../../common/controller/duplicateShorts");
+ const scm = await import("../../common/controller/scanCorruptMedia");
+ const vs = await import("../../common/lib/videoStatus");
+ const paths = getPaths();
+ const rel = (abs: string) => path.relative(SCRATCH, abs);
+ const out: string[] = [];
+ for (const s of slugs) {
+ const ch = path.join("channels", s);
+ out.push(
+ path.join(ch, "config.json"),
+ path.join(ch, "roster.json"),
+ path.join(ch, "maybe-missing.json"),
+ path.join(ch, "metadata-scan.json"),
+ ...shard.SHARD_OPS.map((op) => rel(shard.shardFile(paths, s, op))),
+ );
+ }
+ out.push(
+ rel(paths.schedulerStateFile),
+ rel(paths.autoQueueStateFile),
+ rel(paths.workerDefaultsFile),
+ rel(paths.widgetPresetsFile),
+ rel(paths.homepageConfigFile),
+ rel(dup.duplicateOverridesPath(paths)),
+ rel(scm.mediaScanOverridesPath(paths)),
+ "media-scan.json",
+ );
+ const dirs = [
+ ...sampleDirs(slugs, vs.CUES_JSON_FILENAME),
+ ...sampleDirs(slugs, vs.LIVE_CHAT_FILENAME, LIVE_CHAT_MAX_BYTES),
+ ];
+ for (const d of dirs) {
+ for (const name of sortedDir(d)) {
+ if (MEDIA.test(name) || !fs.statSync(path.join(d, name)).isFile()) continue;
+ out.push(path.relative(LIVE, path.join(d, name)));
+ }
+ }
+ return [...new Set(out)].filter((r) => fs.existsSync(path.join(LIVE, r)));
+}
+
+try {
+ const slugs = sortedDir(path.join(LIVE, "channels")).filter((s) =>
+ fs.existsSync(path.join(LIVE, "channels", s, "config.json")),
+ );
+ if (process.env.FREEZE_TO) {
+ const to = path.resolve(process.env.FREEZE_TO);
+ const files = await inputs(slugs);
+ for (const r of files) {
+ fs.mkdirSync(path.dirname(path.join(to, r)), { recursive: true });
+ fs.copyFileSync(path.join(LIVE, r), path.join(to, r));
+ }
+ console.error(`froze ${files.length} files into ${to}`);
+ fs.rmSync(SCRATCH, { recursive: true, force: true });
+ process.exit(0);
+ }
+ // Configs first: several writers read the channel list.
+ for (const s of slugs) stage(path.join("channels", s, "config.json"));
+ await channelStores(slugs);
+ await globalStores();
+ await cueWriters(slugs);
+} finally {
+ fs.rmSync(SCRATCH, { recursive: true, force: true });
+}