#!/usr/bin/env tsx // The one-core Phase 3 release 4 slice W measurement: the bytes every folded // tmp + rename JSON writer puts on disk, over live samples — printed // deterministically so a run on the parent commit and a run on the branch can // be diffed. Slice W moved these writers onto lib/jsonFile-server.ts and // claims it changed no byte; an empty diff is that claim. // // Model: phase3-files-numbers.ts next door, and the same rules. // // NEVER WRITES THE CORPUS. Every input is COPIED into a scratch tree under // os.tmpdir() first; every read-for-measurement and every write happens on the // copy, and the scratch tree is deleted at the end. The live corpus is only // ever read (readdir, stat, readFile, copyFile source). // // NEVER BOOTS A SERVER. The writers are called in-process; instrumentation.ts // is not loaded, so nothing is armed. // // ONE PROCESS. `TRANSCRIPTS_DIR` is pointed at the scratch tree BEFORE any // module that memoises `getPaths()` is imported. // // THE CLOCK IS FROZEN. Several writers stamp `new Date()` into what they write // (a decision's `decidedAt`, a report's `generatedAt`); `Date` is replaced with // a subclass whose "now" is fixed, so two runs write the same bytes. // // USES ONLY NAMES PRESENT ON BOTH SIDES of slice W (the modules' exported // readers and writers, unchanged by the slice), so the SAME file runs on the // parent and on the branch. // // Per writer and sample it prints the md5 of the bytes written and whether they // equal the OLD idiom's bytes for the same value — `JSON.stringify(v, null, 2) // + "\n"` or, for the compact four, `JSON.stringify(v)` — where v is the parse // of what was written. The old-idiom column pins the FORMAT on each side; the // before/after diff of the md5 column pins the CONTENT. // // Usage, from the repo root: // LIVE_TRANSCRIPTS_DIR=/abs/transcripts node_modules/.bin/tsx plans/tools/phase3-writers-numbers.ts > out.txt // Default LIVE_TRANSCRIPTS_DIR: the primary checkout's, a sibling of this repo. // SAMPLE_N (default 100): video dirs sampled per cue file. // // FREEZING THE INPUTS. The live editor rewrites the scheduler and auto-queue // state, rosters and maybe-missing records all day, so a parent run and a // branch run minutes apart would differ for reasons that are not this code. // `FREEZE_TO=/abs/dir` copies exactly what a measurement reads, in the same // layout, into that directory and exits; both runs then take // `LIVE_TRANSCRIPTS_DIR=/abs/dir`, and a frozen tree samples itself. import { createHash } from "node:crypto"; import fs from "node:fs"; import os from "node:os"; import path from "node:path"; import { fileURLToPath } from "node:url"; const HERE = path.dirname(fileURLToPath(import.meta.url)); const REPO = path.resolve(HERE, "..", ".."); const LIVE = process.env.LIVE_TRANSCRIPTS_DIR ?? path.join(path.dirname(REPO), "yt-dlp-transcript-browser", "transcripts"); const SAMPLE_N = Number(process.env.SAMPLE_N ?? 100); // A live-chat file can be hundreds of MB; the sample takes small ones. const LIVE_CHAT_MAX_BYTES = 4 * 1024 * 1024; const SCRATCH = fs.mkdtempSync(path.join(os.tmpdir(), "phase3-writers-")); process.env.TRANSCRIPTS_DIR = SCRATCH; process.env.TZ = "UTC"; const FIXED_NOW = Date.UTC(2026, 8, 24, 12, 0, 0); const RealDate = Date; class FixedDate extends RealDate { constructor(...args: unknown[]) { if (args.length === 0) super(FIXED_NOW); else super(...(args as [number])); } static now(): number { return FIXED_NOW; } } globalThis.Date = FixedDate as DateConstructor; type Format = "pretty" | "compact"; function md5(b: string | Buffer): string { return createHash("md5").update(b).digest("hex"); } function oldIdiom(v: unknown, format: Format): string { return format === "compact" ? JSON.stringify(v) : JSON.stringify(v, null, 2) + "\n"; } function sortedDir(dir: string): string[] { try { return fs.readdirSync(dir).sort(); } catch { return []; } } // Copy a live file (path relative to the corpus root) into the scratch tree. function stage(rel: string): boolean { const src = path.join(LIVE, rel); if (!fs.existsSync(src)) return false; const dest = path.join(SCRATCH, rel); fs.mkdirSync(path.dirname(dest), { recursive: true }); fs.copyFileSync(src, dest); return true; } // Report the file a writer just wrote. function report(label: string, file: string, format: Format): void { if (!fs.existsSync(file)) { console.log(`${label} ABSENT`); return; } const bytes = fs.readFileSync(file); let idiom: string; try { idiom = oldIdiom(JSON.parse(bytes.toString("utf8")), format) === bytes.toString("utf8") ? "old-idiom=same" : "old-idiom=DIFF"; } catch { idiom = "old-idiom=UNPARSEABLE"; } const leftovers = sortedDir(path.dirname(file)).filter((n) => n.startsWith(`${path.basename(file)}.tmp-`), ); console.log( `${label} ${bytes.length}B md5=${md5(bytes)} ${idiom}` + (leftovers.length ? ` LEFT ${leftovers.length} temp(s)` : ""), ); } async function run(label: string, fn: () => Promise): Promise { try { await fn(); } catch (e) { console.log(`${label} THREW: ${(e as Error).message}`); } } async function channelStores(slugs: string[]): Promise { const { getPaths } = await import("../../common/lib/paths"); const roster = await import("../../common/controller/rosterStore"); const mm = await import("../../common/controller/maybeMissingStore"); const ms = await import("../../common/controller/metadataScanStore"); const shard = await import("../../common/controller/shard"); const paths = getPaths(); console.log("# per-channel stores (every live file)"); for (const slug of slugs) { const ch = path.join("channels", slug); if (stage(path.join(ch, roster.ROSTER_FILENAME))) { await run(`roster ${slug}`, async () => { await roster.writeRoster(paths, slug, await roster.loadRoster(paths, slug)); report(`roster ${slug}`, roster.rosterPath(paths, slug), "pretty"); }); } if (stage(path.join(ch, mm.MAYBE_MISSING_FILENAME))) { await run(`maybe-missing ${slug}`, async () => { const rec = await mm.loadMaybeMissing(paths, slug); if (!rec) return void console.log(`maybe-missing ${slug} load=null`); await mm.writeMaybeMissing(paths, slug, rec); report( `maybe-missing ${slug}`, path.join(paths.channelsDir, slug, mm.MAYBE_MISSING_FILENAME), "pretty", ); }); } if (stage(path.join(ch, ms.METADATA_SCAN_FILENAME))) { await run(`metadata-scan ${slug}`, async () => { // The writer is private; an upsert that changes one entry's title by a // fixed suffix forces exactly one write of the whole store. const scan = await ms.loadMetadataScan(paths, slug); const id = Object.keys(scan.entries).sort()[0]; const upsert = id ? { entries: { [id]: { ...scan.entries[id], title: `${scan.entries[id].title}·` } } } : {}; await ms.upsertMetadataScan(paths, slug, upsert, "2026-09-24T12:00:00.000Z"); report(`metadata-scan ${slug}`, ms.metadataScanPath(paths, slug), "pretty"); }); } for (const op of shard.SHARD_OPS) { const rel = path.relative(SCRATCH, shard.shardFile(paths, slug, op)); if (!stage(rel)) continue; await run(`shard-${op} ${slug}`, async () => { const cfg = await shard.loadShardConfig(paths, slug, op); if (!cfg) return void console.log(`shard-${op} ${slug} load=null`); await shard.saveShardConfig(paths, slug, op, cfg); report(`shard-${op} ${slug}`, shard.shardFile(paths, slug, op), "pretty"); }); } } } async function globalStores(): Promise { const { getPaths } = await import("../../common/lib/paths"); const paths = getPaths(); const rel = (abs: string) => path.relative(SCRATCH, abs); console.log(""); console.log("# process-global stores"); const sched = await import("../../common/jobs/syncSchedulerState"); if (stage(rel(paths.schedulerStateFile))) { await run("scheduler", async () => { await sched.writeSchedulerState(paths, await sched.readSchedulerState(paths)); report("scheduler", paths.schedulerStateFile, "pretty"); }); } const aq = await import("../../common/jobs/autoQueueState"); if (stage(rel(paths.autoQueueStateFile))) { await run("auto-queue", async () => { await aq.writeAutoQueueState(paths, await aq.readAutoQueueState(paths)); report("auto-queue", paths.autoQueueStateFile, "pretty"); }); } const wd = await import("../../common/jobs/workerDefaults"); if (stage(rel(paths.workerDefaultsFile))) { await run("worker-defaults", async () => { const d = wd.readWorkerDefaults(paths); if (!d) return void console.log("worker-defaults load=null"); await wd.writeWorkerDefaults(paths, d.enabledWorkerIds); report("worker-defaults", paths.workerDefaultsFile, "pretty"); }); } const wp = await import("../../common/lib/widgetPresets"); if (stage(rel(paths.widgetPresetsFile))) { await run("widget-presets", async () => { const list = await wp.readWidgetPresets(paths); if (list.length === 0) return void console.log("widget-presets empty"); // Saving over an existing name rewrites the whole list, unchanged. await wp.addWidgetPreset(paths, list[0].name, list[0].query); report("widget-presets", paths.widgetPresetsFile, "pretty"); }); } const hp = await import("../../common/lib/homepage"); if (stage(rel(paths.homepageConfigFile))) { await run("homepage", async () => { await hp.writeHomepageConfig(hp.getHomepageConfig(paths), paths); report("homepage", paths.homepageConfigFile, "pretty"); }); } const dup = await import("../../common/controller/duplicateShorts"); if (stage(rel(dup.duplicateOverridesPath(paths)))) { await run("duplicate-overrides", async () => { const cur = await dup.readDuplicateOverrides(paths); const id = Object.keys(cur.clusters).sort()[0]; if (!id) return void console.log("duplicate-overrides empty"); const c = cur.clusters[id] as Record; await dup.updateDuplicateOverride(paths, id, { ...(typeof c.canonicalSlug === "string" ? { canonicalSlug: c.canonicalSlug } : {}), ...(typeof c.notDuplicate === "boolean" ? { notDuplicate: c.notDuplicate } : {}), ...(typeof c.confirmed === "boolean" ? { confirmed: c.confirmed } : {}), ...(typeof c.note === "string" ? { note: c.note } : {}), }); report("duplicate-overrides", dup.duplicateOverridesPath(paths), "pretty"); }); } // The report writer runs at the end of a detection pass; over the scratch // tree (configs and stores, no data/) that pass sees no videos. await run("duplicates-report", async () => { await dup.detectDuplicateShorts({ paths, onLog: () => {} }); report("duplicates-report (no videos)", path.join(paths.transcriptsDir, "duplicates.json"), "compact"); }); const scm = await import("../../common/controller/scanCorruptMedia"); if (stage(rel(scm.mediaScanOverridesPath(paths)))) { await run("media-scan-overrides", async () => { const cur = await scm.readMediaScanOverrides(paths); const key = Object.keys(cur.reviewed).sort()[0]; if (!key) return void console.log("media-scan-overrides empty"); const note = (cur.reviewed[key] as { note?: string }).note; await scm.updateMediaScanOverride(paths, key, { reviewed: true, ...(note ? { note } : {}) }); report("media-scan-overrides", scm.mediaScanOverridesPath(paths), "pretty"); }); } // A scan of a channel that does not exist merges the live report's every // finding back unchanged, through the report writer. if (stage("media-scan.json")) { await run("media-scan-report", async () => { await scm.scanCorruptMedia({ paths, channels: ["__phase3-writers-none__"], onLog: () => {} }); report("media-scan-report (merge of live)", path.join(paths.transcriptsDir, "media-scan.json"), "compact"); }); } const rd = await import("../../common/controller/relocateDir"); // No live marker exists outside a move; a fixed one exercises the writer. await run("relocation-marker", async () => { const file = path.join(SCRATCH, "channels", ".phase3-writers-marker.json"); fs.mkdirSync(path.dirname(file), { recursive: true }); await rd.writeDirMarker(file, { target: "/mnt/x/slug/data", direction: "out", startedAt: "2026-09-24T12:00:00.000Z", phase: "copy", }); report("relocation-marker (synthetic)", file, "pretty"); }); const bk = await import("../../common/controller/backupSavedVideos"); await run("backup-manifest", async () => { const dest = path.join(SCRATCH, ".phase3-writers-backup"); const r = await bk.backupSavedVideos({ paths, dest, onLog: () => {} }); report(`backup-manifest (${r.entries} saved)`, r.manifestPath, "pretty"); }); } // The first SAMPLE_N video dirs (sorted slugs × sorted ids) holding `filename`. function sampleDirs(slugs: string[], filename: string, maxBytes?: number): string[] { const out: string[] = []; for (const slug of slugs) { const data = path.join(LIVE, "channels", slug, "data"); for (const id of sortedDir(data)) { if (out.length >= SAMPLE_N) return out; const f = path.join(data, id, filename); try { const st = fs.statSync(f); if (maxBytes !== undefined && st.size > maxBytes) continue; out.push(path.join(data, id)); } catch { // not in this dir } } } return out; } const MEDIA = /\.(opus|m4a|mp3|webm|mp4|mkv|wav|ogg|aac|flac|part|ytdl|jpg|jpeg|png|webp)$/i; // Copy a video dir's non-media files (metadata, raw transcripts, the prior cue // file whose format tag a re-normalize reads) into the scratch tree. function stageVideoDir(live: string, n: number): string { const dir = path.join(SCRATCH, "videos", String(n)); fs.mkdirSync(dir, { recursive: true }); for (const name of sortedDir(live)) { if (MEDIA.test(name)) continue; const src = path.join(live, name); if (!fs.statSync(src).isFile()) continue; fs.copyFileSync(src, path.join(dir, name)); } return dir; } async function cueWriters(slugs: string[]): Promise { const nt = await import("../../common/controller/normalizeTranscript"); const nl = await import("../../common/controller/normalizeLiveChat"); const vs = await import("../../common/lib/videoStatus"); let n = 0; console.log(""); console.log(`# ${vs.CUES_JSON_FILENAME} (first ${SAMPLE_N} dirs holding one, force re-normalize)`); for (const live of sampleDirs(slugs, vs.CUES_JSON_FILENAME)) { const rel = path.relative(path.join(LIVE, "channels"), live); const dir = stageVideoDir(live, n++); await run(rel, async () => { const out = await nt.normalizeTranscript({ videoDir: dir, channelSlug: "x", force: true } as never); if ((out as { status: string }).status !== "wrote") { return void console.log(`${rel} ${(out as { status: string }).status}`); } report(rel, path.join(dir, vs.CUES_JSON_FILENAME), "compact"); }); } console.log(""); console.log(`# ${vs.LIVE_CHAT_CUES_FILENAME} (first ${SAMPLE_N} dirs with a live chat ≤ 4 MB)`); for (const live of sampleDirs(slugs, vs.LIVE_CHAT_FILENAME, LIVE_CHAT_MAX_BYTES)) { const rel = path.relative(path.join(LIVE, "channels"), live); const dir = stageVideoDir(live, n++); await run(rel, async () => { const out = await nl.normalizeLiveChat({ videoDir: dir, channelSlug: "x", force: true }); if (out.status !== "wrote") return void console.log(`${rel} ${out.status}`); report(rel, path.join(dir, vs.LIVE_CHAT_CUES_FILENAME), "compact"); }); } } // The files the measurement reads, relative to the corpus root. async function inputs(slugs: string[]): Promise { const { getPaths } = await import("../../common/lib/paths"); const shard = await import("../../common/controller/shard"); const dup = await import("../../common/controller/duplicateShorts"); const scm = await import("../../common/controller/scanCorruptMedia"); const vs = await import("../../common/lib/videoStatus"); const paths = getPaths(); const rel = (abs: string) => path.relative(SCRATCH, abs); const out: string[] = []; for (const s of slugs) { const ch = path.join("channels", s); out.push( path.join(ch, "config.json"), path.join(ch, "roster.json"), path.join(ch, "maybe-missing.json"), path.join(ch, "metadata-scan.json"), ...shard.SHARD_OPS.map((op) => rel(shard.shardFile(paths, s, op))), ); } out.push( rel(paths.schedulerStateFile), rel(paths.autoQueueStateFile), rel(paths.workerDefaultsFile), rel(paths.widgetPresetsFile), rel(paths.homepageConfigFile), rel(dup.duplicateOverridesPath(paths)), rel(scm.mediaScanOverridesPath(paths)), "media-scan.json", ); const dirs = [ ...sampleDirs(slugs, vs.CUES_JSON_FILENAME), ...sampleDirs(slugs, vs.LIVE_CHAT_FILENAME, LIVE_CHAT_MAX_BYTES), ]; for (const d of dirs) { for (const name of sortedDir(d)) { if (MEDIA.test(name) || !fs.statSync(path.join(d, name)).isFile()) continue; out.push(path.relative(LIVE, path.join(d, name))); } } return [...new Set(out)].filter((r) => fs.existsSync(path.join(LIVE, r))); } try { const slugs = sortedDir(path.join(LIVE, "channels")).filter((s) => fs.existsSync(path.join(LIVE, "channels", s, "config.json")), ); if (process.env.FREEZE_TO) { const to = path.resolve(process.env.FREEZE_TO); const files = await inputs(slugs); for (const r of files) { fs.mkdirSync(path.dirname(path.join(to, r)), { recursive: true }); fs.copyFileSync(path.join(LIVE, r), path.join(to, r)); } console.error(`froze ${files.length} files into ${to}`); fs.rmSync(SCRATCH, { recursive: true, force: true }); process.exit(0); } // Configs first: several writers read the channel list. for (const s of slugs) stage(path.join("channels", s, "config.json")); await channelStores(slugs); await globalStores(); await cueWriters(slugs); } finally { fs.rmSync(SCRATCH, { recursive: true, force: true }); }