#!/usr/bin/env tsx // The one-core Phase 3 slice 4b measurement: what the site.json, channel // config.json and per-video sidecar readers ANSWER over the real corpus, and // what a write of that answer puts on disk — printed deterministically so a run // on `main` and a run on the branch can be diffed. // // Model: phase3-settings-numbers.ts next door, and the same rules. // // NEVER WRITES THE CORPUS. Every file is COPIED into a scratch tree under // os.tmpdir() first; every read-for-measurement and every write-back happens on // the copy, and the scratch tree is deleted at the end. The live corpus is only // ever read (readdir, readFile, copyFile source). // // NEVER BOOTS A SERVER. The readers are called in-process; instrumentation.ts // is not loaded, so nothing is armed. // // ONE PROCESS. `TRANSCRIPTS_DIR` is pointed at the scratch tree BEFORE any // module that memoises `getPaths()` is imported, and every reader used here // takes its location explicitly (a Paths, or a video dir). // // USES ONLY NAMES PRESENT ON BOTH SIDES of slice 4b (getSite, writeSite, // readChannelConfig, writeChannelConfig, the loadX/writeX sidecar functions), // so the SAME file runs on `main` and on the branch. The declared-key lists come // from the branch's docs records when they exist and from a literal otherwise; // the literal is checked against the records when both are present. // // Usage, from the repo root: // LIVE_TRANSCRIPTS_DIR=/abs/transcripts node_modules/.bin/tsx plans/tools/phase3-files-numbers.ts > out.txt // Default LIVE_TRANSCRIPTS_DIR: the primary checkout's, a sibling of this repo. // SAMPLE_N (default 300): sidecar dirs sampled per filename. // // FREEZING THE INPUTS. The live corpus moves while a slice is in flight (the // running editor stamps `lastSyncedAt`, writes outcomes), so a `main` run in the // morning and a branch run in the evening would differ for reasons that are not // this code. `FREEZE_TO=/abs/dir` copies exactly what a measurement reads — // every site.json, every config.json, and the sampled sidecar files, in the same // layout — into that directory and exits; both runs then take // `LIVE_TRANSCRIPTS_DIR=/abs/dir`, and a frozen tree samples itself. import { createHash } from "node:crypto"; import fs from "node:fs"; import os from "node:os"; import path from "node:path"; import { fileURLToPath } from "node:url"; const HERE = path.dirname(fileURLToPath(import.meta.url)); const REPO = path.resolve(HERE, "..", ".."); const LIVE = process.env.LIVE_TRANSCRIPTS_DIR ?? path.join(path.dirname(REPO), "yt-dlp-transcript-browser", "transcripts"); const SAMPLE_N = Number(process.env.SAMPLE_N ?? 300); const SCRATCH = fs.mkdtempSync(path.join(os.tmpdir(), "phase3-files-")); process.env.TRANSCRIPTS_DIR = SCRATCH; process.env.TZ = "UTC"; // The keys each file may carry, as of 2026-09-24 (+ transcriptDownloads, release // 5). On the branch these must // equal the docs records (asserted below). const SITE_KEYS_LITERAL = [ "siteId", "siteTitle", "siteDescription", "headerTitle", "homeTagline", "accent", "socialLinks", "groups", "defaultGroupId", "channels", "cloudflareProject", "siteUrl", "relatedSites", "pwa", "archives", "archiveMaxBytes", "duplicates", "transcriptDownloads", "hubUrl", ]; const CHANNEL_KEYS_LITERAL = [ "handling", "sourceKind", "postFetcher", "socialHandle", "platform", "name", "url", "audioFormat", "downloadFormat", "keepSourceVideo", "keepLatest", "extractionMode", "savedVideosDir", "dataDir", "ytdlpExtraArgs", "subLangs", "lastSyncedAt", "lastFullDownloadAt", "lastFullSweepAt", "excludeFromBuild", "excludeFromCleanup", "syncIntervalMinutes", "fullSweepIntervalMinutes", "skipLiveDownloads", "downloadFilter", "cookiesFromBrowser", "cookieMode", "sleepBetweenDownloadsSeconds", "audioCheck", ]; function sortedKeys(_key: string, value: unknown): unknown { if (!value || typeof value !== "object" || Array.isArray(value)) return value; const src = value as Record; const out: Record = {}; for (const k of Object.keys(src).sort()) out[k] = src[k]; return out; } function canonical(v: unknown): string { // `undefined` members are dropped by JSON; that is also what a write drops. return JSON.stringify(v, sortedKeys, 2) ?? "undefined"; } function md5(text: string | Buffer): string { return createHash("md5").update(text).digest("hex"); } function sortedDir(dir: string): string[] { try { return fs.readdirSync(dir).sort(); } catch (e) { return []; } } function rawJson(file: string): unknown { try { return JSON.parse(fs.readFileSync(file, "utf8")); } catch { return undefined; } } function rawKeys(raw: unknown): string[] { return raw && typeof raw === "object" && !Array.isArray(raw) ? Object.keys(raw as object) : []; } async function declaredKeys(): Promise<{ site: string[]; channel: string[] }> { const siteMod = (await import("../../common/lib/site")) as Record; const chMod = (await import("../../common/lib/channelConfig")) as Record; const siteDocs = siteMod.SITE_FIELD_DOCS as Record | undefined; const chKeys = chMod.CHANNEL_CONFIG_KEYS as readonly string[] | undefined; const same = (a: readonly string[], b: readonly string[]) => [...a].sort().join() === [...b].sort().join(); if (siteDocs && !same(Object.keys(siteDocs), SITE_KEYS_LITERAL)) { throw new Error("SITE_FIELD_DOCS disagrees with this script's literal list"); } if (chKeys && !same(chKeys, CHANNEL_KEYS_LITERAL)) { throw new Error("CHANNEL_CONFIG_KEYS disagrees with this script's literal list"); } return { site: SITE_KEYS_LITERAL, channel: CHANNEL_KEYS_LITERAL }; } async function sites(declared: string[]): Promise { const { getPaths } = await import("../../common/lib/paths"); const { getSite, writeSite, siteConfigFile } = await import("../../common/lib/site"); const paths = getPaths(); console.log("# site.json"); for (const id of sortedDir(path.join(LIVE, "sites"))) { const live = path.join(LIVE, "sites", id, "site.json"); if (!fs.existsSync(live)) continue; const file = siteConfigFile(paths, id); fs.mkdirSync(path.dirname(file), { recursive: true }); fs.copyFileSync(live, file); console.log(`## ${id}`); const unknown = rawKeys(rawJson(file)).filter((k) => !declared.includes(k)); console.log(`unknown keys: ${JSON.stringify(unknown.sort())}`); const site = getSite(id, paths); console.log(canonical(site)); try { await writeSite(site, paths); console.log("### written by writeSite(getSite())"); console.log(fs.readFileSync(file, "utf8").trimEnd()); } catch (e) { console.log(`WRITE THREW: ${(e as Error).message}`); } } } async function channels(declared: string[]): Promise { const { getPaths } = await import("../../common/lib/paths"); const { readChannelConfig, writeChannelConfig } = await import( "../../common/controller/channels" ); const paths = getPaths(); console.log(""); console.log("# channel config.json"); const slugs: string[] = []; for (const slug of sortedDir(path.join(LIVE, "channels"))) { const live = path.join(LIVE, "channels", slug, "config.json"); if (!fs.existsSync(live)) continue; slugs.push(slug); const file = path.join(paths.channelsDir, slug, "config.json"); fs.mkdirSync(path.dirname(file), { recursive: true }); fs.copyFileSync(live, file); console.log(`## ${slug}`); const unknown = rawKeys(rawJson(file)).filter((k) => !declared.includes(k)); console.log(`unknown keys: ${JSON.stringify(unknown.sort())}`); const config = await readChannelConfig(paths, slug); console.log(canonical(config)); if (!config) continue; try { await writeChannelConfig(paths, slug, config); console.log(`written md5: ${md5(fs.readFileSync(file))}`); } catch (e) { console.log(`WRITE THREW: ${(e as Error).message}`); } } return slugs; } type Pair = { filename: string; load: (dir: string) => Promise; // Absent for the two markers whose only writer stamps the clock. write?: (dir: string, value: never) => Promise; }; async function sidecarPairs(): Promise { const attribution = await import("../../common/lib/attribution-server"); const diarization = await import("../../common/lib/diarization-server"); const availability = await import("../../common/lib/availability-server"); const dlo = await import("../../common/lib/downloadOutcome-server"); const tro = await import("../../common/lib/transcribeOutcome-server"); const dnc = await import("../../common/lib/doNotClean-server"); const etc = await import("../../common/lib/excludeTruncatedCheck-server"); const digest = await import("../../common/lib/digest-server"); const { ATTRIBUTION_FILENAME } = await import("../../common/lib/attribution"); const { DIARIZATION_FILENAME } = await import("../../common/lib/diarization"); const { AVAILABILITY_FILENAME } = await import("../../common/lib/availability"); const { DOWNLOAD_OUTCOME_FILENAME } = await import("../../common/lib/downloadOutcome"); const { TRANSCRIBE_OUTCOME_FILENAME } = await import("../../common/lib/transcribeOutcome"); const { DO_NOT_CLEAN_FILENAME } = await import("../../common/lib/doNotClean"); const { EXCLUDE_TRUNCATED_CHECK_FILENAME } = await import( "../../common/lib/excludeTruncatedCheck" ); const { DIGEST_FILENAME, DIGEST_OVERRIDES_FILENAME } = await import( "../../common/lib/digest" ); return [ { filename: ATTRIBUTION_FILENAME, load: attribution.loadAttribution, write: attribution.writeAttribution }, { filename: DIARIZATION_FILENAME, load: diarization.loadDiarization, write: diarization.writeDiarization }, { filename: AVAILABILITY_FILENAME, load: availability.loadAvailability, write: availability.writeAvailability }, { filename: DOWNLOAD_OUTCOME_FILENAME, load: dlo.loadDownloadOutcome, write: dlo.writeDownloadOutcome }, { filename: TRANSCRIBE_OUTCOME_FILENAME, load: tro.loadTranscribeOutcome, write: tro.writeTranscribeOutcome }, { filename: DO_NOT_CLEAN_FILENAME, load: dnc.loadDoNotClean }, { filename: EXCLUDE_TRUNCATED_CHECK_FILENAME, load: etc.loadExcludeTruncatedCheck }, { filename: DIGEST_FILENAME, load: digest.loadDigest, write: digest.writeDigest }, { filename: DIGEST_OVERRIDES_FILENAME, load: digest.loadDigestOverrides, write: digest.writeDigestOverrides }, ] as Pair[]; } // A deterministic sample: channels in sorted order, video ids in sorted order, // the first SAMPLE_N dirs holding each filename. function sample(slugs: string[], filenames: string[]): Map { const out = new Map(filenames.map((f) => [f, []])); for (const slug of slugs) { const data = path.join(LIVE, "channels", slug, "data"); for (const id of sortedDir(data)) { const dir = path.join(data, id); const present = new Set(sortedDir(dir)); for (const f of filenames) { const list = out.get(f)!; if (list.length < SAMPLE_N && present.has(f)) list.push(dir); } } if ([...out.values()].every((l) => l.length >= SAMPLE_N)) break; } return out; } async function sidecars(slugs: string[]): Promise { const pairs = await sidecarPairs(); const picked = sample(slugs, pairs.map((p) => p.filename)); console.log(""); console.log(`# sidecars (first ${SAMPLE_N} per filename, sorted slugs × sorted ids)`); let n = 0; for (const pair of pairs) { const dirs = picked.get(pair.filename)!; console.log(`## ${pair.filename} — ${dirs.length} dirs`); let nulls = 0; for (const live of dirs) { const rel = path.relative(path.join(LIVE, "channels"), live); const dir = path.join(SCRATCH, "sidecars", String(n++)); fs.mkdirSync(dir, { recursive: true }); fs.copyFileSync(path.join(live, pair.filename), path.join(dir, pair.filename)); const loaded = await pair.load(dir); if (loaded === null) nulls++; let written = "-"; if (loaded !== null && pair.write) { await pair.write(dir, loaded as never); const file = path.join(dir, pair.filename); written = fs.existsSync(file) ? md5(fs.readFileSync(file)) : "removed"; } console.log(`${rel} load=${md5(canonical(loaded))} write=${written}`); } console.log(`null loads: ${nulls}`); } } async function freeze(to: string): Promise { const copy = (rel: string) => { const dest = path.join(to, rel); fs.mkdirSync(path.dirname(dest), { recursive: true }); fs.copyFileSync(path.join(LIVE, rel), dest); }; for (const id of sortedDir(path.join(LIVE, "sites"))) { const rel = path.join("sites", id, "site.json"); if (fs.existsSync(path.join(LIVE, rel))) copy(rel); } const slugs: string[] = []; for (const slug of sortedDir(path.join(LIVE, "channels"))) { const rel = path.join("channels", slug, "config.json"); if (!fs.existsSync(path.join(LIVE, rel))) continue; slugs.push(slug); copy(rel); } const pairs = await sidecarPairs(); const picked = sample(slugs, pairs.map((p) => p.filename)); let files = 0; for (const [filename, dirs] of picked) { for (const dir of dirs) { copy(path.relative(LIVE, path.join(dir, filename))); files++; } } console.error(`froze ${slugs.length} configs and ${files} sidecar files into ${to}`); } try { if (process.env.FREEZE_TO) { await freeze(path.resolve(process.env.FREEZE_TO)); process.exit(0); } const declared = await declaredKeys(); await sites(declared.site); const slugs = await channels(declared.channel); await sidecars(slugs); } finally { fs.rmSync(SCRATCH, { recursive: true, force: true }); }