// Walk every channel/video and ensure each transcript has a sibling // transcript.cues.json (compact, self-describing). Cheap on rebuilds // because normalizeTranscript short-circuits when cues.json is already // at least as new as the metadata + raw transcript. import path from "node:path"; import { readdir } from "node:fs/promises"; import pLimit from "p-limit"; import { listChannelStatsFromDisk } from "./channels"; import { normalizeTranscript } from "./normalizeTranscript"; import type { Paths } from "../lib/paths"; import { assertChannelTextReadable } from "../lib/channelMedia"; export type NormalizeAllOptions = { paths: Paths; // Restrict to these channel slugs. Empty/omitted = the whole corpus, which is // what the Pool's button on /sites has always done. Channel scoping exists // because the videos that need this are CONCENTRATED — 1,683 of the 1,942 // unnormalized videos on this corpus are one channel — and clearing one // channel should not mean walking all ~79,000 dirs from a page nobody opens // during digest work. // // Same shape as diarizeAll's `channelSlugs`, deliberately: that controller // says in its own header that it was modelled on this one, so this is the // symmetry being completed rather than a new pattern. channelSlugs?: string[]; // Restrict to these video ids within the channels walked (release 19 A7, // `pnpm ops build-cues {slug, ids}`). An id the channel does not hold is // passed over silently here; the action refuses it before any job. videoIds?: string[]; // Rewrite cues.json even when it is fresh. force?: boolean; onLog?: (msg: string) => void; signal?: AbortSignal; concurrency?: number; }; export type NormalizeAllResult = { wrote: number; fresh: number; skipped: number; failed: number; }; export async function normalizeAllTranscripts( opts: NormalizeAllOptions, ): Promise { const log = opts.onLog ?? ((m: string) => console.log(m)); const limit = pLimit(opts.concurrency ?? 8); const wanted = new Set(opts.channelSlugs ?? []); const channels = (await listChannelStatsFromDisk(opts.paths)).filter( (ch) => wanted.size === 0 || wanted.has(ch.slug), ); const result: NormalizeAllResult = { wrote: 0, fresh: 0, skipped: 0, failed: 0, }; for (const ch of channels) { if (opts.signal?.aborted) break; const dataDir = path.join(opts.paths.channelsDir, ch.slug, "data"); // GUARD, same reason as the snapshot's and the batch's: `readdir(dataDir) // .catch(() => [])` cannot tell "this channel has downloaded nothing" from // "this channel's drive is not mounted", and here the second reads as a // clean run over zero videos that reports 0/0/0/0 and moves on. A sweep // that silently skips a channel is worse than one that stops on it. // THE TEXT GUARD (release 17): normalize reads and writes text only, so a // moving, stalled or unmounted MEDIA drive does not stop it. try { await assertChannelTextReadable(opts.paths, ch.slug, ch.config); } catch (err) { log(`Normalize ${ch.slug}: SKIPPED — ${(err as Error).message}`); result.failed++; continue; } const onlyIds = opts.videoIds ? new Set(opts.videoIds) : null; const videoIds = (await readdir(dataDir).catch(() => [] as string[])).filter( (id) => !onlyIds || onlyIds.has(id), ); log(`Normalize ${ch.slug}: ${videoIds.length} videos`); let wrote = 0; let fresh = 0; let skipped = 0; let failed = 0; await Promise.all( videoIds.map((id) => limit(async () => { if (opts.signal?.aborted) return; try { const outcome = await normalizeTranscript({ videoDir: path.join(dataDir, id), channelSlug: ch.slug, configName: ch.config.name, ...(opts.force ? { force: true } : {}), }); if (outcome.status === "wrote") wrote++; else if (outcome.status === "fresh") fresh++; else skipped++; } catch (err) { failed++; log(` ! ${ch.slug}/${id}: ${(err as Error).message}`); } }), ), ); log( ` ${ch.slug}: wrote=${wrote} fresh=${fresh} skipped=${skipped} failed=${failed}`, ); result.wrote += wrote; result.fresh += fresh; result.skipped += skipped; result.failed += failed; } log( `Done. wrote=${result.wrote} fresh=${result.fresh} skipped=${result.skipped} failed=${result.failed}`, ); return result; } // One channel. The digest lane's `deferred` bucket is exactly this run's work // list, so the button that clears it lives on the digest stage card — the count // and the fix in the same place. // // `skipped` here is the honest floor, not a failure: it counts videos with no // raw transcript or no metadata, which normalizing cannot help. On this corpus // that is 1,626 videos, and they are `blocked` on transcription rather than // deferred — a different number on the same card. export async function normalizeChannelTranscripts( opts: Omit & { channelSlug: string }, ): Promise { const { channelSlug, ...rest } = opts; return normalizeAllTranscripts({ ...rest, channelSlugs: [channelSlug] }); }