// Produce a self-describing compact transcript file alongside the raw // VTT/Whisper output so that downstream consumers (buildIndex, archives) // can skip re-parsing and re-summarising. Mirrors the TranscriptDetail // shape that buildIndex would emit, plus a `source` marker. import path from "node:path"; import { open, readdir, readFile, stat } from "node:fs/promises"; import { writeJsonAtomic } from "../lib/jsonFile-server"; import { hasWordTiming, type Cue } from "../lib/vtt"; import { detectTranscriptFormat, parseTranscriptJson, } from "../lib/whisper"; import type { TranscriptOutputFormat } from "../lib/transcriptionApps"; import { summarize, type RawMetadata } from "../lib/transcripts-server"; import type { TranscriptDetail } from "../lib/transcripts"; import { transcriptCoverage, type TranscriptCoverage, } from "../lib/transcriptCoverage"; import { CAPTION_TRACK_RULE_VERSION, CUES_JSON_FILENAME, META_FILENAME, ORIG_VTT_FILENAME, VTT_FILENAME, WHISPER_FILENAME, captionInputs, englishVttsByPreference, pickIndexTranscript, readEnglishVttCues, readVideoFiles, type IndexTranscript, } from "../lib/videoStatus"; // v2: added `hlsUrl` to the summary/detail shape (Kick VOD playback), so stale // v1 cues.json files must regenerate to pick it up. export const CUES_FILE_VERSION = 2; export type NormalizedTranscript = TranscriptDetail & { version: number; source: IndexTranscript["kind"] | "live_chat"; // source "vtt" only: the English VTT the cues were read from, and the // caption-track rule that chose it (CAPTION_TRACK_RULE_VERSION). Written // right after `source`, so they sit in the file's first bytes and // isCuesJsonFresh can read them without parsing the cues. vttFile?: string; captionTrackRule?: number; // The raw transcript format this was parsed from. Durable per-video record so // a later re-normalize knows how to read the raw file without re-sniffing. transcriptFormat?: TranscriptOutputFormat; }; export type NormalizedLiveChat = NormalizedTranscript & { source: "live_chat" }; export type NormalizeOptions = { videoDir: string; channelSlug: string; configName?: string; // Authoritative raw-transcript format, passed by transcribeOne right after a // run (it knows which app produced the file). When omitted, the format is // resolved from the prior cues.json tag, then by content sniff. formatHint?: TranscriptOutputFormat; log?: (msg: string) => void; force?: boolean; }; export type NormalizeOutcome = | { status: "wrote"; cuesPath: string } | { status: "fresh"; cuesPath: string } | { status: "skipped"; reason: "no-raw-transcript" | "no-metadata" }; async function mtimeMs(p: string): Promise { try { return (await stat(p)).mtimeMs; } catch { return null; } } export async function normalizeTranscript( opts: NormalizeOptions, ): Promise { const files = await readVideoFiles(opts.videoDir); if (!files.hasMeta) return { status: "skipped", reason: "no-metadata" }; const picked = pickIndexTranscript(files); if (!picked) return { status: "skipped", reason: "no-raw-transcript" }; // The same question every reader of cues.json asks (isCuesJsonFresh), so a // normalize pass rewrites exactly the files they refuse. const freshness = await isCuesJsonFresh(opts.videoDir); const cuesPath = freshness.cuesPath; if (!opts.force && freshness.fresh) { return { status: "fresh", cuesPath }; } const built = await buildNormalizedTranscript(opts, files); if (built.status === "skipped") return built; const out = built.transcript; // Compact, no trailing newline: transcript.cues.json's historical bytes. await writeJsonAtomic(cuesPath, out, { indent: 0, newline: false }); opts.log?.( `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${out.vttFile ?? out.transcriptFormat}, ${out.cues?.length ?? 0} cues)`, ); return { status: "wrote", cuesPath }; } // WHAT normalizeTranscript WOULD WRITE, built in memory and not written: the // summary from metadata.info.json and the cues of the transcript the index // would pick (the caption-track rule for VTTs). Shared by normalize and by a // reader that must not write (the ops transcript read, release 19 A7). export async function buildNormalizedTranscript( opts: Pick, known?: Awaited>, ): Promise< | { status: "built"; transcript: NormalizedTranscript } | { status: "skipped"; reason: "no-raw-transcript" | "no-metadata" } > { const files = known ?? (await readVideoFiles(opts.videoDir)); if (!files.hasMeta) return { status: "skipped", reason: "no-metadata" }; const picked = pickIndexTranscript(files); if (!picked) return { status: "skipped", reason: "no-raw-transcript" }; const metaPath = path.join(opts.videoDir, META_FILENAME); const transcriptPath = path.join(opts.videoDir, picked.filename); const cuesPath = path.join(opts.videoDir, CUES_JSON_FILENAME); const metaRaw = await readFile(metaPath, "utf8"); const parsedMeta = JSON.parse(metaRaw) as RawMetadata; const summary = summarize( opts.channelSlug, path.basename(opts.videoDir), parsedMeta, opts.configName, ); let cues: Cue[]; let transcriptFormat: TranscriptOutputFormat; let vttFile: string | undefined; try { if (picked.kind === "vtt") { // The caption-track rule: the first English VTT, in preference order, // with cues (lib/videoStatus.ts). const read = await readEnglishVttCues(opts.videoDir, files.entries); if (!read) throw new Error("no English VTT could be read"); cues = read.cues; vttFile = read.filename; transcriptFormat = "vtt"; } else { const rawTranscript = await readFile(transcriptPath, "utf8"); // Resolve the JSON format: authoritative hint -> recorded per-video tag // -> content sniff -> whisper fallback (the only app before chough). transcriptFormat = opts.formatHint ?? (await readRecordedFormat(cuesPath)) ?? detectTranscriptFormat(rawTranscript) ?? "whisper-json"; cues = parseTranscriptJson(rawTranscript, transcriptFormat); } } catch (err) { throw new Error( `Failed to parse ${picked.filename} for ${opts.channelSlug}/${path.basename(opts.videoDir)}: ${(err as Error).message}`, ); } return { status: "built", transcript: { version: CUES_FILE_VERSION, source: picked.kind, ...(vttFile !== undefined ? { vttFile, captionTrackRule: CAPTION_TRACK_RULE_VERSION } : {}), transcriptFormat, ...summary, cues, }, }; } // Read the format recorded in a prior transcript.cues.json, if any. This is the // durable per-video tag that lets a re-normalize parse the raw file the same way // it was first parsed, without re-sniffing. async function readRecordedFormat( cuesPath: string, ): Promise { const prior = await readNormalizedTranscript(cuesPath); const f = prior?.transcriptFormat; return f === "whisper-json" || f === "chough-json" || f === "vtt" ? f : undefined; } export async function readNormalizedTranscript( cuesPath: string, ): Promise { try { const raw = await readFile(cuesPath, "utf8"); const parsed = JSON.parse(raw) as NormalizedTranscript; if (typeof parsed.version !== "number") return null; return parsed; } catch { return null; } } // Read transcript coverage (last cue end vs. duration) for a video dir from its // transcript.cues.json. Returns null when there's no normalized transcript. // Used by the snapshot builder and the editor pages to flag truncated downloads // — the shared math lives in ../lib/transcriptCoverage. export async function readTranscriptCoverage( videoDir: string, ): Promise<{ cov: TranscriptCoverage; isLivestream: boolean } | null> { const cuesPath = path.join(videoDir, CUES_JSON_FILENAME); const t = await readNormalizedTranscript(cuesPath); if (!t) return null; return { cov: transcriptCoverage(t.cues, t.duration), isLivestream: Boolean(t.isLivestream), }; } // WHY a cues.json is not usable. `fresh` is the answer every caller had before // and still gets; `reason` is the answer the DIGEST CARD needs, because the two // not-fresh cases have opposite meanings for an operator: // // missing — there is no transcript.cues.json at all. NOTHING PRODUCES ONE // AUTOMATICALLY for these: transcribeOne is the only automatic // caller of normalizeTranscript, and a `handling: "youtube"` channel // downloads subtitles with --skip-download and never runs it. So // this state is permanent until someone runs the normalize pass. // Measured on this corpus: 1,942 videos, 1,683 of them piratesoftware. // stale — a cues.json exists but the metadata or the raw transcript has been // rewritten under it. This IS the "superseded" case the digest lane // was written for. Measured: 47 videos, all shondo-vods. // // Both are fixed by the same normalize pass, which is why they share the // `deferred` classification — but only `missing` was ever the whole story, and // the copy that claimed the stale cause for both was wrong for 97.6% of them. export type CuesFreshReason = | "fresh" | "missing" | "stale" | "no-raw" | "no-meta"; // Helper: given a video dir, decide whether transcript.cues.json (if present) // is at least as new as metadata.info.json and the raw transcript file. Used // by buildIndex to know whether it can trust cues.json without re-parsing. // // CAPTIONS: the raw transcript is EVERY caption input (captionInputs — each // English VTT, since the content fallback may read any of them, and the // operator's pin), and a cues.json that is new enough must also have been made // under the current caption-track rule. Made under an older one, it is `stale` // where the rule could read it differently: its cues may come from a track the // rule no longer picks (a served `en` that differs from the `en-orig` beside // it), or be the zero cues a cue-block VTT used to parse to // (followsCaptionTrackRule). Telling costs a few small reads, never a parse. export async function isCuesJsonFresh( videoDir: string, ): Promise<{ fresh: boolean; reason: CuesFreshReason; cuesPath: string }> { const cuesPath = path.join(videoDir, CUES_JSON_FILENAME); const metaPath = path.join(videoDir, META_FILENAME); const entries = await readdir(videoDir).catch(() => [] as string[]); const whisperPath = path.join(videoDir, WHISPER_FILENAME); const inputs = captionInputs(entries); const [cuesMs, metaMs, whisperMs, ...inputMs] = await Promise.all([ mtimeMs(cuesPath), mtimeMs(metaPath), mtimeMs(whisperPath), ...inputs.map((n) => mtimeMs(path.join(videoDir, n))), ]); const vttMs = inputMs.reduce( (max, ms) => (ms !== null && (max === null || ms > max) ? ms : max), null, ); // Prefer whisper if present (matches pickIndexTranscript priority). const rawMs = whisperMs ?? vttMs; // Ordered to match normalizeTranscript's OWN precedence (metadata, then raw, // then the sidecar) so the reason names the thing a normalize run would // actually report — `no-meta` and `no-raw` are its two `skipped` outcomes, and // neither is fixable by running it. Every branch below returned fresh:false // before this change too, so no caller's behaviour moves. if (metaMs === null) return { fresh: false, reason: "no-meta", cuesPath }; if (rawMs === null) return { fresh: false, reason: "no-raw", cuesPath }; if (cuesMs === null) return { fresh: false, reason: "missing", cuesPath }; if (cuesMs < metaMs || cuesMs < rawMs) { return { fresh: false, reason: "stale", cuesPath }; } if (whisperMs === null && !(await followsCaptionTrackRule(videoDir, cuesPath, entries))) { return { fresh: false, reason: "stale", cuesPath }; } return { fresh: true, reason: "fresh", cuesPath }; } const HEAD_BYTES = 2048; async function readHead(file: string, bytes = HEAD_BYTES): Promise { const fh = await open(file, "r"); try { const buf = Buffer.alloc(bytes); const { bytesRead } = await fh.read(buf, 0, bytes, 0); return buf.subarray(0, bytesRead).toString("utf8"); } finally { await fh.close(); } } async function readTail(file: string, bytes = 64): Promise { const fh = await open(file, "r"); try { const { size } = await fh.stat(); const start = Math.max(0, size - bytes); const buf = Buffer.alloc(size - start); const { bytesRead } = await fh.read(buf, 0, buf.length, start); return buf.subarray(0, bytesRead).toString("utf8"); } finally { await fh.close(); } } // Whether a caption cues.json, already new enough by mtime, was made under the // current caption-track rule — normalize records it as `captionTrackRule` in // the file's first bytes — or, made before the rule had a version, holds what // the rule would read anyway. The older rule ranked transcript.en.vtt first // and parsed a cue-block VTT to no cues, so an unversioned file is stale only: // // when it holds NO cues (the cue-block parse, or the fallback to the next // track, may find text now) — the cues are the last key normalize writes, // so that is the file's tail; unless its one English VTT has word timing, // whose parse did not change and which had nothing to choose from; // when transcript.en.vtt and transcript.en-orig.vtt are both present and // differ — it was read from en, and en-orig is read now. A byte-identical // pair (equal size: on this corpus every one of 50,461 equal-size pairs was // byte-identical) reads the same either way. // // One to three small reads (a head, a tail, two stats), never a parse. A // mis-read only costs a rewrite, after which the record is there. async function followsCaptionTrackRule( videoDir: string, cuesPath: string, entries: readonly string[], ): Promise { const vtts = englishVttsByPreference(entries); if (vtts.length === 0) return true; try { if (vtts.length === 1 && hasWordTiming(await readHead(path.join(videoDir, vtts[0])))) { return true; } const m = (await readHead(cuesPath)).match(/"captionTrackRule"\s*:\s*(\d+)/); if (m !== null && Number(m[1]) === CAPTION_TRACK_RULE_VERSION) return true; if (/"cues"\s*:\s*\[\s*\]\s*\}\s*$/.test(await readTail(cuesPath))) return false; if (!entries.includes(VTT_FILENAME) || !entries.includes(ORIG_VTT_FILENAME)) return true; const [en, orig] = await Promise.all([ stat(path.join(videoDir, VTT_FILENAME)), stat(path.join(videoDir, ORIG_VTT_FILENAME)), ]); return en.size === orig.size; } catch { return false; } }