Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a8e506b28987a7f621d9773d4de051f401dc281f
parent 12353c3cc6b1c40d6bace8529aa5ec0e88d144a6
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 09:51:32 -0400

index: alternate English tracks ride on the transcript record where their words differ

A record's other English tracks (the served en beside en-orig, regional and
auto-translated tracks, the captions a local transcription replaced) are
kept where their words differ from the primary's and from every track kept
before them; identical ones add nothing. One notion of "track" lives in
lib/captionTracks.ts (ids, plain labels, the dedupe, the cross-track hit rule);
lib/captionTracks-server.ts reads them off disk.

buildIndex stores them in a sparse `alts` sub-DB and writes `track` +
`altTracks` onto the transcript page record only when one differs, so every
other page stays byte-identical. A one-shot pass (meta key `altTracks`,
ALT_TRACKS_VERSION) re-reads the records that can hold an alternate — two or
more English VTTs, or a transcription beside captions — and nothing else. A
transcribed record's caption inputs now count toward its change time.

English VTTs are no longer shipped as subtitle tracks: nothing read them there,
and the identical ones were most of the subs pages' English bulk.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/buildIndex.ts | 85++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Acommon/controller/buildIndexAltTracks.test.ts | 211+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/captionTrack.test.ts | 4++--
Acommon/lib/captionTracks-server.ts | 129+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/captionTracks.test.ts | 179+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/captionTracks.ts | 202+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/corpus.ts | 5++++-
Mcommon/lib/transcripts.ts | 6+++++-
Mcommon/lib/videoStatus.ts | 17++++++++---------
9 files changed, 822 insertions(+), 16 deletions(-)

diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -65,6 +65,8 @@ import type { DisplaySummary, } from "../lib/transcripts"; import type { StoredSubs, SubsDetail } from "../lib/subs"; +import type { TrackFields } from "../lib/captionTracks"; +import { readTrackFields } from "../lib/captionTracks-server"; import type { Manifest, ChannelEntry, @@ -234,6 +236,17 @@ function storedCuesEmpty(db: { getBinaryFast(key: IndexKey): Buffer | undefined return raw === undefined || raw.length <= 4; } +// ALTERNATE TRACKS — the same one-shot shape again (lib/captionTracks.ts). A +// record's other English tracks, kept where their words differ from the +// primary's, live in the `alts` sub-DB and ride on its transcript page +// (`track` + `altTracks`). The first build that sees a new ALT_TRACKS_VERSION +// re-reads every record that CAN hold one — a caption record with two or more +// English VTTs, a transcribed record with any — and nothing else; then records +// the version, only when no channel is held. Bump it when what an alternate is +// changes. +const ALT_TRACKS_VERSION = 1; +const ALT_TRACKS_KEY = "altTracks"; + export type CaptionTrackChannelReport = { reread: number; // Re-read records whose caption text is not what the index held. @@ -469,9 +482,13 @@ async function scanSource( let transcriptMs: number | null = null; if (picked) { // Captions: the newest of every caption input (each English VTT and - // the operator's pin), since the cues may come from any of them. + // the operator's pin), since the cues may come from any of them. A + // transcribed record's captions are its alternate tracks + // (lib/captionTracks.ts), so they count for it too. const inputs = - picked.kind === "vtt" ? captionInputs(files.entries) : [picked.filename]; + picked.kind === "vtt" + ? captionInputs(files.entries) + : [picked.filename, ...captionInputs(files.entries)]; for (const name of inputs) { try { const ms = (await stat(path.join(fullVideoDir, name))).mtimeMs; @@ -651,6 +668,12 @@ export async function buildIndex({ name: "subs", encoding: "msgpack", }); + // A record's alternate English tracks (ALT_TRACKS_KEY), only where one + // differs from the primary — sparse, like `subs`. + const alts = root.openDB<TrackFields, IndexKey>({ + name: "alts", + encoding: "msgpack", + }); const mtimes = root.openDB<MtimeRecord, PathKey>({ name: "mtimes", encoding: "msgpack", @@ -761,6 +784,7 @@ export async function buildIndex({ await sums.clearAsync(); await cues.clearAsync(); await subs.clearAsync(); + await alts.clearAsync(); await mtimes.clearAsync(); await byChannel.clearAsync(); await pageHashes.clearAsync(); @@ -878,6 +902,28 @@ export async function buildIndex({ } const captionReport = new Map<string, CaptionTrackChannelReport>(); + // Records that can hold an alternate track (ALT_TRACKS_KEY). A schema bump + // re-reads everything anyway. + const altTracksDue = meta.get(ALT_TRACKS_KEY) !== ALT_TRACKS_VERSION; + if (altTracksDue && !schemaBumped) { + const queued = new Set( + [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])), + ); + let reread = 0; + for (const s of live) { + const can = + (s.transcriptKind === "vtt" && s.englishVttCount >= 2) || + (s.transcriptKind === "whisper" && s.englishVttCount >= 1); + if (!can) continue; + const pk: PathKey = [s.channelSlug, s.videoDir]; + if (!mtimes.get(pk)) continue; + reread++; + if (!queued.has(pathKeyId(pk))) changed.push(s); + } + log(`Alternate tracks v${ALT_TRACKS_VERSION}: ${reread} record(s) re-read.`); + } + let altTrackRecords = 0; + const anyMutations = added.length > 0 || changed.length > 0 || removed.length > 0; @@ -1012,6 +1058,7 @@ export async function buildIndex({ sums.remove(prev.indexKey); cues.remove(prev.indexKey); subs.remove(prev.indexKey); + alts.remove(prev.indexKey); digests.remove(prev.indexKey); byChannel.remove(indexToChannelKey(prev.indexKey)); } @@ -1019,6 +1066,20 @@ export async function buildIndex({ if (cueList) cues.put(indexKey, cueList); else cues.remove(indexKey); + // The other English tracks, where their words differ from the + // primary's (lib/captionTracks.ts). Only a record that can hold one + // reads anything. + const canHoldAlts = + (s.transcriptKind === "vtt" && s.englishVttCount >= 2) || + (s.transcriptKind === "whisper" && s.englishVttCount >= 1); + const trackFields: TrackFields = canHoldAlts + ? await readTrackFields(videoFullDir, s.transcriptKind!, cueList) + : {}; + if (trackFields.altTracks) { + alts.put(indexKey, trackFields); + altTrackRecords++; + } else alts.remove(indexKey); + if (heldBefore !== undefined) { const r = captionReport.get(s.channelSlug) ?? { reread: 0, changed: 0, zeroToText: 0 }; r.reread++; @@ -1211,10 +1272,15 @@ export async function buildIndex({ ); } + if (altTrackRecords > 0) { + log(`Alternate tracks: ${altTrackRecords} re-read record(s) hold a track whose words differ from the primary's.`); + } + for (const { pathKey, indexKey } of removed) { sums.remove(indexKey); cues.remove(indexKey); subs.remove(indexKey); + alts.remove(indexKey); digests.remove(indexKey); byChannel.remove(indexToChannelKey(indexKey)); mtimes.remove(pathKey); @@ -1254,6 +1320,7 @@ export async function buildIndex({ await sums.flushed; await cues.flushed; await subs.flushed; + await alts.flushed; await digests.flushed; await byChannel.flushed; await mtimes.flushed; @@ -1436,7 +1503,16 @@ export async function buildIndex({ const summary = sums.get(indexKey); if (!summary) continue; const cueList = cues.get(indexKey); - const detail: TranscriptDetail = { ...summary, cues: cueList }; + // `track` + `altTracks` only on a record that has an alternate, so every + // other record's page bytes are what they were. + const trackFields = alts.get(indexKey); + const detail: TranscriptDetail = { + ...summary, + cues: cueList, + ...(trackFields?.altTracks?.length + ? { track: trackFields.track, altTracks: trackFields.altTracks } + : {}), + }; const encoded = JSON.stringify(detail); await writer.push(encoded, summary.id); } @@ -2434,6 +2510,9 @@ export async function buildIndex({ if (captionTrackDue && held.size === 0) { await meta.put(CAPTION_TRACK_KEY, CAPTION_TRACK_RULE_VERSION); } + if (altTracksDue && held.size === 0) { + await meta.put(ALT_TRACKS_KEY, ALT_TRACKS_VERSION); + } await meta.flushed; await root.close(); diff --git a/common/controller/buildIndexAltTracks.test.ts b/common/controller/buildIndexAltTracks.test.ts @@ -0,0 +1,211 @@ +// Integration: ALTERNATE TRACKS (lib/captionTracks.ts) through the REAL +// buildIndex over a temp corpus. A record whose served `en` says something its +// en-orig does not carries that track on its transcript page — so a word only +// in `en` is findable, with the track named — and a record whose tracks are +// identical carries nothing extra; no English VTT is shipped again as a +// subtitle track; and the one-shot pass (ALT_TRACKS_VERSION) re-reads exactly +// the records that can hold an alternate. +// +// Run with: node_modules/.bin/tsx --test common/controller/buildIndexAltTracks.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-alts-")); +const PINNED: Record<string, string> = { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), + EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), + CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), + SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), + CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), + ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), +}; +Object.assign(process.env, PINNED); +delete process.env.ARCHILYZER_INDEX_ALLOW_HELD; +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const { getPaths } = await import("../lib/paths"); +const { buildIndex } = await import("./buildIndex"); +const { open } = await import("lmdb"); +const { hitsAcrossTracks } = await import("../lib/captionTracks"); + +const paths = getPaths(); +const CHANNEL = "example-channel"; +const SITE = "testsite"; +const DIFFER = "DifferTrack1"; // en-orig + an en that says other words +const SAME = "SameTracks01"; // en-orig + an identical en +const WHISPER = "Transcribed1"; // transcript.json + captions that differ +const SPANISH = "SpanishSub01"; // en-orig + a Spanish subtitle track + +const fixture = (name: string) => + readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", name), "utf8"); +const ROLLING = fixture("vtt-rolling.vtt"); +const vtt = (...lines: [string, string, string][]) => + "WEBVTT\nKind: captions\nLanguage: en\n\n" + + lines.map(([a, b, t]) => `${a} --> ${b}\n${t}\n`).join("\n"); +// What the served `en` says: the same opening, then a word en-orig never has, +// far from anything en-orig matches. +const SERVED_EN = vtt( + ["00:00:03.080", "00:00:05.670", "are talking about the harbor"], + ["00:01:40.000", "00:01:43.000", "the zeppelin landed in nineteen thirty"], +); +const WHISPER_JSON = JSON.stringify({ + transcription: [ + { offsets: { from: 0, to: 2000 }, text: " a local transcription says hello" }, + { offsets: { from: 2000, to: 4000 }, text: " and nothing else" }, + ], +}); + +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; +const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id); + +function seed(): void { + rmSync(paths.transcriptsDir, { recursive: true, force: true }); + rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true }); + writeFileSync(paths.settingsFile, "{}"); + writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { + handling: "youtube", + name: CHANNEL, + }); + writeJson(path.join(paths.sitesDir, SITE, "site.json"), { + siteId: SITE, + siteTitle: "Test Site", + siteDescription: "fixture", + headerTitle: "Test Site", + homeTagline: "", + socialLinks: [], + groups: [{ id: "default", name: "All channels", selectedByDefault: true }], + defaultGroupId: "default", + channels: [{ slug: CHANNEL, groupId: "default" }], + }); + for (const [i, id] of [DIFFER, SAME, WHISPER, SPANISH].entries()) { + writeJson(path.join(dirOf(id), "metadata.info.json"), { + id, + title: `Video ${id}`, + upload_date: `2026060${i + 1}`, + duration: 200, + webpage_url: `https://www.youtube.com/watch?v=${id}`, + extractor_key: "Youtube", + }); + } + writeFileSync(path.join(dirOf(DIFFER), "transcript.en-orig.vtt"), ROLLING); + writeFileSync(path.join(dirOf(DIFFER), "transcript.en.vtt"), SERVED_EN); + writeFileSync(path.join(dirOf(SAME), "transcript.en-orig.vtt"), ROLLING); + writeFileSync(path.join(dirOf(SAME), "transcript.en.vtt"), ROLLING); + writeFileSync(path.join(dirOf(WHISPER), "transcript.json"), WHISPER_JSON); + writeFileSync(path.join(dirOf(WHISPER), "transcript.en-orig.vtt"), ROLLING); + writeFileSync(path.join(dirOf(SPANISH), "transcript.en-orig.vtt"), ROLLING); + writeFileSync( + path.join(dirOf(SPANISH), "transcript.es.vtt"), + vtt(["00:00:01.000", "00:00:02.000", "hola a todos"]), + ); +} + +type Cue = { start: number; end: number; text: string }; +type Rec = { + id: string; + cues?: Cue[]; + track?: string; + altTracks?: { track: string; cues: Cue[] }[]; +}; + +// Every record of the channel's shared transcript pages, by id. +function pageRecords(): Map<string, Rec> { + const dir = path.join(paths.exportSharedTranscriptsDir, CHANNEL); + const out = new Map<string, Rec>(); + for (const name of readdirSync(dir)) { + if (!/^page-\d+\.json$/.test(name)) continue; + for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as Rec[]) { + out.set(r.id, r); + } + } + return out; +} +function subsTracks(): Record<string, string[]> { + const dir = path.join(paths.exportSharedSubsDir, CHANNEL); + const out: Record<string, string[]> = {}; + if (!existsSync(dir)) return out; + for (const name of readdirSync(dir)) { + if (!/^page-\d+\.json$/.test(name)) continue; + for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as { + id: string; + tracks: Record<string, unknown>; + }[]) { + out[r.id] = Object.keys(r.tracks).sort(); + } + } + return out; +} + +async function runIndex(): Promise<string[]> { + const log: string[] = []; + await buildIndex({ paths, onLog: (s) => log.push(s) }); + return log; +} + +test("a differing served en rides on the page as an alternate; identical tracks add nothing", async () => { + seed(); + await runIndex(); + const recs = pageRecords(); + + const differ = recs.get(DIFFER)!; + assert.equal(differ.track, "en-orig"); + assert.deepEqual(differ.altTracks?.map((t) => t.track), ["en"]); + // The word only `en` has is found — in `en`, named — and the opening both + // say is found once, in the primary. + const find = (q: string) => + hitsAcrossTracks(differ, (cues) => cues.filter((c) => c.text.includes(q))); + assert.deepEqual( + find("zeppelin").map((h) => [h.track, Math.round(h.start)]), + [["en", 100]], + ); + assert.deepEqual(find("harbor").map((h) => h.track), [undefined]); + + // Identical tracks: no fields at all, so the record is what it always was. + const same = recs.get(SAME)!; + assert.equal("track" in same, false); + assert.equal("altTracks" in same, false); + + // A transcription is the primary; the captions it replaced are an alternate. + const whisper = recs.get(WHISPER)!; + assert.equal(whisper.track, "transcription"); + assert.equal(whisper.cues?.[0].text, "a local transcription says hello"); + assert.deepEqual(whisper.altTracks?.map((t) => t.track), ["en-orig"]); + + // No English VTT is a subtitle track any more; a Spanish one still is. + assert.deepEqual(subsTracks(), { [SPANISH]: ["es"] }); +}); + +test("the one-shot pass re-reads exactly the records that can hold an alternate, once", async () => { + // The index as a build before alternate tracks left it: no `alts` records, + // no version. + const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); + try { + await root.openDB({ name: "alts", encoding: "msgpack" }).clearAsync(); + await root.openDB({ name: "meta", encoding: "msgpack" }).remove("altTracks"); + } finally { + await root.close(); + } + const log = await runIndex(); + // DIFFER, SAME (two English VTTs) and WHISPER (a transcription beside + // captions); not SPANISH. + assert.ok(log.includes("Alternate tracks v1: 3 record(s) re-read."), log.join("\n")); + assert.deepEqual(pageRecords().get(DIFFER)?.altTracks?.map((t) => t.track), ["en"]); + + const again = await runIndex(); + assert.equal(again.some((l) => /Alternate tracks v1/.test(l)), false, again.join("\n")); +}); diff --git a/common/lib/captionTrack.test.ts b/common/lib/captionTrack.test.ts @@ -97,14 +97,14 @@ test("readEnglishVttCues: every track empty is the first track with no cues; non assert.equal(await readEnglishVttCues(videoDir({ "transcript.es.vtt": ROLLING })), null); }); -test("readSubTracks lists the served en as an alternate beside an en-orig primary", async () => { +test("readSubTracks lists no English VTT: those are caption tracks (lib/captionTracks.ts)", async () => { const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS, "transcript.en-orig.vtt": ROLLING, "transcript.es.vtt": ROLLING, }); const tracks = (await readSubTracks(dir)).map((t) => t.track).sort(); - assert.deepEqual(tracks, ["en", "es"]); + assert.deepEqual(tracks, ["es"]); }); test("umtool's copy of the caption-track rule version matches", () => { diff --git a/common/lib/captionTracks-server.ts b/common/lib/captionTracks-server.ts @@ -0,0 +1,129 @@ +// The alternate tracks of one video dir, read from disk (lib/captionTracks.ts +// says what an alternate is). SERVER-ONLY (node:fs). + +import path from "node:path"; +import { readdir, readFile } from "node:fs/promises"; +import { parseVtt, type Cue } from "./vtt"; +import { parseTranscriptJson } from "./whisper"; +import { + TRANSCRIPT_PIN_FILENAME, + VTT_FILENAME, + WHISPER_FILENAME, + englishVttsByPreference, + readEnglishVttCues, +} from "./videoStatus"; +import { + PINNED_TRACK, + TRANSCRIPTION_TRACK, + distinctAltTracks, + trackOfVttFile, + type AltTrack, + type TrackFields, +} from "./captionTracks"; + +// The track id of an English caption file in a listing: its language code, or +// `pinned` for transcript.en.vtt while the operator's pin stands. +export function trackIdOfVtt(filename: string, entries: readonly string[]): string { + if (filename === VTT_FILENAME && entries.includes(TRANSCRIPT_PIN_FILENAME)) { + return PINNED_TRACK; + } + return trackOfVttFile(filename) ?? filename; +} + +// Whether a listing can hold an alternate at all — no file is read. A caption +// record needs two English VTTs; a transcribed one, one. +export function mayHaveAltTracks( + primaryKind: "vtt" | "whisper", + entries: readonly string[], +): boolean { + const n = englishVttsByPreference(entries).length; + return primaryKind === "whisper" ? n >= 1 : n >= 2; +} + +// Every English VTT of a listing, parsed, in preference order. One that cannot +// be read is left out. +export async function readEnglishVttTracks( + videoDir: string, + entries: readonly string[], +): Promise<{ filename: string; cues: Cue[] }[]> { + const out: { filename: string; cues: Cue[] }[] = []; + for (const filename of englishVttsByPreference(entries)) { + try { + out.push({ + filename, + cues: parseVtt(await readFile(path.join(videoDir, filename), "utf8")), + }); + } catch { + // unreadable: not a track + } + } + return out; +} + +// A record's track fields: the primary's id and the English tracks whose words +// differ from it. `primaryCues` is what the record's transcript holds (the +// dedupe compares against it); for a caption record, the primary is the first +// track in preference order that has a cue — the caption-track rule's content +// fallback, so the id names the track the words really came from. Empty +// (no fields) when nothing differs. +export async function readTrackFields( + videoDir: string, + primaryKind: "vtt" | "whisper", + primaryCues: readonly Cue[] | undefined, + entries?: readonly string[], +): Promise<TrackFields> { + const listing = entries ?? (await readdir(videoDir).catch(() => [] as string[])); + if (!mayHaveAltTracks(primaryKind, listing)) return {}; + const vtts = await readEnglishVttTracks(videoDir, listing); + let primaryTrack: string; + let primary: readonly Cue[] | undefined = primaryCues; + let candidates: { filename: string; cues: Cue[] }[]; + if (primaryKind === "whisper") { + primaryTrack = TRANSCRIPTION_TRACK; + candidates = vtts; + } else { + if (vtts.length === 0) return {}; + const idx = Math.max(0, vtts.findIndex((t) => t.cues.length > 0)); + primaryTrack = trackIdOfVtt(vtts[idx].filename, listing); + primary ??= vtts[idx].cues; + candidates = vtts.filter((_, i) => i !== idx); + } + const alts: AltTrack[] = distinctAltTracks( + primary, + candidates.map((t) => ({ track: trackIdOfVtt(t.filename, listing), cues: t.cues })), + ); + if (alts.length === 0) return {}; + return { track: primaryTrack, altTracks: alts }; +} + +// Every track of a video dir, read from disk, primary first: what the editor's +// transcript reader shows and switches between. The primary is the record's +// transcript by the same rules the index reads it with — a local +// transcription (transcript.json) over captions, captions by the caption-track +// rule — and the rest are the alternates readTrackFields keeps. Null when the +// dir holds no transcript. +export async function readVideoTracks( + videoDir: string, +): Promise<{ tracks: AltTrack[] } | null> { + const entries = await readdir(videoDir).catch(() => [] as string[]); + let primary: AltTrack | null = null; + let kind: "vtt" | "whisper" = "vtt"; + if (entries.includes(WHISPER_FILENAME)) { + try { + primary = { + track: TRANSCRIPTION_TRACK, + cues: parseTranscriptJson(await readFile(path.join(videoDir, WHISPER_FILENAME), "utf8")), + }; + kind = "whisper"; + } catch { + primary = null; + } + } + if (!primary) { + const read = await readEnglishVttCues(videoDir, entries); + if (!read) return null; + primary = { track: trackIdOfVtt(read.filename, entries), cues: read.cues }; + } + const fields = await readTrackFields(videoDir, kind, primary.cues, entries); + return { tracks: [primary, ...(fields.altTracks ?? [])] }; +} diff --git a/common/lib/captionTracks.test.ts b/common/lib/captionTracks.test.ts @@ -0,0 +1,179 @@ +// The tracks of a transcript (lib/captionTracks.ts + captionTracks-server.ts): +// labels from ids, which alternates are kept, and how a search finds a word +// across them. +// +// Run with: node_modules/.bin/tsx --test common/lib/captionTracks.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { + cuesOfTrack, + distinctAltTracks, + hitsAcrossTracks, + inTrackLabel, + recordTracks, + trackKind, + trackLabel, + uncoveredAltHits, +} from "./captionTracks"; +import { readTrackFields, readVideoTracks } from "./captionTracks-server"; +import { TRANSCRIPT_PIN_FILENAME } from "./videoStatus"; + +const ROOT = mkdtempSync(path.join(tmpdir(), "caption-tracks-")); +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const cue = (start: number, text: string) => ({ start, end: start + 2, text }); +const vtt = (...cues: [number, string][]) => + "WEBVTT\n\n" + + cues + .map(([s, t]) => { + const ts = (x: number) => `00:${String(Math.floor(x / 60)).padStart(2, "0")}:${String(x % 60).padStart(2, "0")}.000`; + return `${ts(s)} --> ${ts(s + 2)}\n${t}\n`; + }) + .join("\n"); + +let n = 0; +function videoDir(files: Record<string, string>): string { + const dir = path.join(ROOT, `v${n++}`); + mkdirSync(dir, { recursive: true }); + for (const [name, body] of Object.entries(files)) writeFileSync(path.join(dir, name), body); + return dir; +} + +test("labels are plain words derived from the track id", () => { + assert.equal(trackLabel("en-orig"), "original audio captions"); + assert.equal(trackLabel("en"), "uploaded captions"); + assert.equal(trackLabel("en-en-US"), "auto-translated captions"); + assert.equal(trackLabel("en-GB"), "UK English captions"); + assert.equal(trackLabel("en-x-foo"), "regional captions (en-x-foo)"); + assert.equal(trackLabel("transcription"), "transcription"); + assert.equal(trackLabel("pinned"), "chosen captions"); + assert.equal(inTrackLabel("en"), "in uploaded captions"); + assert.equal(trackKind("en-US"), "regional"); + assert.equal(trackKind("live_chat"), "other"); +}); + +test("an alternate is kept only where its words differ from the primary and every kept one", () => { + const primary = [cue(0, "hello there")]; + const kept = distinctAltTracks(primary, [ + { track: "en", cues: [cue(0, "hello there")] }, // identical words + { track: "en-GB", cues: [cue(5, "hello there")] }, // timing alone differs + { track: "en-US", cues: [cue(0, "hello their")] }, // other words: kept + { track: "en-en-US", cues: [cue(0, "hello their")] }, // same as en-US + { track: "en-CA", cues: [] }, // empty + ]); + assert.deepEqual(kept.map((t) => t.track), ["en-US"]); +}); + +test("recordTracks and cuesOfTrack: primary first; an unknown track is undefined", () => { + const rec = { + cues: [cue(0, "a")], + track: "en-orig", + altTracks: [{ track: "en", cues: [cue(0, "b")] }], + }; + assert.deepEqual(recordTracks(rec), ["en-orig", "en"]); + assert.equal(cuesOfTrack(rec, null)?.[0].text, "a"); + assert.equal(cuesOfTrack(rec, "en-orig")?.[0].text, "a"); + assert.equal(cuesOfTrack(rec, "en")?.[0].text, "b"); + assert.equal(cuesOfTrack(rec, "en-GB"), undefined); + assert.deepEqual(recordTracks({ cues: [] }), []); +}); + +test("a search finds a word every track says once, in the primary, and an alternate's own word there", () => { + const rec = { + cues: [cue(10, "the bridge opened"), cue(300, "and then we left")], + track: "en-orig", + altTracks: [ + { + track: "en", + cues: [cue(11, "the bridge opened in 1932"), cue(200, "a zeppelin flew over the bridge")], + }, + ], + }; + const find = (q: string) => + hitsAcrossTracks(rec, (cues) => cues.filter((c) => c.text.includes(q))); + assert.deepEqual( + find("bridge").map((h) => [h.start, h.track]), + [ + [10, undefined], + [200, "en"], + ], + ); + assert.deepEqual(find("1932"), [{ ...cue(11, "the bridge opened in 1932"), track: "en" }]); + assert.deepEqual(find("nothing"), []); + // A record with no alternates is its primary alone. + assert.deepEqual( + hitsAcrossTracks({ cues: rec.cues }, (cues) => cues.filter((c) => c.text.includes("left"))), + [cue(300, "and then we left")], + ); +}); + +test("uncoveredAltHits drops an alternate hit within the window of a primary one", () => { + assert.deepEqual( + uncoveredAltHits([{ start: 100 }, { start: 500 }], [{ start: 90 }, { start: 130 }, { start: 515 }, { start: 900 }]), + [{ start: 130 }, { start: 900 }], + ); + assert.deepEqual(uncoveredAltHits([], [{ start: 1 }]), [{ start: 1 }]); +}); + +test("readTrackFields: a differing en beside en-orig is an alternate; identical tracks are none", async () => { + const differ = videoDir({ + "transcript.en-orig.vtt": vtt([1, "said words"]), + "transcript.en.vtt": vtt([1, "uploaded words"]), + }); + assert.deepEqual(await readTrackFields(differ, "vtt", undefined), { + track: "en-orig", + altTracks: [{ track: "en", cues: [{ start: 1, end: 3, text: "uploaded words" }] }], + }); + const same = videoDir({ + "transcript.en-orig.vtt": vtt([1, "said words"]), + "transcript.en.vtt": vtt([1, "said words"]), + }); + assert.deepEqual(await readTrackFields(same, "vtt", undefined), {}); + // One English VTT: nothing to read. + const lone = videoDir({ "transcript.en.vtt": vtt([1, "x"]) }); + assert.deepEqual(await readTrackFields(lone, "vtt", undefined), {}); +}); + +test("readTrackFields: an empty en-orig falls through to en as the primary, as the rule reads it", async () => { + const dir = videoDir({ + "transcript.en-orig.vtt": "WEBVTT\n\n", + "transcript.en.vtt": vtt([1, "served words"]), + "transcript.en-GB.vtt": vtt([1, "british words"]), + }); + assert.deepEqual(await readTrackFields(dir, "vtt", undefined), { + track: "en", + altTracks: [{ track: "en-GB", cues: [{ start: 1, end: 3, text: "british words" }] }], + }); +}); + +test("readTrackFields: the operator's pin names the primary `pinned`; its source is not repeated", async () => { + const dir = videoDir({ + "transcript.en-orig.vtt": vtt([1, "said words"]), + "transcript.en.vtt": vtt([1, "said words"]), // the copy of en-orig the pin made + "transcript.en-US.vtt": vtt([1, "other words"]), + [TRANSCRIPT_PIN_FILENAME]: JSON.stringify({ from: "transcript.en-orig.vtt", pinnedAt: "" }), + }); + assert.deepEqual(await readTrackFields(dir, "vtt", undefined), { + track: "pinned", + altTracks: [{ track: "en-US", cues: [{ start: 1, end: 3, text: "other words" }] }], + }); +}); + +test("readVideoTracks: a transcription is the primary and its captions the alternate", async () => { + const dir = videoDir({ + "transcript.json": JSON.stringify({ + transcription: [{ offsets: { from: 0, to: 1000 }, text: " machine words" }], + }), + "transcript.en-orig.vtt": vtt([1, "caption words"]), + }); + const read = await readVideoTracks(dir); + assert.deepEqual(read?.tracks.map((t) => [t.track, t.cues[0].text]), [ + ["transcription", "machine words"], + ["en-orig", "caption words"], + ]); + assert.equal(await readVideoTracks(videoDir({ "metadata.info.json": "{}" })), null); +}); diff --git a/common/lib/captionTracks.ts b/common/lib/captionTracks.ts @@ -0,0 +1,202 @@ +// THE TRACKS OF A TRANSCRIPT — one notion of "track" for every reader. +// +// A record's transcript is read from ONE track, its primary, chosen by the +// caption-track rule (lib/videoStatus.ts: en-orig first, the operator's pin +// above all, a local transcription above captions). A record may hold other +// English tracks beside it: the served `en`, a regional en-GB, an en→en +// auto-translation, or the captions a local transcription replaced. Human +// captions are not always a transcript of what was said, so those stay +// readable and searchable as ALTERNATES — a viewer's choice, never a change to +// the primary (the pin, transcript-pin.json, is how the primary changes). +// +// An alternate is kept only where its words differ from the primary's and from +// every alternate kept before it: most served `en` tracks are byte-identical to +// en-orig, and shipping or indexing those would double the corpus for nothing. +// +// Track ids are the caption's language code as its file names it +// (transcript.<id>.vtt), plus two that no file names: `transcription` (a local +// whisper/parakeet transcript) and `pinned` (transcript.en.vtt while the +// operator's pin stands — a copy of whichever track was picked). Labels are +// derived from the id, here and nowhere else, so the editor, the export site, +// a search hit and the MCP all say the same words. +// +// Pure: no node built-ins, safe in a client bundle. + +import type { Cue } from "./vtt"; + +export const TRANSCRIPTION_TRACK = "transcription"; +export const PINNED_TRACK = "pinned"; + +export type AltTrack = { track: string; cues: Cue[] }; + +// What a transcript record carries about its tracks. Both fields are OMITTED +// when the record has no alternate that differs (the common case), so those +// records' pages stay byte-identical to the ones already on disk. +export type TrackFields = { + // The primary's track id. + track?: string; + // The English tracks whose words differ from the primary's, in preference + // order. + altTracks?: AltTrack[]; +}; + +const REGIONS: Record<string, string> = { + US: "US", + GB: "UK", + UK: "UK", + CA: "Canadian", + AU: "Australian", + IE: "Irish", + IN: "Indian", + NZ: "New Zealand", + ZA: "South African", +}; + +// The kind of a track, from its id. +export type TrackKind = + | "original" + | "uploaded" + | "regional" + | "translated" + | "transcription" + | "pinned" + | "other"; + +export function trackKind(track: string): TrackKind { + if (track === TRANSCRIPTION_TRACK) return "transcription"; + if (track === PINNED_TRACK) return "pinned"; + if (track === "en-orig") return "original"; + if (track === "en") return "uploaded"; + if (/^en-en(?:-|$)/.test(track)) return "translated"; + if (/^en-/.test(track)) return "regional"; + return "other"; +} + +// A plain label for a track — what a person reads in the switcher and on a +// search hit ("in uploaded captions"). +export function trackLabel(track: string): string { + switch (trackKind(track)) { + case "original": + return "original audio captions"; + case "uploaded": + return "uploaded captions"; + case "translated": + return "auto-translated captions"; + case "transcription": + return "transcription"; + case "pinned": + return "chosen captions"; + case "regional": { + const region = track.slice(3); + const name = REGIONS[region.toUpperCase()]; + return name ? `${name} English captions` : `regional captions (${track})`; + } + default: + return `captions (${track})`; + } +} + +// The track id of a caption file name (transcript.<id>.vtt), or null. +export function trackOfVttFile(filename: string): string | null { + const m = filename.match(/^transcript\.([^.]+)\.vtt$/); + return m ? m[1] : null; +} + +// A cue list's words, for "do these two tracks say the same thing" — timing +// alone is not a different transcript. +export function cueWords(cues: readonly Cue[]): string { + return cues.map((c) => c.text).join("\n"); +} + +// The alternates worth keeping: tracks with at least one cue whose words differ +// from the primary's and from every track kept before them. `candidates` in +// preference order; the primary is not among them. +export function distinctAltTracks( + primary: readonly Cue[] | undefined, + candidates: readonly AltTrack[], +): AltTrack[] { + const seen = new Set<string>([cueWords(primary ?? [])]); + const out: AltTrack[] = []; + for (const c of candidates) { + if (c.cues.length === 0) continue; + const words = cueWords(c.cues); + if (seen.has(words)) continue; + seen.add(words); + out.push(c); + } + return out; +} + +// The tracks a record offers, primary first: what a switcher lists. A record +// with no alternates offers just its primary (or nothing, with no track id). +export function recordTracks(rec: TrackFields): string[] { + if (!rec.altTracks || rec.altTracks.length === 0) return rec.track ? [rec.track] : []; + return [rec.track ?? "", ...rec.altTracks.map((t) => t.track)].filter((t) => t !== ""); +} + +// The cues of one track of a record: the primary when `track` is absent or +// names it, an alternate when it names one, undefined when the record has no +// such track. +export function cuesOfTrack<R extends TrackFields & { cues?: Cue[] | undefined }>( + rec: R, + track: string | null | undefined, +): Cue[] | undefined { + if (!track || track === rec.track) return rec.cues; + return rec.altTracks?.find((t) => t.track === track)?.cues; +} + +// How far (seconds) a primary hit covers an alternate's. An alternate hit with +// a primary hit this close is the same moment found twice — the primary's is +// the one shown. Only a match the primary has nowhere near is the alternate's +// to report. +export const ALT_HIT_COVERED_SEC = 20; + +// Drop the alternate hits a primary hit already covers (ALT_HIT_COVERED_SEC). +// Both lists carry `start` in seconds; the primary's needs no order. +export function uncoveredAltHits<H extends { start: number }>( + primary: readonly { start: number }[], + alt: readonly H[], +): H[] { + if (primary.length === 0) return [...alt]; + const starts = primary.map((h) => h.start).sort((a, b) => a - b); + return alt.filter((h) => { + // Binary search for the nearest primary start. + let lo = 0; + let hi = starts.length - 1; + while (lo < hi) { + const mid = (lo + hi) >> 1; + if (starts[mid] < h.start) lo = mid + 1; + else hi = mid; + } + const near = [starts[lo], starts[lo - 1]].filter((s) => s !== undefined); + return !near.some((s) => Math.abs(s - h.start) <= ALT_HIT_COVERED_SEC); + }); +} + +// THE ONE RULE for matching a record's words across its tracks, used by every +// search (the viewer's leaf pipeline, the MCP, the query-tree evaluator): +// `find` is run over the primary's cues, then over each alternate's, and an +// alternate's hit is kept only where no hit already kept is within +// ALT_HIT_COVERED_SEC — so a word every track says is found once, in the +// primary, and a word only an alternate says is found there, wearing that +// alternate's track id. Ordered by start when an alternate adds anything. +export function hitsAcrossTracks<H extends { start: number }>( + rec: TrackFields & { cues?: readonly Cue[] | undefined }, + find: (cues: readonly Cue[]) => H[], +): (H & { track?: string })[] { + const out: (H & { track?: string })[] = rec.cues ? find(rec.cues) : []; + let added = false; + for (const alt of rec.altTracks ?? []) { + for (const h of uncoveredAltHits(out, find(alt.cues))) { + out.push({ ...h, track: alt.track }); + added = true; + } + } + if (added) out.sort((a, b) => a.start - b.start); + return out; +} + +// "in uploaded captions" — what a hit from an alternate says about itself. +export function inTrackLabel(track: string): string { + return `in ${trackLabel(track)}`; +} diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts @@ -80,7 +80,10 @@ const SHARD_SCHEME = { "<channel.manifests.transcripts> -> { pageCount, slugToPage: { <videoId>: <pageNumber> } }", transcriptPage: "/transcripts/<slug>/page-<NNNN>.json -> array of { id, title, uploadDate, " + - "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }", + "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }; " + + "a record with other English caption tracks whose words differ from its transcript " + + "also carries `track` (the transcript's track id, e.g. \"en-orig\") and " + + "`altTracks: [{ track, cues }]` (e.g. \"en\", the uploaded captions) — absent otherwise", subsManifest: "<channel.manifests.subs> -> lighter list-view records under the same slugToPage scheme", summariesIndex: diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts @@ -1,5 +1,6 @@ import path from "node:path"; import type { Cue } from "./vtt"; +import type { TrackFields } from "./captionTracks"; import { readFile } from "fs-extra"; import { getPaths } from "./paths"; @@ -69,9 +70,12 @@ export type DisplaySummary = { curatedTags?: string[]; }; +// `track` / `altTracks` (lib/captionTracks.ts): the primary's track id and the +// other English tracks whose words differ from it — present only on a record +// that has such a track. export type TranscriptDetail = TranscriptSummary & { cues: Cue[] | undefined; -}; +} & TrackFields; export type TranscriptPage = TranscriptDetail[]; diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -82,9 +82,9 @@ export type SubTrack = { ext: "vtt" | "json" | "json3" | "srv1" | "srv2" | "srv3"; }; -// Treat any transcript.<x>.<y> file as a sub track when it isn't one of the -// primary transcript outputs (transcript.en.vtt, transcript.json) or a -// derived/auxiliary file (transcript.cues.json). Live chat lands as +// Treat any transcript.<x>.<y> file as a sub track when it isn't a transcript +// (transcript.json, or ANY English VTT — the primary and its alternate tracks, +// lib/captionTracks.ts) or a derived/auxiliary file (transcript.cues.json). Live chat lands as // transcript.live_chat.json; non-en languages as transcript.<lang>.vtt. // Exported for lib/sidecar-server.ts, which refuses to declare a sidecar this // matches (a sidecar so named would be read as a subtitle track). @@ -274,15 +274,14 @@ export async function readVideoFiles( export async function readSubTracks(videoDir: string): Promise<SubTrack[]> { const entries = await readdir(videoDir).catch(() => [] as string[]); - const primaryVtt = resolvePrimaryVtt(entries); const tracks: SubTrack[] = []; for (const entry of entries) { if (entry === WHISPER_FILENAME) continue; - // The resolved primary English VTT (transcript.en-orig.vtt, or e.g. - // transcript.en-US.vtt when there is nothing better) is the main - // transcript, not an alternate sub-track. Any other English track — the - // served transcript.en.vtt beside an en-orig — is an alternate. - if (entry === primaryVtt) continue; + // Every English VTT is a CAPTION track, not a sub track: the primary is + // the transcript, and the others are its alternate tracks, kept only + // where their words differ (lib/captionTracks.ts) — not shipped again, + // identical or not, as subtitles. + if (isEnglishVtt(entry)) continue; if (entry === CUES_JSON_FILENAME) continue; if (entry === LIVE_CHAT_CUES_FILENAME) continue; const m = entry.match(SUB_FILE_RE);