commit a8e506b28987a7f621d9773d4de051f401dc281f
parent 12353c3cc6b1c40d6bace8529aa5ec0e88d144a6
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 09:51:32 -0400
index: alternate English tracks ride on the transcript record where their words differ
A record's other English tracks (the served en beside en-orig, regional and
auto-translated tracks, the captions a local transcription replaced) are
kept where their words differ from the primary's and from every track kept
before them; identical ones add nothing. One notion of "track" lives in
lib/captionTracks.ts (ids, plain labels, the dedupe, the cross-track hit rule);
lib/captionTracks-server.ts reads them off disk.
buildIndex stores them in a sparse `alts` sub-DB and writes `track` +
`altTracks` onto the transcript page record only when one differs, so every
other page stays byte-identical. A one-shot pass (meta key `altTracks`,
ALT_TRACKS_VERSION) re-reads the records that can hold an alternate — two or
more English VTTs, or a transcription beside captions — and nothing else. A
transcribed record's caption inputs now count toward its change time.
English VTTs are no longer shipped as subtitle tracks: nothing read them there,
and the identical ones were most of the subs pages' English bulk.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
9 files changed, 822 insertions(+), 16 deletions(-)
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -65,6 +65,8 @@ import type {
DisplaySummary,
} from "../lib/transcripts";
import type { StoredSubs, SubsDetail } from "../lib/subs";
+import type { TrackFields } from "../lib/captionTracks";
+import { readTrackFields } from "../lib/captionTracks-server";
import type {
Manifest,
ChannelEntry,
@@ -234,6 +236,17 @@ function storedCuesEmpty(db: { getBinaryFast(key: IndexKey): Buffer | undefined
return raw === undefined || raw.length <= 4;
}
+// ALTERNATE TRACKS — the same one-shot shape again (lib/captionTracks.ts). A
+// record's other English tracks, kept where their words differ from the
+// primary's, live in the `alts` sub-DB and ride on its transcript page
+// (`track` + `altTracks`). The first build that sees a new ALT_TRACKS_VERSION
+// re-reads every record that CAN hold one — a caption record with two or more
+// English VTTs, a transcribed record with any — and nothing else; then records
+// the version, only when no channel is held. Bump it when what an alternate is
+// changes.
+const ALT_TRACKS_VERSION = 1;
+const ALT_TRACKS_KEY = "altTracks";
+
export type CaptionTrackChannelReport = {
reread: number;
// Re-read records whose caption text is not what the index held.
@@ -469,9 +482,13 @@ async function scanSource(
let transcriptMs: number | null = null;
if (picked) {
// Captions: the newest of every caption input (each English VTT and
- // the operator's pin), since the cues may come from any of them.
+ // the operator's pin), since the cues may come from any of them. A
+ // transcribed record's captions are its alternate tracks
+ // (lib/captionTracks.ts), so they count for it too.
const inputs =
- picked.kind === "vtt" ? captionInputs(files.entries) : [picked.filename];
+ picked.kind === "vtt"
+ ? captionInputs(files.entries)
+ : [picked.filename, ...captionInputs(files.entries)];
for (const name of inputs) {
try {
const ms = (await stat(path.join(fullVideoDir, name))).mtimeMs;
@@ -651,6 +668,12 @@ export async function buildIndex({
name: "subs",
encoding: "msgpack",
});
+ // A record's alternate English tracks (ALT_TRACKS_KEY), only where one
+ // differs from the primary — sparse, like `subs`.
+ const alts = root.openDB<TrackFields, IndexKey>({
+ name: "alts",
+ encoding: "msgpack",
+ });
const mtimes = root.openDB<MtimeRecord, PathKey>({
name: "mtimes",
encoding: "msgpack",
@@ -761,6 +784,7 @@ export async function buildIndex({
await sums.clearAsync();
await cues.clearAsync();
await subs.clearAsync();
+ await alts.clearAsync();
await mtimes.clearAsync();
await byChannel.clearAsync();
await pageHashes.clearAsync();
@@ -878,6 +902,28 @@ export async function buildIndex({
}
const captionReport = new Map<string, CaptionTrackChannelReport>();
+ // Records that can hold an alternate track (ALT_TRACKS_KEY). A schema bump
+ // re-reads everything anyway.
+ const altTracksDue = meta.get(ALT_TRACKS_KEY) !== ALT_TRACKS_VERSION;
+ if (altTracksDue && !schemaBumped) {
+ const queued = new Set(
+ [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])),
+ );
+ let reread = 0;
+ for (const s of live) {
+ const can =
+ (s.transcriptKind === "vtt" && s.englishVttCount >= 2) ||
+ (s.transcriptKind === "whisper" && s.englishVttCount >= 1);
+ if (!can) continue;
+ const pk: PathKey = [s.channelSlug, s.videoDir];
+ if (!mtimes.get(pk)) continue;
+ reread++;
+ if (!queued.has(pathKeyId(pk))) changed.push(s);
+ }
+ log(`Alternate tracks v${ALT_TRACKS_VERSION}: ${reread} record(s) re-read.`);
+ }
+ let altTrackRecords = 0;
+
const anyMutations =
added.length > 0 || changed.length > 0 || removed.length > 0;
@@ -1012,6 +1058,7 @@ export async function buildIndex({
sums.remove(prev.indexKey);
cues.remove(prev.indexKey);
subs.remove(prev.indexKey);
+ alts.remove(prev.indexKey);
digests.remove(prev.indexKey);
byChannel.remove(indexToChannelKey(prev.indexKey));
}
@@ -1019,6 +1066,20 @@ export async function buildIndex({
if (cueList) cues.put(indexKey, cueList);
else cues.remove(indexKey);
+ // The other English tracks, where their words differ from the
+ // primary's (lib/captionTracks.ts). Only a record that can hold one
+ // reads anything.
+ const canHoldAlts =
+ (s.transcriptKind === "vtt" && s.englishVttCount >= 2) ||
+ (s.transcriptKind === "whisper" && s.englishVttCount >= 1);
+ const trackFields: TrackFields = canHoldAlts
+ ? await readTrackFields(videoFullDir, s.transcriptKind!, cueList)
+ : {};
+ if (trackFields.altTracks) {
+ alts.put(indexKey, trackFields);
+ altTrackRecords++;
+ } else alts.remove(indexKey);
+
if (heldBefore !== undefined) {
const r = captionReport.get(s.channelSlug) ?? { reread: 0, changed: 0, zeroToText: 0 };
r.reread++;
@@ -1211,10 +1272,15 @@ export async function buildIndex({
);
}
+ if (altTrackRecords > 0) {
+ log(`Alternate tracks: ${altTrackRecords} re-read record(s) hold a track whose words differ from the primary's.`);
+ }
+
for (const { pathKey, indexKey } of removed) {
sums.remove(indexKey);
cues.remove(indexKey);
subs.remove(indexKey);
+ alts.remove(indexKey);
digests.remove(indexKey);
byChannel.remove(indexToChannelKey(indexKey));
mtimes.remove(pathKey);
@@ -1254,6 +1320,7 @@ export async function buildIndex({
await sums.flushed;
await cues.flushed;
await subs.flushed;
+ await alts.flushed;
await digests.flushed;
await byChannel.flushed;
await mtimes.flushed;
@@ -1436,7 +1503,16 @@ export async function buildIndex({
const summary = sums.get(indexKey);
if (!summary) continue;
const cueList = cues.get(indexKey);
- const detail: TranscriptDetail = { ...summary, cues: cueList };
+ // `track` + `altTracks` only on a record that has an alternate, so every
+ // other record's page bytes are what they were.
+ const trackFields = alts.get(indexKey);
+ const detail: TranscriptDetail = {
+ ...summary,
+ cues: cueList,
+ ...(trackFields?.altTracks?.length
+ ? { track: trackFields.track, altTracks: trackFields.altTracks }
+ : {}),
+ };
const encoded = JSON.stringify(detail);
await writer.push(encoded, summary.id);
}
@@ -2434,6 +2510,9 @@ export async function buildIndex({
if (captionTrackDue && held.size === 0) {
await meta.put(CAPTION_TRACK_KEY, CAPTION_TRACK_RULE_VERSION);
}
+ if (altTracksDue && held.size === 0) {
+ await meta.put(ALT_TRACKS_KEY, ALT_TRACKS_VERSION);
+ }
await meta.flushed;
await root.close();
diff --git a/common/controller/buildIndexAltTracks.test.ts b/common/controller/buildIndexAltTracks.test.ts
@@ -0,0 +1,211 @@
+// Integration: ALTERNATE TRACKS (lib/captionTracks.ts) through the REAL
+// buildIndex over a temp corpus. A record whose served `en` says something its
+// en-orig does not carries that track on its transcript page — so a word only
+// in `en` is findable, with the track named — and a record whose tracks are
+// identical carries nothing extra; no English VTT is shipped again as a
+// subtitle track; and the one-shot pass (ALT_TRACKS_VERSION) re-reads exactly
+// the records that can hold an alternate.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildIndexAltTracks.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-alts-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_INDEX_ALLOW_HELD;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { open } = await import("lmdb");
+const { hitsAcrossTracks } = await import("../lib/captionTracks");
+
+const paths = getPaths();
+const CHANNEL = "example-channel";
+const SITE = "testsite";
+const DIFFER = "DifferTrack1"; // en-orig + an en that says other words
+const SAME = "SameTracks01"; // en-orig + an identical en
+const WHISPER = "Transcribed1"; // transcript.json + captions that differ
+const SPANISH = "SpanishSub01"; // en-orig + a Spanish subtitle track
+
+const fixture = (name: string) =>
+ readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", name), "utf8");
+const ROLLING = fixture("vtt-rolling.vtt");
+const vtt = (...lines: [string, string, string][]) =>
+ "WEBVTT\nKind: captions\nLanguage: en\n\n" +
+ lines.map(([a, b, t]) => `${a} --> ${b}\n${t}\n`).join("\n");
+// What the served `en` says: the same opening, then a word en-orig never has,
+// far from anything en-orig matches.
+const SERVED_EN = vtt(
+ ["00:00:03.080", "00:00:05.670", "are talking about the harbor"],
+ ["00:01:40.000", "00:01:43.000", "the zeppelin landed in nineteen thirty"],
+);
+const WHISPER_JSON = JSON.stringify({
+ transcription: [
+ { offsets: { from: 0, to: 2000 }, text: " a local transcription says hello" },
+ { offsets: { from: 2000, to: 4000 }, text: " and nothing else" },
+ ],
+});
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id);
+
+function seed(): void {
+ rmSync(paths.transcriptsDir, { recursive: true, force: true });
+ rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true });
+ writeFileSync(paths.settingsFile, "{}");
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: CHANNEL,
+ });
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+ for (const [i, id] of [DIFFER, SAME, WHISPER, SPANISH].entries()) {
+ writeJson(path.join(dirOf(id), "metadata.info.json"), {
+ id,
+ title: `Video ${id}`,
+ upload_date: `2026060${i + 1}`,
+ duration: 200,
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ }
+ writeFileSync(path.join(dirOf(DIFFER), "transcript.en-orig.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(DIFFER), "transcript.en.vtt"), SERVED_EN);
+ writeFileSync(path.join(dirOf(SAME), "transcript.en-orig.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(SAME), "transcript.en.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(WHISPER), "transcript.json"), WHISPER_JSON);
+ writeFileSync(path.join(dirOf(WHISPER), "transcript.en-orig.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(SPANISH), "transcript.en-orig.vtt"), ROLLING);
+ writeFileSync(
+ path.join(dirOf(SPANISH), "transcript.es.vtt"),
+ vtt(["00:00:01.000", "00:00:02.000", "hola a todos"]),
+ );
+}
+
+type Cue = { start: number; end: number; text: string };
+type Rec = {
+ id: string;
+ cues?: Cue[];
+ track?: string;
+ altTracks?: { track: string; cues: Cue[] }[];
+};
+
+// Every record of the channel's shared transcript pages, by id.
+function pageRecords(): Map<string, Rec> {
+ const dir = path.join(paths.exportSharedTranscriptsDir, CHANNEL);
+ const out = new Map<string, Rec>();
+ for (const name of readdirSync(dir)) {
+ if (!/^page-\d+\.json$/.test(name)) continue;
+ for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as Rec[]) {
+ out.set(r.id, r);
+ }
+ }
+ return out;
+}
+function subsTracks(): Record<string, string[]> {
+ const dir = path.join(paths.exportSharedSubsDir, CHANNEL);
+ const out: Record<string, string[]> = {};
+ if (!existsSync(dir)) return out;
+ for (const name of readdirSync(dir)) {
+ if (!/^page-\d+\.json$/.test(name)) continue;
+ for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as {
+ id: string;
+ tracks: Record<string, unknown>;
+ }[]) {
+ out[r.id] = Object.keys(r.tracks).sort();
+ }
+ }
+ return out;
+}
+
+async function runIndex(): Promise<string[]> {
+ const log: string[] = [];
+ await buildIndex({ paths, onLog: (s) => log.push(s) });
+ return log;
+}
+
+test("a differing served en rides on the page as an alternate; identical tracks add nothing", async () => {
+ seed();
+ await runIndex();
+ const recs = pageRecords();
+
+ const differ = recs.get(DIFFER)!;
+ assert.equal(differ.track, "en-orig");
+ assert.deepEqual(differ.altTracks?.map((t) => t.track), ["en"]);
+ // The word only `en` has is found — in `en`, named — and the opening both
+ // say is found once, in the primary.
+ const find = (q: string) =>
+ hitsAcrossTracks(differ, (cues) => cues.filter((c) => c.text.includes(q)));
+ assert.deepEqual(
+ find("zeppelin").map((h) => [h.track, Math.round(h.start)]),
+ [["en", 100]],
+ );
+ assert.deepEqual(find("harbor").map((h) => h.track), [undefined]);
+
+ // Identical tracks: no fields at all, so the record is what it always was.
+ const same = recs.get(SAME)!;
+ assert.equal("track" in same, false);
+ assert.equal("altTracks" in same, false);
+
+ // A transcription is the primary; the captions it replaced are an alternate.
+ const whisper = recs.get(WHISPER)!;
+ assert.equal(whisper.track, "transcription");
+ assert.equal(whisper.cues?.[0].text, "a local transcription says hello");
+ assert.deepEqual(whisper.altTracks?.map((t) => t.track), ["en-orig"]);
+
+ // No English VTT is a subtitle track any more; a Spanish one still is.
+ assert.deepEqual(subsTracks(), { [SPANISH]: ["es"] });
+});
+
+test("the one-shot pass re-reads exactly the records that can hold an alternate, once", async () => {
+ // The index as a build before alternate tracks left it: no `alts` records,
+ // no version.
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ try {
+ await root.openDB({ name: "alts", encoding: "msgpack" }).clearAsync();
+ await root.openDB({ name: "meta", encoding: "msgpack" }).remove("altTracks");
+ } finally {
+ await root.close();
+ }
+ const log = await runIndex();
+ // DIFFER, SAME (two English VTTs) and WHISPER (a transcription beside
+ // captions); not SPANISH.
+ assert.ok(log.includes("Alternate tracks v1: 3 record(s) re-read."), log.join("\n"));
+ assert.deepEqual(pageRecords().get(DIFFER)?.altTracks?.map((t) => t.track), ["en"]);
+
+ const again = await runIndex();
+ assert.equal(again.some((l) => /Alternate tracks v1/.test(l)), false, again.join("\n"));
+});
diff --git a/common/lib/captionTrack.test.ts b/common/lib/captionTrack.test.ts
@@ -97,14 +97,14 @@ test("readEnglishVttCues: every track empty is the first track with no cues; non
assert.equal(await readEnglishVttCues(videoDir({ "transcript.es.vtt": ROLLING })), null);
});
-test("readSubTracks lists the served en as an alternate beside an en-orig primary", async () => {
+test("readSubTracks lists no English VTT: those are caption tracks (lib/captionTracks.ts)", async () => {
const dir = videoDir({
"transcript.en.vtt": CUE_BLOCKS,
"transcript.en-orig.vtt": ROLLING,
"transcript.es.vtt": ROLLING,
});
const tracks = (await readSubTracks(dir)).map((t) => t.track).sort();
- assert.deepEqual(tracks, ["en", "es"]);
+ assert.deepEqual(tracks, ["es"]);
});
test("umtool's copy of the caption-track rule version matches", () => {
diff --git a/common/lib/captionTracks-server.ts b/common/lib/captionTracks-server.ts
@@ -0,0 +1,129 @@
+// The alternate tracks of one video dir, read from disk (lib/captionTracks.ts
+// says what an alternate is). SERVER-ONLY (node:fs).
+
+import path from "node:path";
+import { readdir, readFile } from "node:fs/promises";
+import { parseVtt, type Cue } from "./vtt";
+import { parseTranscriptJson } from "./whisper";
+import {
+ TRANSCRIPT_PIN_FILENAME,
+ VTT_FILENAME,
+ WHISPER_FILENAME,
+ englishVttsByPreference,
+ readEnglishVttCues,
+} from "./videoStatus";
+import {
+ PINNED_TRACK,
+ TRANSCRIPTION_TRACK,
+ distinctAltTracks,
+ trackOfVttFile,
+ type AltTrack,
+ type TrackFields,
+} from "./captionTracks";
+
+// The track id of an English caption file in a listing: its language code, or
+// `pinned` for transcript.en.vtt while the operator's pin stands.
+export function trackIdOfVtt(filename: string, entries: readonly string[]): string {
+ if (filename === VTT_FILENAME && entries.includes(TRANSCRIPT_PIN_FILENAME)) {
+ return PINNED_TRACK;
+ }
+ return trackOfVttFile(filename) ?? filename;
+}
+
+// Whether a listing can hold an alternate at all — no file is read. A caption
+// record needs two English VTTs; a transcribed one, one.
+export function mayHaveAltTracks(
+ primaryKind: "vtt" | "whisper",
+ entries: readonly string[],
+): boolean {
+ const n = englishVttsByPreference(entries).length;
+ return primaryKind === "whisper" ? n >= 1 : n >= 2;
+}
+
+// Every English VTT of a listing, parsed, in preference order. One that cannot
+// be read is left out.
+export async function readEnglishVttTracks(
+ videoDir: string,
+ entries: readonly string[],
+): Promise<{ filename: string; cues: Cue[] }[]> {
+ const out: { filename: string; cues: Cue[] }[] = [];
+ for (const filename of englishVttsByPreference(entries)) {
+ try {
+ out.push({
+ filename,
+ cues: parseVtt(await readFile(path.join(videoDir, filename), "utf8")),
+ });
+ } catch {
+ // unreadable: not a track
+ }
+ }
+ return out;
+}
+
+// A record's track fields: the primary's id and the English tracks whose words
+// differ from it. `primaryCues` is what the record's transcript holds (the
+// dedupe compares against it); for a caption record, the primary is the first
+// track in preference order that has a cue — the caption-track rule's content
+// fallback, so the id names the track the words really came from. Empty
+// (no fields) when nothing differs.
+export async function readTrackFields(
+ videoDir: string,
+ primaryKind: "vtt" | "whisper",
+ primaryCues: readonly Cue[] | undefined,
+ entries?: readonly string[],
+): Promise<TrackFields> {
+ const listing = entries ?? (await readdir(videoDir).catch(() => [] as string[]));
+ if (!mayHaveAltTracks(primaryKind, listing)) return {};
+ const vtts = await readEnglishVttTracks(videoDir, listing);
+ let primaryTrack: string;
+ let primary: readonly Cue[] | undefined = primaryCues;
+ let candidates: { filename: string; cues: Cue[] }[];
+ if (primaryKind === "whisper") {
+ primaryTrack = TRANSCRIPTION_TRACK;
+ candidates = vtts;
+ } else {
+ if (vtts.length === 0) return {};
+ const idx = Math.max(0, vtts.findIndex((t) => t.cues.length > 0));
+ primaryTrack = trackIdOfVtt(vtts[idx].filename, listing);
+ primary ??= vtts[idx].cues;
+ candidates = vtts.filter((_, i) => i !== idx);
+ }
+ const alts: AltTrack[] = distinctAltTracks(
+ primary,
+ candidates.map((t) => ({ track: trackIdOfVtt(t.filename, listing), cues: t.cues })),
+ );
+ if (alts.length === 0) return {};
+ return { track: primaryTrack, altTracks: alts };
+}
+
+// Every track of a video dir, read from disk, primary first: what the editor's
+// transcript reader shows and switches between. The primary is the record's
+// transcript by the same rules the index reads it with — a local
+// transcription (transcript.json) over captions, captions by the caption-track
+// rule — and the rest are the alternates readTrackFields keeps. Null when the
+// dir holds no transcript.
+export async function readVideoTracks(
+ videoDir: string,
+): Promise<{ tracks: AltTrack[] } | null> {
+ const entries = await readdir(videoDir).catch(() => [] as string[]);
+ let primary: AltTrack | null = null;
+ let kind: "vtt" | "whisper" = "vtt";
+ if (entries.includes(WHISPER_FILENAME)) {
+ try {
+ primary = {
+ track: TRANSCRIPTION_TRACK,
+ cues: parseTranscriptJson(await readFile(path.join(videoDir, WHISPER_FILENAME), "utf8")),
+ };
+ kind = "whisper";
+ } catch {
+ primary = null;
+ }
+ }
+ if (!primary) {
+ const read = await readEnglishVttCues(videoDir, entries);
+ if (!read) return null;
+ primary = { track: trackIdOfVtt(read.filename, entries), cues: read.cues };
+ }
+ const fields = await readTrackFields(videoDir, kind, primary.cues, entries);
+ return { tracks: [primary, ...(fields.altTracks ?? [])] };
+}
diff --git a/common/lib/captionTracks.test.ts b/common/lib/captionTracks.test.ts
@@ -0,0 +1,179 @@
+// The tracks of a transcript (lib/captionTracks.ts + captionTracks-server.ts):
+// labels from ids, which alternates are kept, and how a search finds a word
+// across them.
+//
+// Run with: node_modules/.bin/tsx --test common/lib/captionTracks.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import {
+ cuesOfTrack,
+ distinctAltTracks,
+ hitsAcrossTracks,
+ inTrackLabel,
+ recordTracks,
+ trackKind,
+ trackLabel,
+ uncoveredAltHits,
+} from "./captionTracks";
+import { readTrackFields, readVideoTracks } from "./captionTracks-server";
+import { TRANSCRIPT_PIN_FILENAME } from "./videoStatus";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "caption-tracks-"));
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const cue = (start: number, text: string) => ({ start, end: start + 2, text });
+const vtt = (...cues: [number, string][]) =>
+ "WEBVTT\n\n" +
+ cues
+ .map(([s, t]) => {
+ const ts = (x: number) => `00:${String(Math.floor(x / 60)).padStart(2, "0")}:${String(x % 60).padStart(2, "0")}.000`;
+ return `${ts(s)} --> ${ts(s + 2)}\n${t}\n`;
+ })
+ .join("\n");
+
+let n = 0;
+function videoDir(files: Record<string, string>): string {
+ const dir = path.join(ROOT, `v${n++}`);
+ mkdirSync(dir, { recursive: true });
+ for (const [name, body] of Object.entries(files)) writeFileSync(path.join(dir, name), body);
+ return dir;
+}
+
+test("labels are plain words derived from the track id", () => {
+ assert.equal(trackLabel("en-orig"), "original audio captions");
+ assert.equal(trackLabel("en"), "uploaded captions");
+ assert.equal(trackLabel("en-en-US"), "auto-translated captions");
+ assert.equal(trackLabel("en-GB"), "UK English captions");
+ assert.equal(trackLabel("en-x-foo"), "regional captions (en-x-foo)");
+ assert.equal(trackLabel("transcription"), "transcription");
+ assert.equal(trackLabel("pinned"), "chosen captions");
+ assert.equal(inTrackLabel("en"), "in uploaded captions");
+ assert.equal(trackKind("en-US"), "regional");
+ assert.equal(trackKind("live_chat"), "other");
+});
+
+test("an alternate is kept only where its words differ from the primary and every kept one", () => {
+ const primary = [cue(0, "hello there")];
+ const kept = distinctAltTracks(primary, [
+ { track: "en", cues: [cue(0, "hello there")] }, // identical words
+ { track: "en-GB", cues: [cue(5, "hello there")] }, // timing alone differs
+ { track: "en-US", cues: [cue(0, "hello their")] }, // other words: kept
+ { track: "en-en-US", cues: [cue(0, "hello their")] }, // same as en-US
+ { track: "en-CA", cues: [] }, // empty
+ ]);
+ assert.deepEqual(kept.map((t) => t.track), ["en-US"]);
+});
+
+test("recordTracks and cuesOfTrack: primary first; an unknown track is undefined", () => {
+ const rec = {
+ cues: [cue(0, "a")],
+ track: "en-orig",
+ altTracks: [{ track: "en", cues: [cue(0, "b")] }],
+ };
+ assert.deepEqual(recordTracks(rec), ["en-orig", "en"]);
+ assert.equal(cuesOfTrack(rec, null)?.[0].text, "a");
+ assert.equal(cuesOfTrack(rec, "en-orig")?.[0].text, "a");
+ assert.equal(cuesOfTrack(rec, "en")?.[0].text, "b");
+ assert.equal(cuesOfTrack(rec, "en-GB"), undefined);
+ assert.deepEqual(recordTracks({ cues: [] }), []);
+});
+
+test("a search finds a word every track says once, in the primary, and an alternate's own word there", () => {
+ const rec = {
+ cues: [cue(10, "the bridge opened"), cue(300, "and then we left")],
+ track: "en-orig",
+ altTracks: [
+ {
+ track: "en",
+ cues: [cue(11, "the bridge opened in 1932"), cue(200, "a zeppelin flew over the bridge")],
+ },
+ ],
+ };
+ const find = (q: string) =>
+ hitsAcrossTracks(rec, (cues) => cues.filter((c) => c.text.includes(q)));
+ assert.deepEqual(
+ find("bridge").map((h) => [h.start, h.track]),
+ [
+ [10, undefined],
+ [200, "en"],
+ ],
+ );
+ assert.deepEqual(find("1932"), [{ ...cue(11, "the bridge opened in 1932"), track: "en" }]);
+ assert.deepEqual(find("nothing"), []);
+ // A record with no alternates is its primary alone.
+ assert.deepEqual(
+ hitsAcrossTracks({ cues: rec.cues }, (cues) => cues.filter((c) => c.text.includes("left"))),
+ [cue(300, "and then we left")],
+ );
+});
+
+test("uncoveredAltHits drops an alternate hit within the window of a primary one", () => {
+ assert.deepEqual(
+ uncoveredAltHits([{ start: 100 }, { start: 500 }], [{ start: 90 }, { start: 130 }, { start: 515 }, { start: 900 }]),
+ [{ start: 130 }, { start: 900 }],
+ );
+ assert.deepEqual(uncoveredAltHits([], [{ start: 1 }]), [{ start: 1 }]);
+});
+
+test("readTrackFields: a differing en beside en-orig is an alternate; identical tracks are none", async () => {
+ const differ = videoDir({
+ "transcript.en-orig.vtt": vtt([1, "said words"]),
+ "transcript.en.vtt": vtt([1, "uploaded words"]),
+ });
+ assert.deepEqual(await readTrackFields(differ, "vtt", undefined), {
+ track: "en-orig",
+ altTracks: [{ track: "en", cues: [{ start: 1, end: 3, text: "uploaded words" }] }],
+ });
+ const same = videoDir({
+ "transcript.en-orig.vtt": vtt([1, "said words"]),
+ "transcript.en.vtt": vtt([1, "said words"]),
+ });
+ assert.deepEqual(await readTrackFields(same, "vtt", undefined), {});
+ // One English VTT: nothing to read.
+ const lone = videoDir({ "transcript.en.vtt": vtt([1, "x"]) });
+ assert.deepEqual(await readTrackFields(lone, "vtt", undefined), {});
+});
+
+test("readTrackFields: an empty en-orig falls through to en as the primary, as the rule reads it", async () => {
+ const dir = videoDir({
+ "transcript.en-orig.vtt": "WEBVTT\n\n",
+ "transcript.en.vtt": vtt([1, "served words"]),
+ "transcript.en-GB.vtt": vtt([1, "british words"]),
+ });
+ assert.deepEqual(await readTrackFields(dir, "vtt", undefined), {
+ track: "en",
+ altTracks: [{ track: "en-GB", cues: [{ start: 1, end: 3, text: "british words" }] }],
+ });
+});
+
+test("readTrackFields: the operator's pin names the primary `pinned`; its source is not repeated", async () => {
+ const dir = videoDir({
+ "transcript.en-orig.vtt": vtt([1, "said words"]),
+ "transcript.en.vtt": vtt([1, "said words"]), // the copy of en-orig the pin made
+ "transcript.en-US.vtt": vtt([1, "other words"]),
+ [TRANSCRIPT_PIN_FILENAME]: JSON.stringify({ from: "transcript.en-orig.vtt", pinnedAt: "" }),
+ });
+ assert.deepEqual(await readTrackFields(dir, "vtt", undefined), {
+ track: "pinned",
+ altTracks: [{ track: "en-US", cues: [{ start: 1, end: 3, text: "other words" }] }],
+ });
+});
+
+test("readVideoTracks: a transcription is the primary and its captions the alternate", async () => {
+ const dir = videoDir({
+ "transcript.json": JSON.stringify({
+ transcription: [{ offsets: { from: 0, to: 1000 }, text: " machine words" }],
+ }),
+ "transcript.en-orig.vtt": vtt([1, "caption words"]),
+ });
+ const read = await readVideoTracks(dir);
+ assert.deepEqual(read?.tracks.map((t) => [t.track, t.cues[0].text]), [
+ ["transcription", "machine words"],
+ ["en-orig", "caption words"],
+ ]);
+ assert.equal(await readVideoTracks(videoDir({ "metadata.info.json": "{}" })), null);
+});
diff --git a/common/lib/captionTracks.ts b/common/lib/captionTracks.ts
@@ -0,0 +1,202 @@
+// THE TRACKS OF A TRANSCRIPT — one notion of "track" for every reader.
+//
+// A record's transcript is read from ONE track, its primary, chosen by the
+// caption-track rule (lib/videoStatus.ts: en-orig first, the operator's pin
+// above all, a local transcription above captions). A record may hold other
+// English tracks beside it: the served `en`, a regional en-GB, an en→en
+// auto-translation, or the captions a local transcription replaced. Human
+// captions are not always a transcript of what was said, so those stay
+// readable and searchable as ALTERNATES — a viewer's choice, never a change to
+// the primary (the pin, transcript-pin.json, is how the primary changes).
+//
+// An alternate is kept only where its words differ from the primary's and from
+// every alternate kept before it: most served `en` tracks are byte-identical to
+// en-orig, and shipping or indexing those would double the corpus for nothing.
+//
+// Track ids are the caption's language code as its file names it
+// (transcript.<id>.vtt), plus two that no file names: `transcription` (a local
+// whisper/parakeet transcript) and `pinned` (transcript.en.vtt while the
+// operator's pin stands — a copy of whichever track was picked). Labels are
+// derived from the id, here and nowhere else, so the editor, the export site,
+// a search hit and the MCP all say the same words.
+//
+// Pure: no node built-ins, safe in a client bundle.
+
+import type { Cue } from "./vtt";
+
+export const TRANSCRIPTION_TRACK = "transcription";
+export const PINNED_TRACK = "pinned";
+
+export type AltTrack = { track: string; cues: Cue[] };
+
+// What a transcript record carries about its tracks. Both fields are OMITTED
+// when the record has no alternate that differs (the common case), so those
+// records' pages stay byte-identical to the ones already on disk.
+export type TrackFields = {
+ // The primary's track id.
+ track?: string;
+ // The English tracks whose words differ from the primary's, in preference
+ // order.
+ altTracks?: AltTrack[];
+};
+
+const REGIONS: Record<string, string> = {
+ US: "US",
+ GB: "UK",
+ UK: "UK",
+ CA: "Canadian",
+ AU: "Australian",
+ IE: "Irish",
+ IN: "Indian",
+ NZ: "New Zealand",
+ ZA: "South African",
+};
+
+// The kind of a track, from its id.
+export type TrackKind =
+ | "original"
+ | "uploaded"
+ | "regional"
+ | "translated"
+ | "transcription"
+ | "pinned"
+ | "other";
+
+export function trackKind(track: string): TrackKind {
+ if (track === TRANSCRIPTION_TRACK) return "transcription";
+ if (track === PINNED_TRACK) return "pinned";
+ if (track === "en-orig") return "original";
+ if (track === "en") return "uploaded";
+ if (/^en-en(?:-|$)/.test(track)) return "translated";
+ if (/^en-/.test(track)) return "regional";
+ return "other";
+}
+
+// A plain label for a track — what a person reads in the switcher and on a
+// search hit ("in uploaded captions").
+export function trackLabel(track: string): string {
+ switch (trackKind(track)) {
+ case "original":
+ return "original audio captions";
+ case "uploaded":
+ return "uploaded captions";
+ case "translated":
+ return "auto-translated captions";
+ case "transcription":
+ return "transcription";
+ case "pinned":
+ return "chosen captions";
+ case "regional": {
+ const region = track.slice(3);
+ const name = REGIONS[region.toUpperCase()];
+ return name ? `${name} English captions` : `regional captions (${track})`;
+ }
+ default:
+ return `captions (${track})`;
+ }
+}
+
+// The track id of a caption file name (transcript.<id>.vtt), or null.
+export function trackOfVttFile(filename: string): string | null {
+ const m = filename.match(/^transcript\.([^.]+)\.vtt$/);
+ return m ? m[1] : null;
+}
+
+// A cue list's words, for "do these two tracks say the same thing" — timing
+// alone is not a different transcript.
+export function cueWords(cues: readonly Cue[]): string {
+ return cues.map((c) => c.text).join("\n");
+}
+
+// The alternates worth keeping: tracks with at least one cue whose words differ
+// from the primary's and from every track kept before them. `candidates` in
+// preference order; the primary is not among them.
+export function distinctAltTracks(
+ primary: readonly Cue[] | undefined,
+ candidates: readonly AltTrack[],
+): AltTrack[] {
+ const seen = new Set<string>([cueWords(primary ?? [])]);
+ const out: AltTrack[] = [];
+ for (const c of candidates) {
+ if (c.cues.length === 0) continue;
+ const words = cueWords(c.cues);
+ if (seen.has(words)) continue;
+ seen.add(words);
+ out.push(c);
+ }
+ return out;
+}
+
+// The tracks a record offers, primary first: what a switcher lists. A record
+// with no alternates offers just its primary (or nothing, with no track id).
+export function recordTracks(rec: TrackFields): string[] {
+ if (!rec.altTracks || rec.altTracks.length === 0) return rec.track ? [rec.track] : [];
+ return [rec.track ?? "", ...rec.altTracks.map((t) => t.track)].filter((t) => t !== "");
+}
+
+// The cues of one track of a record: the primary when `track` is absent or
+// names it, an alternate when it names one, undefined when the record has no
+// such track.
+export function cuesOfTrack<R extends TrackFields & { cues?: Cue[] | undefined }>(
+ rec: R,
+ track: string | null | undefined,
+): Cue[] | undefined {
+ if (!track || track === rec.track) return rec.cues;
+ return rec.altTracks?.find((t) => t.track === track)?.cues;
+}
+
+// How far (seconds) a primary hit covers an alternate's. An alternate hit with
+// a primary hit this close is the same moment found twice — the primary's is
+// the one shown. Only a match the primary has nowhere near is the alternate's
+// to report.
+export const ALT_HIT_COVERED_SEC = 20;
+
+// Drop the alternate hits a primary hit already covers (ALT_HIT_COVERED_SEC).
+// Both lists carry `start` in seconds; the primary's needs no order.
+export function uncoveredAltHits<H extends { start: number }>(
+ primary: readonly { start: number }[],
+ alt: readonly H[],
+): H[] {
+ if (primary.length === 0) return [...alt];
+ const starts = primary.map((h) => h.start).sort((a, b) => a - b);
+ return alt.filter((h) => {
+ // Binary search for the nearest primary start.
+ let lo = 0;
+ let hi = starts.length - 1;
+ while (lo < hi) {
+ const mid = (lo + hi) >> 1;
+ if (starts[mid] < h.start) lo = mid + 1;
+ else hi = mid;
+ }
+ const near = [starts[lo], starts[lo - 1]].filter((s) => s !== undefined);
+ return !near.some((s) => Math.abs(s - h.start) <= ALT_HIT_COVERED_SEC);
+ });
+}
+
+// THE ONE RULE for matching a record's words across its tracks, used by every
+// search (the viewer's leaf pipeline, the MCP, the query-tree evaluator):
+// `find` is run over the primary's cues, then over each alternate's, and an
+// alternate's hit is kept only where no hit already kept is within
+// ALT_HIT_COVERED_SEC — so a word every track says is found once, in the
+// primary, and a word only an alternate says is found there, wearing that
+// alternate's track id. Ordered by start when an alternate adds anything.
+export function hitsAcrossTracks<H extends { start: number }>(
+ rec: TrackFields & { cues?: readonly Cue[] | undefined },
+ find: (cues: readonly Cue[]) => H[],
+): (H & { track?: string })[] {
+ const out: (H & { track?: string })[] = rec.cues ? find(rec.cues) : [];
+ let added = false;
+ for (const alt of rec.altTracks ?? []) {
+ for (const h of uncoveredAltHits(out, find(alt.cues))) {
+ out.push({ ...h, track: alt.track });
+ added = true;
+ }
+ }
+ if (added) out.sort((a, b) => a.start - b.start);
+ return out;
+}
+
+// "in uploaded captions" — what a hit from an alternate says about itself.
+export function inTrackLabel(track: string): string {
+ return `in ${trackLabel(track)}`;
+}
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -80,7 +80,10 @@ const SHARD_SCHEME = {
"<channel.manifests.transcripts> -> { pageCount, slugToPage: { <videoId>: <pageNumber> } }",
transcriptPage:
"/transcripts/<slug>/page-<NNNN>.json -> array of { id, title, uploadDate, " +
- "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }",
+ "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }; " +
+ "a record with other English caption tracks whose words differ from its transcript " +
+ "also carries `track` (the transcript's track id, e.g. \"en-orig\") and " +
+ "`altTracks: [{ track, cues }]` (e.g. \"en\", the uploaded captions) — absent otherwise",
subsManifest:
"<channel.manifests.subs> -> lighter list-view records under the same slugToPage scheme",
summariesIndex:
diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts
@@ -1,5 +1,6 @@
import path from "node:path";
import type { Cue } from "./vtt";
+import type { TrackFields } from "./captionTracks";
import { readFile } from "fs-extra";
import { getPaths } from "./paths";
@@ -69,9 +70,12 @@ export type DisplaySummary = {
curatedTags?: string[];
};
+// `track` / `altTracks` (lib/captionTracks.ts): the primary's track id and the
+// other English tracks whose words differ from it — present only on a record
+// that has such a track.
export type TranscriptDetail = TranscriptSummary & {
cues: Cue[] | undefined;
-};
+} & TrackFields;
export type TranscriptPage = TranscriptDetail[];
diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts
@@ -82,9 +82,9 @@ export type SubTrack = {
ext: "vtt" | "json" | "json3" | "srv1" | "srv2" | "srv3";
};
-// Treat any transcript.<x>.<y> file as a sub track when it isn't one of the
-// primary transcript outputs (transcript.en.vtt, transcript.json) or a
-// derived/auxiliary file (transcript.cues.json). Live chat lands as
+// Treat any transcript.<x>.<y> file as a sub track when it isn't a transcript
+// (transcript.json, or ANY English VTT — the primary and its alternate tracks,
+// lib/captionTracks.ts) or a derived/auxiliary file (transcript.cues.json). Live chat lands as
// transcript.live_chat.json; non-en languages as transcript.<lang>.vtt.
// Exported for lib/sidecar-server.ts, which refuses to declare a sidecar this
// matches (a sidecar so named would be read as a subtitle track).
@@ -274,15 +274,14 @@ export async function readVideoFiles(
export async function readSubTracks(videoDir: string): Promise<SubTrack[]> {
const entries = await readdir(videoDir).catch(() => [] as string[]);
- const primaryVtt = resolvePrimaryVtt(entries);
const tracks: SubTrack[] = [];
for (const entry of entries) {
if (entry === WHISPER_FILENAME) continue;
- // The resolved primary English VTT (transcript.en-orig.vtt, or e.g.
- // transcript.en-US.vtt when there is nothing better) is the main
- // transcript, not an alternate sub-track. Any other English track — the
- // served transcript.en.vtt beside an en-orig — is an alternate.
- if (entry === primaryVtt) continue;
+ // Every English VTT is a CAPTION track, not a sub track: the primary is
+ // the transcript, and the others are its alternate tracks, kept only
+ // where their words differ (lib/captionTracks.ts) — not shipped again,
+ // identical or not, as subtitles.
+ if (isEnglishVtt(entry)) continue;
if (entry === CUES_JSON_FILENAME) continue;
if (entry === LIVE_CHAT_CUES_FILENAME) continue;
const m = entry.match(SUB_FILE_RE);