// The tracks of a transcript (lib/captionTracks.ts + captionTracks-server.ts): // labels from ids, which alternates are kept, and how a search finds a word // across them. // // Run with: node_modules/.bin/tsx --test common/lib/captionTracks.test.ts import { after, test } from "node:test"; import assert from "node:assert/strict"; import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; import { cuesOfTrack, distinctAltTracks, hitsAcrossTracks, inTrackLabel, recordTracks, trackKind, trackLabel, trackLabels, uncoveredAltHits, } from "./captionTracks"; import { readTrackFields, readVideoTracks } from "./captionTracks-server"; import { TRANSCRIPT_PIN_FILENAME } from "./videoStatus"; const ROOT = mkdtempSync(path.join(tmpdir(), "caption-tracks-")); after(() => rmSync(ROOT, { recursive: true, force: true })); const cue = (start: number, text: string) => ({ start, end: start + 2, text }); const vtt = (...cues: [number, string][]) => "WEBVTT\n\n" + cues .map(([s, t]) => { const ts = (x: number) => `00:${String(Math.floor(x / 60)).padStart(2, "0")}:${String(x % 60).padStart(2, "0")}.000`; return `${ts(s)} --> ${ts(s + 2)}\n${t}\n`; }) .join("\n"); let n = 0; function videoDir(files: Record): string { const dir = path.join(ROOT, `v${n++}`); mkdirSync(dir, { recursive: true }); for (const [name, body] of Object.entries(files)) writeFileSync(path.join(dir, name), body); return dir; } test("labels are plain words derived from the track id", () => { assert.equal(trackLabel("en-orig"), "original audio captions"); assert.equal(trackLabel("en"), "uploaded captions"); assert.equal(trackLabel("en-en-US"), "auto-translated captions"); assert.equal(trackLabel("en-GB"), "UK English captions"); assert.equal(trackLabel("en-NG"), "regional captions (en-NG)"); assert.equal(trackLabel("en-US-orig"), "original audio captions (US)"); // A track YouTube named itself: another uploaded one. assert.equal(trackLabel("en-uYU-mmqFLq8"), "other uploaded captions"); assert.equal(trackKind("en-JkeT_87f4cc"), "uploaded"); assert.deepEqual(trackLabels(["en-orig", "en-aa-1", "en-bb-2"]), [ "original audio captions", "other uploaded captions (en-aa-1)", "other uploaded captions (en-bb-2)", ]); assert.equal(trackLabel("transcription"), "transcription"); assert.equal(trackLabel("pinned"), "chosen captions"); assert.equal(inTrackLabel("en"), "in uploaded captions"); assert.equal(trackKind("en-US"), "regional"); assert.equal(trackKind("live_chat"), "other"); }); test("an alternate is kept only where its words differ from the primary and every kept one", () => { const primary = [cue(0, "hello there")]; const kept = distinctAltTracks(primary, [ { track: "en", cues: [cue(0, "hello there")] }, // identical words { track: "en-GB", cues: [cue(5, "hello there")] }, // timing alone differs { track: "en-US", cues: [cue(0, "hello their")] }, // other words: kept { track: "en-en-US", cues: [cue(0, "hello their")] }, // same as en-US { track: "en-CA", cues: [] }, // empty ]); assert.deepEqual(kept.map((t) => t.track), ["en-US"]); }); test("recordTracks and cuesOfTrack: primary first; an unknown track is undefined", () => { const rec = { cues: [cue(0, "a")], track: "en-orig", altTracks: [{ track: "en", cues: [cue(0, "b")] }], }; assert.deepEqual(recordTracks(rec), ["en-orig", "en"]); assert.equal(cuesOfTrack(rec, null)?.[0].text, "a"); assert.equal(cuesOfTrack(rec, "en-orig")?.[0].text, "a"); assert.equal(cuesOfTrack(rec, "en")?.[0].text, "b"); assert.equal(cuesOfTrack(rec, "en-GB"), undefined); assert.deepEqual(recordTracks({}), []); }); test("a search finds a word every track says once, in the primary, and an alternate's own word there", () => { const rec = { cues: [cue(10, "the bridge opened"), cue(300, "and then we left")], track: "en-orig", altTracks: [ { track: "en", cues: [cue(11, "the bridge opened in 1932"), cue(200, "a zeppelin flew over the bridge")], }, ], }; const find = (q: string) => hitsAcrossTracks(rec, (cues) => cues.filter((c) => c.text.includes(q))); assert.deepEqual( find("bridge").map((h) => [h.start, h.track]), [ [10, undefined], [200, "en"], ], ); assert.deepEqual(find("1932"), [{ ...cue(11, "the bridge opened in 1932"), track: "en" }]); assert.deepEqual(find("nothing"), []); // A record with no alternates is its primary alone. assert.deepEqual( hitsAcrossTracks({ cues: rec.cues }, (cues) => cues.filter((c) => c.text.includes("left"))), [cue(300, "and then we left")], ); }); test("uncoveredAltHits drops an alternate hit within the window of a primary one", () => { assert.deepEqual( uncoveredAltHits([{ start: 100 }, { start: 500 }], [{ start: 90 }, { start: 130 }, { start: 515 }, { start: 900 }]), [{ start: 130 }, { start: 900 }], ); assert.deepEqual(uncoveredAltHits([], [{ start: 1 }]), [{ start: 1 }]); }); test("readTrackFields: a differing en beside en-orig is an alternate; identical tracks are none", async () => { const differ = videoDir({ "transcript.en-orig.vtt": vtt([1, "said words"]), "transcript.en.vtt": vtt([1, "uploaded words"]), }); assert.deepEqual(await readTrackFields(differ, "vtt", undefined), { track: "en-orig", altTracks: [{ track: "en", cues: [{ start: 1, end: 3, text: "uploaded words" }] }], }); const same = videoDir({ "transcript.en-orig.vtt": vtt([1, "said words"]), "transcript.en.vtt": vtt([1, "said words"]), }); assert.deepEqual(await readTrackFields(same, "vtt", undefined), {}); // One English VTT: nothing to read. const lone = videoDir({ "transcript.en.vtt": vtt([1, "x"]) }); assert.deepEqual(await readTrackFields(lone, "vtt", undefined), {}); }); test("readTrackFields: an empty en-orig falls through to en as the primary, as the rule reads it", async () => { const dir = videoDir({ "transcript.en-orig.vtt": "WEBVTT\n\n", "transcript.en.vtt": vtt([1, "served words"]), "transcript.en-GB.vtt": vtt([1, "british words"]), }); assert.deepEqual(await readTrackFields(dir, "vtt", undefined), { track: "en", altTracks: [{ track: "en-GB", cues: [{ start: 1, end: 3, text: "british words" }] }], }); }); test("readTrackFields: the operator's pin names the primary `pinned`; its source is not repeated", async () => { const dir = videoDir({ "transcript.en-orig.vtt": vtt([1, "said words"]), "transcript.en.vtt": vtt([1, "said words"]), // the copy of en-orig the pin made "transcript.en-US.vtt": vtt([1, "other words"]), [TRANSCRIPT_PIN_FILENAME]: JSON.stringify({ from: "transcript.en-orig.vtt", pinnedAt: "" }), }); assert.deepEqual(await readTrackFields(dir, "vtt", undefined), { track: "pinned", altTracks: [{ track: "en-US", cues: [{ start: 1, end: 3, text: "other words" }] }], }); }); test("readVideoTracks: a transcription is the primary and its captions the alternate", async () => { const dir = videoDir({ "transcript.json": JSON.stringify({ transcription: [{ offsets: { from: 0, to: 1000 }, text: " machine words" }], }), "transcript.en-orig.vtt": vtt([1, "caption words"]), }); const read = await readVideoTracks(dir); assert.deepEqual(read?.tracks.map((t) => [t.track, t.cues[0].text]), [ ["transcription", "machine words"], ["en-orig", "caption words"], ]); assert.equal(await readVideoTracks(videoDir({ "metadata.info.json": "{}" })), null); });