import { test } from "node:test"; import assert from "node:assert/strict"; import { readFileSync } from "node:fs"; import path from "node:path"; import { cuesToText, cuesToSrt, hasWordTiming, parseVtt, type Cue } from "./vtt"; // Both fixtures keep the structure of real YouTube downloads (one rolling // auto-caption track, one served `en` track of a livestream VOD) with the // words replaced. const fixture = (name: string) => readFileSync(path.join(import.meta.dirname, "__fixtures__", name), "utf8"); // Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/vtt.test.ts const cues: Cue[] = [ { start: 0, end: 2.5, text: "Hello there" }, { start: 3661.25, end: 3663, text: "General Kenobi" }, ]; test("cuesToText writes one line per cue, trailing newline", () => { assert.equal(cuesToText(cues), "Hello there\nGeneral Kenobi\n"); }); test("cuesToText on empty cues is a lone newline", () => { assert.equal(cuesToText([]), "\n"); }); test("cuesToSrt numbers blocks and formats HH:MM:SS,mmm timing", () => { const srt = cuesToSrt(cues); assert.equal( srt, "1\n00:00:00,000 --> 00:00:02,500\nHello there\n" + "\n" + "2\n01:01:01,250 --> 01:01:03,000\nGeneral Kenobi\n", ); }); test("cuesToSrt falls back to start when end is missing", () => { const noEnd = [{ start: 5, end: undefined as unknown as number, text: "x" }]; const srt = cuesToSrt(noEnd); assert.equal(srt, "1\n00:00:05,000 --> 00:00:05,000\nx\n"); }); test("cuesToSrt clamps negative times to zero", () => { const neg = [{ start: -3, end: -1, text: "neg" }]; assert.equal(cuesToSrt(neg), "1\n00:00:00,000 --> 00:00:00,000\nneg\n"); }); test("parseVtt reads a rolling auto-caption track one new line per cue", () => { const cues = parseVtt(fixture("vtt-rolling.vtt")); // Not asserted: the FIRST cue, whose carried-over line is a lone space — // the body read stops at a whitespace-only line, so that cue reads as empty. // That is the rolling parse as it stands, unchanged here. assert.deepEqual( cues.slice(-3).map((c) => c.text), [ "are talking about the harbor", "bridge finally finally", "I have been teasing that", ], ); assert.equal(cues.at(-3)!.start, 3.08); }); test("parseVtt reads a cue-block track (no word timing,   line ends) — every line of every cue", () => { const src = fixture("vtt-cue-blocks.vtt"); assert.equal(hasWordTiming(src), false); const cues = parseVtt(src); assert.equal(cues.length, 7); assert.deepEqual(cues[0], { start: 0.96, end: 6, text: "welcome back everyone today we are talking about the harbor bridge finally finally", }); // Entities decoded after the tags are stripped: an escaped "" is text. assert.equal(cues[3].text, "we will look at the tide tables for the north pier & the lighthouse"); assert.equal(cues[4].text, "[Applause]"); // A real tag is stripped. assert.equal(cues[6].text, "right to begin let us start with the end the conclusion here I have said this"); }); test("parseVtt: the shape is decided per document — a rolling track's untagged repeat cues are not read as text", () => { const src = fixture("vtt-rolling.vtt"); assert.equal(hasWordTiming(src), true); // Seven cues in the file; the repeats add nothing. assert.equal(parseVtt(src).length, 3); }); test("parseVtt reads cue identifiers and settings as structure, not text", () => { const src = "WEBVTT\n\nNOTE a comment\n\n1\n00:00:01.000 --> 00:00:02.500 line:90%\nFirst line\n\n" + "2\n00:00:03.000 --> 00:00:04.000\nSecond line\n"; assert.deepEqual(parseVtt(src), [ { start: 1, end: 2.5, text: "First line" }, { start: 3, end: 4, text: "Second line" }, ]); });