commit 4974c94023d518fe75742c78080d16c20473a675
parent 9b649912101756b6a746db374b3b40af5966564d
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 08:47:18 -0400
vtt: read cue-block captions — a VTT with no word timing is no longer zero cues
parseVtt decides the shape once per document. A rolling auto-caption track
(inline <hh:mm:ss.mmm> word timing) is read as before: the tagged new line of
each cue. A document with no timing tag anywhere — uploaded captions, and the
en track YouTube serves for some livestream recordings (two lines a cue,
at each line end) — is read cue by cue, every line, tags stripped and
entities decoded after. It used to parse to no cues at all.
Fixtures keep both real shapes with the words replaced.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
4 files changed, 154 insertions(+), 5 deletions(-)
diff --git a/common/lib/__fixtures__/vtt-cue-blocks.vtt b/common/lib/__fixtures__/vtt-cue-blocks.vtt
@@ -0,0 +1,29 @@
+WEBVTT
+Kind: captions
+Language: en
+
+00:00:00.960 --> 00:00:06.000
+welcome back everyone today we are talking
+about the harbor bridge finally finally
+
+00:00:06.000 --> 00:00:09.800
+I have been teasing that all week and let me
+just say to every reader out there that
+
+00:00:09.800 --> 00:00:16.200
+no matter what you build in life you never
+ever skip the load test plus later on
+
+00:00:16.200 --> 00:00:20.383
+we will look at the tide tables for the
+north pier & the <old> lighthouse
+
+00:00:20.383 --> 00:00:20.393
+[Applause]
+
+00:00:20.393 --> 00:00:20.400
+[Music]
+
+00:00:20.400 --> 00:00:26.640
+<i>right</i> to begin let us start with the
+end the conclusion here I have said this
diff --git a/common/lib/__fixtures__/vtt-rolling.vtt b/common/lib/__fixtures__/vtt-rolling.vtt
@@ -0,0 +1,31 @@
+WEBVTT
+Kind: captions
+Language: en
+
+00:00:00.960 --> 00:00:03.070 align:start position:0%
+
+welcome<00:00:01.160><c> back</c><00:00:01.319><c> everyone</c><00:00:01.520><c> today</c><00:00:01.760><c> we</c>
+
+00:00:03.070 --> 00:00:03.080 align:start position:0%
+welcome back everyone today we
+
+
+00:00:03.080 --> 00:00:05.670 align:start position:0%
+welcome back everyone today we
+are<00:00:03.240><c> talking</c><00:00:03.639><c> about</c><00:00:04.319><c> the</c><00:00:04.759><c> harbor</c>
+
+00:00:05.670 --> 00:00:05.680 align:start position:0%
+are talking about the harbor
+
+
+00:00:05.680 --> 00:00:06.869 align:start position:0%
+are talking about the harbor
+bridge<00:00:06.000><c> finally</c><00:00:06.120><c> finally</c>
+
+00:00:06.869 --> 00:00:06.879 align:start position:0%
+bridge finally finally
+
+
+00:00:06.879 --> 00:00:09.310 align:start position:0%
+bridge finally finally
+I<00:00:07.000><c> have</c><00:00:07.120><c> been</c><00:00:07.279><c> teasing</c><00:00:07.560><c> that</c>
diff --git a/common/lib/vtt.test.ts b/common/lib/vtt.test.ts
@@ -1,6 +1,14 @@
import { test } from "node:test";
import assert from "node:assert/strict";
-import { cuesToText, cuesToSrt, type Cue } from "./vtt";
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { cuesToText, cuesToSrt, hasWordTiming, parseVtt, type Cue } from "./vtt";
+
+// Both fixtures keep the structure of real YouTube downloads (one rolling
+// auto-caption track, one served `en` track of a livestream VOD) with the
+// words replaced.
+const fixture = (name: string) =>
+ readFileSync(path.join(import.meta.dirname, "__fixtures__", name), "utf8");
// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/vtt.test.ts
@@ -37,3 +45,53 @@ test("cuesToSrt clamps negative times to zero", () => {
const neg = [{ start: -3, end: -1, text: "neg" }];
assert.equal(cuesToSrt(neg), "1\n00:00:00,000 --> 00:00:00,000\nneg\n");
});
+
+test("parseVtt reads a rolling auto-caption track one new line per cue", () => {
+ const cues = parseVtt(fixture("vtt-rolling.vtt"));
+ // Not asserted: the FIRST cue, whose carried-over line is a lone space —
+ // the body read stops at a whitespace-only line, so that cue reads as empty.
+ // That is the rolling parse as it stands, unchanged here.
+ assert.deepEqual(
+ cues.slice(-3).map((c) => c.text),
+ [
+ "are talking about the harbor",
+ "bridge finally finally",
+ "I have been teasing that",
+ ],
+ );
+ assert.equal(cues.at(-3)!.start, 3.08);
+});
+
+test("parseVtt reads a cue-block track (no word timing, line ends) — every line of every cue", () => {
+ const src = fixture("vtt-cue-blocks.vtt");
+ assert.equal(hasWordTiming(src), false);
+ const cues = parseVtt(src);
+ assert.equal(cues.length, 7);
+ assert.deepEqual(cues[0], {
+ start: 0.96,
+ end: 6,
+ text: "welcome back everyone today we are talking about the harbor bridge finally finally",
+ });
+ // Entities decoded after the tags are stripped: an escaped "<old>" is text.
+ assert.equal(cues[3].text, "we will look at the tide tables for the north pier & the <old> lighthouse");
+ assert.equal(cues[4].text, "[Applause]");
+ // A real tag is stripped.
+ assert.equal(cues[6].text, "right to begin let us start with the end the conclusion here I have said this");
+});
+
+test("parseVtt: the shape is decided per document — a rolling track's untagged repeat cues are not read as text", () => {
+ const src = fixture("vtt-rolling.vtt");
+ assert.equal(hasWordTiming(src), true);
+ // Seven cues in the file; the repeats add nothing.
+ assert.equal(parseVtt(src).length, 3);
+});
+
+test("parseVtt reads cue identifiers and settings as structure, not text", () => {
+ const src =
+ "WEBVTT\n\nNOTE a comment\n\n1\n00:00:01.000 --> 00:00:02.500 line:90%\n<v Host>First line\n\n" +
+ "2\n00:00:03.000 --> 00:00:04.000\nSecond line\n";
+ assert.deepEqual(parseVtt(src), [
+ { start: 1, end: 2.5, text: "First line" },
+ { start: 3, end: 4, text: "Second line" },
+ ]);
+});
diff --git a/common/lib/vtt.ts b/common/lib/vtt.ts
@@ -2,6 +2,13 @@ export type Cue = { start: number; end: number; text: string };
const TIMING_TAG_RE = /<\d{2}:\d{2}:\d{2}\.\d{3}>/;
+// Whether VTT text carries inline word timing (<hh:mm:ss.mmm>) — the mark of
+// YouTube's rolling auto-caption shape, which parseVtt reads differently from
+// plain cue blocks.
+export function hasWordTiming(src: string): boolean {
+ return TIMING_TAG_RE.test(src);
+}
+
function parseTimestamp(ts: string): number {
const m = ts.match(/(\d+):(\d+):(\d+)\.(\d+)/);
if (!m) return 0;
@@ -27,9 +34,32 @@ function stripTags(s: string): string {
.trim();
}
+// Strip EVERY tag from a plain cue-block line (<i>, <b>, <v Speaker>, a stray
+// <c>), then decode entities — in that order, so an escaped "<3" survives as
+// text instead of being read as the start of a tag.
+function stripPlainLine(s: string): string {
+ return decodeEntities(s.replace(/<[^>]*>/g, ""))
+ .replace(/\s+/g, " ")
+ .trim();
+}
+
+// Two VTT shapes, decided once per DOCUMENT:
+//
+// ROLLING (YouTube auto-captions, word timing). Each cue repeats the line
+// before it and adds one new line carrying inline <hh:mm:ss.mmm><c>…</c> word
+// timing; between them sits a ~10 ms cue that repeats the text with no tags.
+// Only the tagged line is new, so only it is kept — reading every line would
+// say each sentence two or three times.
+//
+// CUE BLOCKS (uploaded captions, and the `en` track YouTube serves for some
+// livestream VODs: two lines a cue, ` ` at each line end, no word
+// timing). Every line of a cue is its text. A document of this shape has no
+// timing tag anywhere, which is what tells the two apart — read as ROLLING it
+// came out as zero cues, and the video as textless.
export function parseVtt(src: string): Cue[] {
const lines = src.replace(/\r\n/g, "\n").split("\n");
const cues: Cue[] = [];
+ const rolling = hasWordTiming(src);
let i = 0;
while (i < lines.length) {
@@ -50,12 +80,13 @@ export function parseVtt(src: string): Cue[] {
i++;
}
- const taggedLines = body.filter((l) => TIMING_TAG_RE.test(l));
let text: string;
- if (taggedLines.length > 0) {
- text = stripTags(taggedLines[taggedLines.length - 1]);
+ if (!rolling) {
+ text = body.map(stripPlainLine).filter(Boolean).join(" ");
} else {
- continue;
+ const taggedLines = body.filter((l) => TIMING_TAG_RE.test(l));
+ if (taggedLines.length === 0) continue;
+ text = stripTags(taggedLines[taggedLines.length - 1]);
}
if (text) cues.push({ start, end, text });