import { test } from "node:test"; import assert from "node:assert/strict"; import { parseChapters, parseTags, snapToCueStart } from "./digestParse"; import type { DigestChunkOutput } from "./digestParse"; import type { Cue } from "./vtt"; // Each guard here corresponds to a failure MEASURED on qwen2.5:7b before the // prompt/schema were hardened, so these tests are regression pins for real // output, not hypotheticals. function cue(start: number, text = `t${start}`): Cue { return { start, end: start + 5, text }; } // A 20-minute transcript with a cue every 10s. const CUES: Cue[] = Array.from({ length: 120 }, (_, i) => cue(i * 10)); function chunk( chapters: unknown[], opts: { index?: number; start?: number; end?: number } = {}, ): DigestChunkOutput { return { index: opts.index ?? 0, startSeconds: opts.start ?? 0, endSeconds: opts.end ?? 1200, data: { chapters }, }; } test("guard 1: rejects a malformed timestamp and records it verbatim", () => { // ":00:27" is the exact shape the naive prompt produced. const { chapters, warnings } = parseChapters( [ chunk([ { start: ":00:27", title: "Bad stamp" }, { start: "00:01:00", title: "Good stamp" }, ]), ], CUES, ); assert.deepEqual( chapters.map((c) => c.title), ["Good stamp"], ); const w = warnings.find((x) => x.code === "malformed-timestamp"); assert.ok(w, "the rejection is recorded, never silently dropped"); assert.equal(w?.value, ":00:27"); assert.equal(w?.section, "chapters"); assert.equal(w?.chunk, 0); }); test("guard 1: rejects single-digit fields the regex pin forbids", () => { const { chapters, warnings } = parseChapters( [chunk([{ start: "1:2:3", title: "Loose stamp" }])], CUES, ); assert.equal(chapters.length, 0); assert.equal(warnings[0].code, "malformed-timestamp"); }); test("guard 1: rejects an in-shape stamp with an impossible field", () => { const { chapters, warnings } = parseChapters( [chunk([{ start: "00:99:00", title: "Ninety-nine minutes" }])], CUES, ); assert.equal(chapters.length, 0); assert.equal(warnings[0].code, "malformed-timestamp"); }); test("guard 2: flags a title that drifted out of English", () => { // The measured drift was into Chinese on an 8.2k-token chunk. const { chapters, warnings } = parseChapters( [ chunk([ { start: "00:00:10", title: "法庭文件截止日期" }, { start: "00:02:00", title: "Court filing deadlines" }, ]), ], CUES, ); assert.deepEqual( chapters.map((c) => c.title), ["Court filing deadlines"], ); const w = warnings.find((x) => x.code === "language-drift"); assert.ok(w); assert.equal(w?.value, "法庭文件截止日期"); }); test("guard 3: clamps to the CHUNK's range, not the video's", () => { // The measured failure: 01:10:29 emitted for an input spanning 00:04:45 to // 00:15:36. The video was 176 minutes long, so a whole-video range check would // have ACCEPTED it. This is why the clamp is per-chunk. const { chapters, warnings } = parseChapters( [ chunk([{ start: "01:10:29", title: "Way past the end" }], { start: 285, end: 936, }), ], CUES, ); assert.equal(chapters.length, 0); const w = warnings.find((x) => x.code === "out-of-range"); assert.ok(w); assert.equal(w?.value, "01:10:29"); assert.match(w?.detail ?? "", /00:04:45/); }); test("guard 3: accepts a start inside the chunk's own range", () => { const { chapters } = parseChapters( [ chunk([{ start: "00:05:00", title: "Inside the window" }], { start: 285, end: 936, }), ], CUES, ); assert.equal(chapters.length, 1); assert.equal(chapters[0].title, "Inside the window"); }); test("guard 4a: drops a non-monotonic entry within a chunk", () => { const { chapters, warnings } = parseChapters( [ chunk([ { start: "00:01:00", title: "First" }, { start: "00:00:30", title: "Backwards" }, { start: "00:02:00", title: "Third" }, ]), ], CUES, ); assert.deepEqual( chapters.map((c) => c.title), ["First", "Third"], ); assert.ok(warnings.some((w) => w.code === "non-monotonic")); }); test("guard 4b: de-dups equivalent chapters across a chunk seam", () => { const { chapters, warnings } = parseChapters( [ chunk([{ start: "00:05:00", title: "Court filing deadlines" }], { index: 0, start: 0, end: 400, }), chunk([{ start: "00:05:10", title: "court filing deadlines." }], { index: 1, start: 300, end: 700, }), ], CUES, ); assert.equal(chapters.length, 1, "the overlap's duplicate view is collapsed"); assert.ok(warnings.some((w) => w.code === "seam-duplicate")); }); test("guard 4b: keeps a genuinely different topic near a seam", () => { const { chapters } = parseChapters( [ chunk([{ start: "00:05:00", title: "Court filing deadlines" }], { index: 0, start: 0, end: 400, }), chunk([{ start: "00:05:20", title: "Jury selection" }], { index: 1, start: 300, end: 700, }), ], CUES, ); assert.equal(chapters.length, 2); }); test("guard 4b: collapses two chunks that snap onto the same cue", () => { const { chapters, warnings } = parseChapters( [ chunk([{ start: "00:05:00", title: "Alpha" }], { index: 0, end: 700 }), chunk([{ start: "00:05:02", title: "Completely unrelated beta" }], { index: 1, end: 700, }), ], CUES, ); assert.equal(chapters.length, 1); const w = warnings.find((w) => w.code === "seam-duplicate"); assert.match(w?.detail ?? "", /same start/); }); test("guard 5: snaps a start onto the nearest cue boundary", () => { // 00:05:04 (304s) sits between cues at 300s and 310s; 300 is nearer. const { chapters } = parseChapters( [chunk([{ start: "00:05:04", title: "Between cues" }])], CUES, ); assert.equal(chapters[0].start, 300); assert.equal(chapters[0].clock, "00:05:04", "the model's own stamp is kept for audit"); assert.equal(chapters[0].id, "c300", "the id derives from the SNAPPED start"); }); test("guard 5: snaps forward when the later cue is nearer", () => { const { chapters } = parseChapters( [chunk([{ start: "00:05:08", title: "Nearer the next cue" }])], CUES, ); assert.equal(chapters[0].start, 310); }); test("snapToCueStart handles a time before the first cue", () => { assert.equal(snapToCueStart([cue(40), cue(50)], 5), 40); }); test("snapToCueStart handles an empty cue list", () => { assert.equal(snapToCueStart([], 42.7), 42); }); test("every kept chapter is marked decidedBy: ai", () => { const { chapters } = parseChapters( [chunk([{ start: "00:01:00", title: "Machine-decided" }])], CUES, ); assert.equal(chapters[0].decidedBy, "ai"); }); test("a chunk with no chapters array is recorded as parse-failed", () => { const { chapters, warnings } = parseChapters( [{ index: 0, startSeconds: 0, endSeconds: 600, data: { nope: true } }], CUES, ); assert.equal(chapters.length, 0); assert.equal(warnings[0].code, "parse-failed"); }); test("an empty chapters array is recorded as empty-output", () => { const { warnings } = parseChapters([chunk([])], CUES); assert.equal(warnings[0].code, "empty-output"); }); test("an entry with no title is recorded, not silently dropped", () => { const { chapters, warnings } = parseChapters( [chunk([{ start: "00:01:00", title: " " }])], CUES, ); assert.equal(chapters.length, 0); assert.equal(warnings[0].code, "empty-title"); }); // --------------------------------------------------------------------------- // Tags // --------------------------------------------------------------------------- test("parseTags lowercases, dedups across chunks, and keeps ids stable", () => { const { tags } = parseTags([ { index: 0, startSeconds: 0, endSeconds: 600, data: { tags: ["Court Filings", "appeals"] } }, { index: 1, startSeconds: 500, endSeconds: 1200, data: { tags: ["court filings", "sentencing"] } }, ]); assert.deepEqual( tags.map((t) => t.tag), ["appeals", "court filings", "sentencing"], ); assert.equal(tags[1].id, "tcourt-filings"); }); test("parseTags does NOT warn about an expected overlap duplicate", () => { const { warnings } = parseTags([ { index: 0, startSeconds: 0, endSeconds: 600, data: { tags: ["appeals"] } }, { index: 1, startSeconds: 500, endSeconds: 1200, data: { tags: ["appeals"] } }, ]); assert.equal(warnings.length, 0); }); test("parseTags flags a drifted tag", () => { const { tags, warnings } = parseTags([ { index: 0, startSeconds: 0, endSeconds: 600, data: { tags: ["上訴", "appeals"] } }, ]); assert.deepEqual( tags.map((t) => t.tag), ["appeals"], ); assert.equal(warnings[0].code, "language-drift"); }); // --------------------------------------------------------------------------- // chunk-local timestamps // // These pin the EXACT failure that motivated the mode. Digesting // community-notes/v2chrch (2.3 h, 3505 cues -> 3 chunks), the third chunk — // range 01:31:43-02:16:09 — came back with nine starts of 00:00:00, 00:03:54, // 00:12:26 ...: the model had reverted to counting from zero. The per-chunk // clamp rejected all nine, so the chunk yielded nothing. // --------------------------------------------------------------------------- // The real third chunk's range, to the second. const CHUNK3_START = 91 * 60 + 43; // 01:31:43 = 5503 const CHUNK3_END = 2 * 3600 + 16 * 60 + 9; // 02:16:09 = 8169 // Cues every 10s across the whole 2.3h video, so the snap has real boundaries. const LONG_CUES: Cue[] = Array.from({ length: 830 }, (_, i) => cue(i * 10)); function chunk3(chapters: unknown[], mode?: "absolute" | "chunk-local"): DigestChunkOutput { return { index: 2, startSeconds: CHUNK3_START, endSeconds: CHUNK3_END, data: { chapters }, ...(mode ? { timestampMode: mode } : {}), }; } test("chunk-local: a 00:03:00 start in the third chunk resolves to 01:34:43", () => { const { chapters, warnings } = parseChapters( [chunk3([{ start: "00:03:00", title: "Filing deadlines" }], "chunk-local")], LONG_CUES, ); assert.equal(chapters.length, 1, "the offset is added back, so it is in range"); // 5503 + 180 = 5683, snapped to the nearest cue boundary (5680). assert.equal(chapters[0].start, 5680); // `clock` is stored in REAL video time, never in the numbering the model used. assert.equal(chapters[0].clock, "01:34:43"); assert.equal( warnings.length, 0, "nothing is rejected: this is exactly the output the mode exists to accept", ); }); test("absolute: the same 00:03:00 is rejected as out-of-range", () => { const { chapters, warnings } = parseChapters( [chunk3([{ start: "00:03:00", title: "Filing deadlines" }])], LONG_CUES, ); assert.equal(chapters.length, 0); const w = warnings.find((x) => x.code === "out-of-range"); assert.ok(w, "the per-chunk clamp is what caught the measured failure"); assert.equal(w?.value, "00:03:00"); assert.match(String(w?.detail), /01:31:43/); }); test("chunk-local: the measured nine-start chunk yields chapters instead of nothing", () => { // The first three of the nine starts the model actually emitted. const emitted = ["00:00:00", "00:03:54", "00:12:26"]; const absolute = parseChapters( [chunk3(emitted.map((start, i) => ({ start, title: `Topic ${i}` })))], LONG_CUES, ); assert.equal(absolute.chapters.length, 0, "the measured outcome: a lost chunk"); assert.equal( absolute.warnings.filter((w) => w.code === "out-of-range").length, 3, ); const local = parseChapters( [ chunk3( emitted.map((start, i) => ({ start, title: `Topic ${i}` })), "chunk-local", ), ], LONG_CUES, ); assert.equal(local.chapters.length, 3); assert.deepEqual( local.chapters.map((c) => c.clock), ["01:31:43", "01:35:37", "01:44:09"], ); }); test("chunk-local does not weaken the range clamp", () => { // 00:50:00 local is 02:21:43 absolute — past the chunk's real end, so it is // still rejected. The guard checks the chunk's REAL range either way. const { chapters, warnings } = parseChapters( [chunk3([{ start: "00:50:00", title: "Past the end" }], "chunk-local")], LONG_CUES, ); assert.equal(chapters.length, 0); const w = warnings.find((x) => x.code === "out-of-range"); assert.ok(w); // The warning reports REAL video time, with the model's own value recorded // alongside it so a prompt regression stays diagnosable from the artifact. assert.equal(w?.value, "02:21:43"); assert.match(String(w?.detail), /model emitted 00:50:00/); }); test("chunk-local: monotonicity is still checked, in real video time", () => { const { chapters, warnings } = parseChapters( [ chunk3( [ { start: "00:10:00", title: "Second thing" }, { start: "00:02:00", title: "Backwards" }, ], "chunk-local", ), ], LONG_CUES, ); assert.deepEqual( chapters.map((c) => c.title), ["Second thing"], ); const w = warnings.find((x) => x.code === "non-monotonic"); assert.ok(w); assert.equal(w?.value, "01:33:43"); }); test("a chunk starting at 0 is identical under both modes", () => { const entries = [{ start: "00:01:00", title: "Opening" }]; const abs = parseChapters([chunk(entries)], CUES); const local = parseChapters( [{ ...chunk(entries), timestampMode: "chunk-local" as const }], CUES, ); assert.deepEqual(abs.chapters, local.chapters); assert.deepEqual(abs.warnings, local.warnings); });