// `archilyzer reports check`, `reports verify-quotes` and `reports // attach-video`, over a temp corpus. // // One channel with three records: one whose `en` track has the quote, one // whose served `en` track is a REWRITE (the quote matches it) while // `en-orig` has the words as spoken, and one the quote does not match at all. // A site publishing a report on the first; drafts on the others. // // Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test bin/reports-check.test.ts import { after, test } from "node:test"; import assert from "node:assert/strict"; import { execFileSync } from "node:child_process"; import { mkdirSync, mkdtempSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; const ROOT = mkdtempSync(path.join(tmpdir(), "reports-check-")); Object.assign(process.env, { TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), SITES_DIR: path.join(ROOT, "transcripts", "sites"), SETTINGS_FILE: path.join(ROOT, "settings.json"), EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), }); after(() => rmSync(ROOT, { recursive: true, force: true })); const { getPaths } = await import("../lib/paths"); const { checkMain, verifyQuotesMain } = await import("./reports-check"); const { attachVideoMain } = await import("./reports-attach-video"); const { encodePlan, encodeArgs } = await import("../publish/reportVideo"); const paths = getPaths(); const CHAN = "chan"; const writeJson = (file: string, value: unknown) => { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, JSON.stringify(value, null, 2)); }; const writeText = (file: string, text: string) => { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, text); }; const ts = (s: number) => new Date(s * 1000).toISOString().slice(11, 23); const vtt = (cues: [number, number, string][]) => "WEBVTT\nKind: captions\nLanguage: en\n\n" + cues.map(([a, b, text]) => `${ts(a)} --> ${ts(b)} align:start position:0%\n${text}<${ts(a)}>\n`).join("\n"); function record(id: string, tracks: Record) { const dir = path.join(paths.channelsDir, CHAN, "data", id); writeJson(path.join(dir, "metadata.info.json"), { id, title: `Stream ${id}`, channel: CHAN, upload_date: "20260110", duration: 600, webpage_url: `https://www.youtube.com/watch?v=${id}`, extractor_key: "Youtube", }); for (const [track, cues] of Object.entries(tracks)) writeText(path.join(dir, `transcript.${track}.vtt`), vtt(cues)); } const span = (id: string, quote: string) => ({ kind: "video", channel: CHAN, id, start: 10, end: 20, quote }); function report(id: string, citations: Record) { return { format: "archilyzer-report", version: 1, id, kind: "sweep", title: `Report ${id}`, citations, sections: [ { id: "s", title: "S", body: Object.keys(citations) .map((c) => `Said [this](cite:${c}).`) .join(" "), }, ], }; } const reportFile = (siteId: string, id: string) => path.join(paths.sitesDir, siteId, "reports", id, "report.json"); writeJson(paths.settingsFile, {}); writeJson(path.join(paths.channelsDir, CHAN, "config.json"), { handling: "youtube", name: "Chan", url: "https://www.youtube.com/@chan/videos", }); record("good1", { en: [[10, 20, "The bridge opened in the spring, I was there for it."]] }); record("rewr1", { en: [[10, 20, "We will never agree to the deal on the table."]], "en-orig": [[10, 20, "Honestly we might take whatever they offer us now."]], }); record("miss1", { en: [[10, 20, "Something else entirely about the weather today."]] }); writeJson(path.join(paths.sitesDir, "demo", "site.json"), { siteId: "demo", siteTitle: "Demo", siteDescription: "fixture", headerTitle: "demo", homeTagline: "", socialLinks: [], groups: [{ id: "default", name: "All channels", selectedByDefault: true }], defaultGroupId: "default", channels: [{ slug: CHAN, groupId: "default" }], siteUrl: "https://demo.example.test", archives: false, reports: ["pub1"], }); writeJson(reportFile("demo", "pub1"), report("pub1", { c1: span("good1", "The bridge opened in the spring, I was there for it.") })); writeJson( reportFile("demo", "draft1"), report("draft1", { c1: span("miss1", "The bridge opened in the spring, I was there for it.") }), ); writeJson( reportFile("demo", "mixed"), report("mixed", { c1: span("good1", "The bridge opened in the spring, I was there for it."), c2: span("rewr1", "We will never agree to the deal on the table."), c3: span("miss1", "The bridge opened in the spring, I was there for it."), c4: span("ghost1", "nothing here"), }), ); function capture() { const lines: string[] = []; return { lines, text: () => lines.join("\n"), out: { log: (s: string) => lines.push(s), error: (s: string) => lines.push(s) }, }; } test("check: compose's problems with no build — the published report lacks its prepared media", async () => { const c = capture(); assert.equal(await checkMain({ siteId: "demo", paths }, c.out), 1); assert.match(c.text(), /compose would fail/); assert.match(c.text(), /missing-media: chan\/good1\/10\.00-20\.00: no prepared media/); // Nothing was composed. assert.throws(() => statSync(path.join(paths.exportPublicDir, "reports"))); }); test("check --allow-missing-media passes the text, and says what it let through", async () => { const c = capture(); assert.equal(await checkMain({ siteId: "demo", paths, allowMissingMedia: true }, c.out), 0); assert.match(c.text(), /allowed \(--allow-missing-media\): missing-media/); assert.match(c.text(), /pub1: 1 quote\(s\) checked, lowest 1\.00/); assert.match(c.text(), /compose would pass/); }); test("check --reports checks a draft not in site.json: a drifted quote fails it", async () => { const c = capture(); assert.equal(await checkMain({ siteId: "demo", paths, reports: ["draft1"], allowMissingMedia: true }, c.out), 1); assert.match(c.text(), /quote-drift: draft1#c1: the quote matches \d+% of what the record says there/); }); test("check: an unknown site is a usage error", async () => { const c = capture(); assert.equal(await checkMain({ siteId: "nope", paths }, c.out), 2); }); test("verify-quotes: each quote's best track, en-orig where there is one, and what failed", async () => { const c = capture(); const code = await verifyQuotesMain({ file: reportFile("demo", "mixed"), json: true, paths, now: "2026-10-09T00:00:00Z" }, c.out); assert.equal(code, 1); const { results } = JSON.parse(c.text()) as { results: { citation: string; status: string; score?: number; track?: string; enOrig?: number }[]; }; const by = Object.fromEntries(results.map((r) => [r.citation, r])); assert.equal(by.c1.status, "ok"); assert.equal(by.c1.score, 1); assert.equal(by.c1.track, "transcript.en.vtt"); // Compose would pass c2 (its best track, the served `en`, matches) — but // the words as spoken do not. assert.equal(by.c2.status, "en-orig-drift"); assert.equal(by.c2.score, 1); assert.equal(by.c2.track, "transcript.en.vtt"); assert.ok(by.c2.enOrig !== undefined && by.c2.enOrig < 0.6, String(by.c2.enOrig)); assert.equal(by.c3.status, "drift"); assert.equal(by.c4.status, "missing-record"); }); test("verify-quotes: a clean report exits 0, in words", async () => { const c = capture(); assert.equal(await verifyQuotesMain({ file: reportFile("demo", "pub1"), paths }, c.out), 0); assert.match(c.text(), /ok\s+c1 {2}video chan\/good1 10–20 s {2}1\.00 transcript\.en\.vtt/); assert.match(c.text(), /1 quote\(s\): 1 ok/); }); // ─── attach-video ─── test("encodePlan: remux an H.264 mp4 under the limit; encode the rest to fit; refuse what cannot", () => { const mp4 = { durationSec: 120, formatName: "mov,mp4,m4a,3gp,3g2,mj2", videoCodec: "h264", audioCodec: "aac", width: 1920 }; const MiB = 1024 * 1024; assert.deepEqual(encodePlan(mp4, 10 * MiB), { mode: "remux" }); // Over the limit: 24 MiB × 0.96 over 120 s ≈ 1610 kb/s in all. const over = encodePlan(mp4, 80 * MiB); assert.equal(over.mode, "encode"); if (over.mode === "encode") { assert.equal(over.audioKbps, 128); assert.ok(over.videoKbps > 1300 && over.videoKbps < 1500, String(over.videoKbps)); assert.equal(over.maxWidth, 1280); } // A VP9 webm is encoded whatever its size. assert.equal(encodePlan({ ...mp4, formatName: "matroska,webm", videoCodec: "vp9", audioCodec: "opus" }, MiB).mode, "encode"); // Two hours do not fit 24 MiB watchably. const long = encodePlan({ ...mp4, durationSec: 7200 }, 900 * MiB); assert.equal(long.mode, "refuse"); if (long.mode === "refuse") assert.match(long.reason, /120\.0 min .* trim it/); assert.equal(encodePlan({ ...mp4, videoCodec: null }, MiB).mode, "refuse"); const args = encodeArgs("in.webm", "out.mp4", { mode: "encode", videoKbps: 800, audioKbps: 64, maxWidth: 854 }); assert.deepEqual(args.slice(args.indexOf("-b:v"), args.indexOf("-b:v") + 6), ["-b:v", "800k", "-maxrate", "1200k", "-bufsize", "1600k"]); assert.ok(args.includes("scale='min(854,iw)':-2")); assert.equal(args.at(-2), "+faststart"); }); const hasFfmpeg = (() => { try { execFileSync("ffmpeg", ["-version"], { stdio: "ignore" }); execFileSync("ffprobe", ["-version"], { stdio: "ignore" }); return true; } catch { return false; } })(); // A 3 s 640×360 clip with a tone: an H.264/AAC mp4, or an MPEG-4 Part 2 MKV. function makeClip(file: string, container: "mp4" | "mkv") { execFileSync("ffmpeg", [ "-v", "error", "-y", "-f", "lavfi", "-i", "testsrc=size=640x360:rate=25", "-f", "lavfi", "-i", "sine=frequency=440:sample_rate=44100", "-t", "3", ...(container === "mp4" ? ["-c:v", "libx264", "-preset", "ultrafast", "-crf", "8", "-c:a", "aac"] : ["-c:v", "mpeg4", "-q:v", "2", "-c:a", "aac"]), file, ]); } test("attach-video: an H.264 mp4 under the limit is remuxed, a poster drawn, report.json gets `video`", { skip: !hasFfmpeg && "no ffmpeg" }, async () => { const file = reportFile("demo", "pub1"); const src = path.join(ROOT, "clip.mp4"); makeClip(src, "mp4"); const c = capture(); assert.equal(await attachVideoMain({ reportFile: file, video: src, caption: "The stream, cut." }, c.out), 0, c.text()); const dir = path.dirname(file); const doc = JSON.parse(readFileSync(file, "utf8")) as { video: unknown; citations: unknown }; assert.deepEqual(doc.video, { src: "video.mp4", poster: "poster.jpg", caption: "The stream, cut." }); assert.ok(statSync(path.join(dir, "video.mp4")).size > 0); assert.ok(statSync(path.join(dir, "poster.jpg")).size > 0); assert.match(c.text(), /remuxed/); // The rest of the report is untouched. assert.deepEqual(Object.keys(doc.citations as object), ["c1"]); }); test("attach-video: anything else is encoded under the limit (re-encoded smaller on an overshoot); the poster and caption are kept", { skip: !hasFfmpeg && "no ffmpeg" }, async () => { const file = reportFile("demo", "pub1"); const src = path.join(ROOT, "clip.mkv"); makeClip(src, "mkv"); const limit = 200_000; assert.ok(statSync(src).size > limit, "the fixture must start over the limit"); const c = capture(); assert.equal(await attachVideoMain({ reportFile: file, video: src, limitBytes: limit }, c.out), 0, c.text()); const dir = path.dirname(file); const size = statSync(path.join(dir, "video.mp4")).size; assert.ok(size > 0 && size <= limit, `video.mp4 is ${size} bytes, over ${limit}`); const probe = execFileSync("ffprobe", ["-v", "error", "-show_entries", "stream=codec_name", "-of", "csv=p=0", path.join(dir, "video.mp4")], { encoding: "utf8" }); assert.deepEqual(probe.trim().split("\n").sort(), ["aac", "h264"]); const doc = JSON.parse(readFileSync(file, "utf8")) as { video: unknown }; assert.deepEqual(doc.video, { src: "video.mp4", poster: "poster.jpg", caption: "The stream, cut." }); assert.match(c.text(), /encoded/); }); test("attach-video refuses a file that is not a report, and writes nothing", { skip: !hasFfmpeg && "no ffmpeg" }, async () => { const notReport = path.join(ROOT, "nope", "report.json"); writeJson(notReport, { hello: "world" }); const c = capture(); assert.equal(await attachVideoMain({ reportFile: notReport, video: path.join(ROOT, "clip.mp4") }, c.out), 1); assert.match(c.text(), /is not a report/); assert.throws(() => statSync(path.join(ROOT, "nope", "video.mp4"))); });