// `archilyzer reports check`, `reports verify-quotes` and `reports
// attach-video`, over a temp corpus.
//
// One channel with three records: one whose `en` track has the quote, one
// whose served `en` track is a REWRITE (the quote matches it) while
// `en-orig` has the words as spoken, and one the quote does not match at all.
// A site publishing a report on the first; drafts on the others.
//
// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test bin/reports-check.test.ts
import { after, test } from "node:test";
import assert from "node:assert/strict";
import { execFileSync } from "node:child_process";
import { mkdirSync, mkdtempSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import path from "node:path";
const ROOT = mkdtempSync(path.join(tmpdir(), "reports-check-"));
Object.assign(process.env, {
TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
SITES_DIR: path.join(ROOT, "transcripts", "sites"),
SETTINGS_FILE: path.join(ROOT, "settings.json"),
EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
});
after(() => rmSync(ROOT, { recursive: true, force: true }));
const { getPaths } = await import("../lib/paths");
const { checkMain, verifyQuotesMain } = await import("./reports-check");
const { attachVideoMain } = await import("./reports-attach-video");
const { encodePlan, encodeArgs } = await import("../publish/reportVideo");
const paths = getPaths();
const CHAN = "chan";
const writeJson = (file: string, value: unknown) => {
mkdirSync(path.dirname(file), { recursive: true });
writeFileSync(file, JSON.stringify(value, null, 2));
};
const writeText = (file: string, text: string) => {
mkdirSync(path.dirname(file), { recursive: true });
writeFileSync(file, text);
};
const ts = (s: number) => new Date(s * 1000).toISOString().slice(11, 23);
const vtt = (cues: [number, number, string][]) =>
"WEBVTT\nKind: captions\nLanguage: en\n\n" +
cues.map(([a, b, text]) => `${ts(a)} --> ${ts(b)} align:start position:0%\n${text}<${ts(a)}>\n`).join("\n");
function record(id: string, tracks: Record) {
const dir = path.join(paths.channelsDir, CHAN, "data", id);
writeJson(path.join(dir, "metadata.info.json"), {
id,
title: `Stream ${id}`,
channel: CHAN,
upload_date: "20260110",
duration: 600,
webpage_url: `https://www.youtube.com/watch?v=${id}`,
extractor_key: "Youtube",
});
for (const [track, cues] of Object.entries(tracks)) writeText(path.join(dir, `transcript.${track}.vtt`), vtt(cues));
}
const span = (id: string, quote: string) => ({ kind: "video", channel: CHAN, id, start: 10, end: 20, quote });
function report(id: string, citations: Record) {
return {
format: "archilyzer-report",
version: 1,
id,
kind: "sweep",
title: `Report ${id}`,
citations,
sections: [
{
id: "s",
title: "S",
body: Object.keys(citations)
.map((c) => `Said [this](cite:${c}).`)
.join(" "),
},
],
};
}
const reportFile = (siteId: string, id: string) => path.join(paths.sitesDir, siteId, "reports", id, "report.json");
writeJson(paths.settingsFile, {});
writeJson(path.join(paths.channelsDir, CHAN, "config.json"), {
handling: "youtube",
name: "Chan",
url: "https://www.youtube.com/@chan/videos",
});
record("good1", { en: [[10, 20, "The bridge opened in the spring, I was there for it."]] });
record("rewr1", {
en: [[10, 20, "We will never agree to the deal on the table."]],
"en-orig": [[10, 20, "Honestly we might take whatever they offer us now."]],
});
record("miss1", { en: [[10, 20, "Something else entirely about the weather today."]] });
writeJson(path.join(paths.sitesDir, "demo", "site.json"), {
siteId: "demo",
siteTitle: "Demo",
siteDescription: "fixture",
headerTitle: "demo",
homeTagline: "",
socialLinks: [],
groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
defaultGroupId: "default",
channels: [{ slug: CHAN, groupId: "default" }],
siteUrl: "https://demo.example.test",
archives: false,
reports: ["pub1"],
});
writeJson(reportFile("demo", "pub1"), report("pub1", { c1: span("good1", "The bridge opened in the spring, I was there for it.") }));
writeJson(
reportFile("demo", "draft1"),
report("draft1", { c1: span("miss1", "The bridge opened in the spring, I was there for it.") }),
);
writeJson(
reportFile("demo", "mixed"),
report("mixed", {
c1: span("good1", "The bridge opened in the spring, I was there for it."),
c2: span("rewr1", "We will never agree to the deal on the table."),
c3: span("miss1", "The bridge opened in the spring, I was there for it."),
c4: span("ghost1", "nothing here"),
}),
);
function capture() {
const lines: string[] = [];
return {
lines,
text: () => lines.join("\n"),
out: { log: (s: string) => lines.push(s), error: (s: string) => lines.push(s) },
};
}
test("check: compose's problems with no build — the published report lacks its prepared media", async () => {
const c = capture();
assert.equal(await checkMain({ siteId: "demo", paths }, c.out), 1);
assert.match(c.text(), /compose would fail/);
assert.match(c.text(), /missing-media: chan\/good1\/10\.00-20\.00: no prepared media/);
// Nothing was composed.
assert.throws(() => statSync(path.join(paths.exportPublicDir, "reports")));
});
test("check --allow-missing-media passes the text, and says what it let through", async () => {
const c = capture();
assert.equal(await checkMain({ siteId: "demo", paths, allowMissingMedia: true }, c.out), 0);
assert.match(c.text(), /allowed \(--allow-missing-media\): missing-media/);
assert.match(c.text(), /pub1: 1 quote\(s\) checked, lowest 1\.00/);
assert.match(c.text(), /compose would pass/);
});
test("check --reports checks a draft not in site.json: a drifted quote fails it", async () => {
const c = capture();
assert.equal(await checkMain({ siteId: "demo", paths, reports: ["draft1"], allowMissingMedia: true }, c.out), 1);
assert.match(c.text(), /quote-drift: draft1#c1: the quote matches \d+% of what the record says there/);
});
test("check: an unknown site is a usage error", async () => {
const c = capture();
assert.equal(await checkMain({ siteId: "nope", paths }, c.out), 2);
});
test("verify-quotes: each quote's best track, en-orig where there is one, and what failed", async () => {
const c = capture();
const code = await verifyQuotesMain({ file: reportFile("demo", "mixed"), json: true, paths, now: "2026-10-09T00:00:00Z" }, c.out);
assert.equal(code, 1);
const { results } = JSON.parse(c.text()) as {
results: { citation: string; status: string; score?: number; track?: string; enOrig?: number }[];
};
const by = Object.fromEntries(results.map((r) => [r.citation, r]));
assert.equal(by.c1.status, "ok");
assert.equal(by.c1.score, 1);
assert.equal(by.c1.track, "transcript.en.vtt");
// Compose would pass c2 (its best track, the served `en`, matches) — but
// the words as spoken do not.
assert.equal(by.c2.status, "en-orig-drift");
assert.equal(by.c2.score, 1);
assert.equal(by.c2.track, "transcript.en.vtt");
assert.ok(by.c2.enOrig !== undefined && by.c2.enOrig < 0.6, String(by.c2.enOrig));
assert.equal(by.c3.status, "drift");
assert.equal(by.c4.status, "missing-record");
});
test("verify-quotes: a clean report exits 0, in words", async () => {
const c = capture();
assert.equal(await verifyQuotesMain({ file: reportFile("demo", "pub1"), paths }, c.out), 0);
assert.match(c.text(), /ok\s+c1 {2}video chan\/good1 10–20 s {2}1\.00 transcript\.en\.vtt/);
assert.match(c.text(), /1 quote\(s\): 1 ok/);
});
// ─── attach-video ───
test("encodePlan: remux an H.264 mp4 under the limit; encode the rest to fit; refuse what cannot", () => {
const mp4 = { durationSec: 120, formatName: "mov,mp4,m4a,3gp,3g2,mj2", videoCodec: "h264", audioCodec: "aac", width: 1920 };
const MiB = 1024 * 1024;
assert.deepEqual(encodePlan(mp4, 10 * MiB), { mode: "remux" });
// Over the limit: 24 MiB × 0.96 over 120 s ≈ 1610 kb/s in all.
const over = encodePlan(mp4, 80 * MiB);
assert.equal(over.mode, "encode");
if (over.mode === "encode") {
assert.equal(over.audioKbps, 128);
assert.ok(over.videoKbps > 1300 && over.videoKbps < 1500, String(over.videoKbps));
assert.equal(over.maxWidth, 1280);
}
// A VP9 webm is encoded whatever its size.
assert.equal(encodePlan({ ...mp4, formatName: "matroska,webm", videoCodec: "vp9", audioCodec: "opus" }, MiB).mode, "encode");
// Two hours do not fit 24 MiB watchably.
const long = encodePlan({ ...mp4, durationSec: 7200 }, 900 * MiB);
assert.equal(long.mode, "refuse");
if (long.mode === "refuse") assert.match(long.reason, /120\.0 min .* trim it/);
assert.equal(encodePlan({ ...mp4, videoCodec: null }, MiB).mode, "refuse");
const args = encodeArgs("in.webm", "out.mp4", { mode: "encode", videoKbps: 800, audioKbps: 64, maxWidth: 854 });
assert.deepEqual(args.slice(args.indexOf("-b:v"), args.indexOf("-b:v") + 6), ["-b:v", "800k", "-maxrate", "1200k", "-bufsize", "1600k"]);
assert.ok(args.includes("scale='min(854,iw)':-2"));
assert.equal(args.at(-2), "+faststart");
});
const hasFfmpeg = (() => {
try {
execFileSync("ffmpeg", ["-version"], { stdio: "ignore" });
execFileSync("ffprobe", ["-version"], { stdio: "ignore" });
return true;
} catch {
return false;
}
})();
// A 3 s 640×360 clip with a tone: an H.264/AAC mp4, or an MPEG-4 Part 2 MKV.
function makeClip(file: string, container: "mp4" | "mkv") {
execFileSync("ffmpeg", [
"-v", "error", "-y",
"-f", "lavfi", "-i", "testsrc=size=640x360:rate=25",
"-f", "lavfi", "-i", "sine=frequency=440:sample_rate=44100",
"-t", "3",
...(container === "mp4"
? ["-c:v", "libx264", "-preset", "ultrafast", "-crf", "8", "-c:a", "aac"]
: ["-c:v", "mpeg4", "-q:v", "2", "-c:a", "aac"]),
file,
]);
}
test("attach-video: an H.264 mp4 under the limit is remuxed, a poster drawn, report.json gets `video`", { skip: !hasFfmpeg && "no ffmpeg" }, async () => {
const file = reportFile("demo", "pub1");
const src = path.join(ROOT, "clip.mp4");
makeClip(src, "mp4");
const c = capture();
assert.equal(await attachVideoMain({ reportFile: file, video: src, caption: "The stream, cut." }, c.out), 0, c.text());
const dir = path.dirname(file);
const doc = JSON.parse(readFileSync(file, "utf8")) as { video: unknown; citations: unknown };
assert.deepEqual(doc.video, { src: "video.mp4", poster: "poster.jpg", caption: "The stream, cut." });
assert.ok(statSync(path.join(dir, "video.mp4")).size > 0);
assert.ok(statSync(path.join(dir, "poster.jpg")).size > 0);
assert.match(c.text(), /remuxed/);
// The rest of the report is untouched.
assert.deepEqual(Object.keys(doc.citations as object), ["c1"]);
});
test("attach-video: anything else is encoded under the limit (re-encoded smaller on an overshoot); the poster and caption are kept", { skip: !hasFfmpeg && "no ffmpeg" }, async () => {
const file = reportFile("demo", "pub1");
const src = path.join(ROOT, "clip.mkv");
makeClip(src, "mkv");
const limit = 200_000;
assert.ok(statSync(src).size > limit, "the fixture must start over the limit");
const c = capture();
assert.equal(await attachVideoMain({ reportFile: file, video: src, limitBytes: limit }, c.out), 0, c.text());
const dir = path.dirname(file);
const size = statSync(path.join(dir, "video.mp4")).size;
assert.ok(size > 0 && size <= limit, `video.mp4 is ${size} bytes, over ${limit}`);
const probe = execFileSync("ffprobe", ["-v", "error", "-show_entries", "stream=codec_name", "-of", "csv=p=0", path.join(dir, "video.mp4")], { encoding: "utf8" });
assert.deepEqual(probe.trim().split("\n").sort(), ["aac", "h264"]);
const doc = JSON.parse(readFileSync(file, "utf8")) as { video: unknown };
assert.deepEqual(doc.video, { src: "video.mp4", poster: "poster.jpg", caption: "The stream, cut." });
assert.match(c.text(), /encoded/);
});
test("attach-video refuses a file that is not a report, and writes nothing", { skip: !hasFfmpeg && "no ffmpeg" }, async () => {
const notReport = path.join(ROOT, "nope", "report.json");
writeJson(notReport, { hello: "world" });
const c = capture();
assert.equal(await attachVideoMain({ reportFile: notReport, video: path.join(ROOT, "clip.mp4") }, c.out), 1);
assert.match(c.text(), /is not a report/);
assert.throws(() => statSync(path.join(ROOT, "nope", "video.mp4")));
});