// `reports prepare` over a temp site: two reports' citations become clips and // post captures in the site's report-media cache, with a manifest and every // problem listed — through the real report checker, the real tier lookup and // real ffmpeg over a few frames of lavfi. // // Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test publish/reportMedia.test.ts import { after, test } from "node:test"; import assert from "node:assert/strict"; import { execFileSync } from "node:child_process"; import { existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; // Every path getPaths() can resolve to a place this file's code may write is // pinned under ROOT before anything calls it. const ROOT = mkdtempSync(path.join(tmpdir(), "reports-prepare-")); Object.assign(process.env, { TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), SITES_DIR: path.join(ROOT, "transcripts", "sites"), SETTINGS_FILE: path.join(ROOT, "settings.json"), EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), }); after(() => rmSync(ROOT, { recursive: true, force: true })); const { getPaths } = await import("../lib/paths"); const { prepareReportMedia, readReportMediaIndex, reportMediaDir, citedMoments } = await import("./reportMedia"); const { citedCaptureSourceDir } = await import("./citedPostCaptures"); const { main: prepareMain } = await import("../bin/reports-prepare"); const paths = getPaths(); const SITE = "demo-site"; const CH = "demo-channel"; const X = "demo-x"; const writeJson = (file: string, value: unknown) => { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, JSON.stringify(value, null, 2)); }; const ff = (args: string[]) => execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", ...args]); const videoDir = (slug: string, id: string) => path.join(paths.channelsDir, slug, "data", id); // A PNG header is all the copier reads of a screenshot. function png(w: number, h: number): Buffer { const b = Buffer.alloc(33); b.writeUInt32BE(0x89504e47, 0); b.writeUInt32BE(0x0d0a1a0a, 4); b.writeUInt32BE(13, 8); b.write("IHDR", 12, "latin1"); b.writeUInt32BE(w, 16); b.writeUInt32BE(h, 20); return b; } function report(id: string, citations: Record) { return { format: "archilyzer-report", version: 1, id, kind: "sweep", title: `Report ${id}`, citations, sections: [ { id: "s1", title: "One", body: Object.keys(citations).map((c) => `[${c}](cite:${c})`).join(" "), }, ], }; } // The corpus: a video channel with a fetched window and a recording's sound, // an X channel with two captured posts, and a channel no site has. writeJson(path.join(paths.channelsDir, CH, "config.json"), { handling: "transcribe", name: "Demo", url: "https://example.test/demo", }); writeJson(path.join(paths.channelsDir, X, "config.json"), { handling: "transcribe", name: "Demo (X)", url: "https://x.com/demo", sourceKind: "social", platform: "twitter", }); mkdirSync(path.join(videoDir(CH, "abc123"), "clips"), { recursive: true }); ff([ "-f", "lavfi", "-i", "testsrc=size=160x90:rate=10:duration=6", "-f", "lavfi", "-i", "sine=frequency=440:duration=6", "-c:v", "libx264", "-preset", "ultrafast", "-pix_fmt", "yuv420p", "-c:a", "aac", "-shortest", path.join(videoDir(CH, "abc123"), "clips", "0.00-6.00.mp4"), ]); mkdirSync(videoDir(CH, "pod1"), { recursive: true }); ff(["-f", "lavfi", "-i", "sine=frequency=220:duration=6", "-c:a", "aac", path.join(videoDir(CH, "pod1"), "audio.m4a")]); for (const id of ["111", "222"]) { const dir = citedCaptureSourceDir(paths.channelsDir, X, id); mkdirSync(dir, { recursive: true }); writeFileSync(path.join(dir, "shot.png"), png(600, 400)); writeFileSync(path.join(dir, `${id}_1.jpg`), `media of ${id}`); writeFileSync(path.join(dir, "capture.json"), "{}"); } writeJson(path.join(paths.sitesDir, SITE, "site.json"), { title: "Demo", channels: [{ slug: CH }, { slug: X }], search: false, reports: ["r1", "r2", "r-gone"], }); writeJson( path.join(paths.sitesDir, SITE, "reports", "r1", "report.json"), report("r1", { c01: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { before: 0.5 }, quote: "one" }, c02: { kind: "audio", channel: CH, id: "pod1", start: 1, end: 3, quote: "two" }, p01: { kind: "post", channel: X, id: "111", quote: "three" }, c03: { kind: "video", channel: CH, id: "missing1", start: 10, end: 12, quote: "four" }, c04: { kind: "video", channel: "elsewhere", id: "xyz", start: 1, end: 2, quote: "five" }, }), ); // The same moment as r1's c01, with more context after it. writeJson( path.join(paths.sitesDir, SITE, "reports", "r2", "report.json"), report("r2", { k1: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { after: 1 }, quote: "one" } }), ); // A draft: in reports/, not in site.json. Never read. writeJson( path.join(paths.sitesDir, SITE, "reports", "draft", "report.json"), report("draft", { d1: { kind: "post", channel: X, id: "222", quote: "uncited" } }), ); const PUBLIC = { social: { x: { visibility: "public" } } }; const VIDEO_KEY = `${CH}/abc123/1.00-2.00`; const AUDIO_KEY = `${CH}/pod1/1.00-3.00`; const POST_KEY = `${X}/111`; test("citedMoments: one moment per span, cited by both reports, with the wider pad on each side", () => { const moments = citedMoments([ report("r1", { a: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { before: 0.5 }, quote: "q" } }), report("r2", { b: { kind: "audio", channel: CH, id: "abc123", start: 1.001, end: 2, pad: { after: 1 }, quote: "q" } }), ] as never); assert.equal(moments.length, 1); assert.equal(moments[0].key, VIDEO_KEY); assert.equal(moments[0].kind, "video"); assert.deepEqual(moments[0].pad, { before: 0.5, after: 1 }); assert.deepEqual(moments[0].citedBy, ["r1#a", "r2#b"]); }); test("prepare: clips, the cited capture, a manifest, and every problem", async () => { const lines: string[] = []; const index = await prepareReportMedia({ siteId: SITE, settings: PUBLIC, onLog: (l) => lines.push(l) }); const dir = reportMediaDir(paths, SITE); assert.deepEqual(Object.keys(index.moments), [VIDEO_KEY, POST_KEY, AUDIO_KEY].sort()); const video = index.moments[VIDEO_KEY]; assert.equal(video.kind, "video"); assert.match(video.file, /^[0-9a-f]{32}\.mp4$/); assert.equal(video.width, 160); assert.equal(video.height, 90); // 1–2 widened by r1's 0.5 before and r2's 1 after: 0.5–3. assert.ok(Math.abs((video.durationSec ?? 0) - 2.5) < 0.15, `duration ${video.durationSec}`); assert.equal(readFileSync(path.join(dir, video.file)).length, video.bytes); const audio = index.moments[AUDIO_KEY]; assert.equal(audio.kind, "audio"); assert.match(audio.file, /\.m4a$/); assert.equal(audio.width, null); const post = index.moments[POST_KEY]; assert.equal(post.kind, "post"); assert.equal(post.file, `posts/${X}/111/shot.png`); assert.equal(post.width, 600); assert.ok(post.kind === "post" && post.media.map((m) => m.file).join() === `posts/${X}/111/111_1.jpg`); // Only the cited post: not the draft's, not the record. assert.deepEqual(readdirSync(path.join(dir, "posts", X)), ["111"]); assert.deepEqual(readdirSync(path.join(dir, "posts", X, "111")).sort(), ["111_1.jpg", "shot.png"]); const byKind = (k: string) => index.problems.filter((p) => p.kind === k); assert.equal(byKind("missing-report").length, 1); assert.equal(byKind("missing-report")[0].report, "r-gone"); assert.deepEqual(byKind("missing-media").map((p) => p.moment), [`${CH}/missing1/10.00-12.00`]); assert.deepEqual(byKind("missing-media")[0].citations, ["r1#c03"]); assert.deepEqual(byKind("not-in-site").map((p) => p.moment), ["elsewhere/xyz/1.00-2.00"]); assert.equal(index.problems.length, 3); // The manifest on disk is what was returned. assert.deepEqual(await readReportMediaIndex(paths, SITE), index); assert.ok(lines.some((l) => l.includes(VIDEO_KEY) && l.includes("corpus-window"))); }); test("a second run cuts nothing it already has; the CLI exits 1 and names each problem", async () => { const first = await readReportMediaIndex(paths, SITE); const out: string[] = []; const err: string[] = []; const code = await prepareMain({ siteId: SITE }, { log: (s) => out.push(s), error: (s) => err.push(s) }); assert.equal(code, 1); const second = await readReportMediaIndex(paths, SITE); assert.equal(second?.moments[VIDEO_KEY].file, first?.moments[VIDEO_KEY].file); assert.ok(out.some((l) => l.includes(VIDEO_KEY) && l.includes("cached"))); assert.ok(err.some((l) => l.startsWith(" missing-media:") && l.includes("missing1") && l.includes("r1#c03"))); assert.ok(err.some((l) => l.includes("missing-report"))); assert.equal(await prepareMain({ siteId: "no-such-site" }, { log: () => {}, error: () => {} }), 2); }); test("a post the site may not carry is a problem, and its capture leaves the cache", async () => { const index = await prepareReportMedia({ siteId: SITE, settings: { social: { x: { visibility: "private" } } } }); assert.equal(index.moments[POST_KEY], undefined); assert.deepEqual( index.problems.filter((p) => p.kind === "not-visible").map((p) => p.moment), [POST_KEY], ); assert.equal(existsSync(path.join(reportMediaDir(paths, SITE), "posts", X)), false); }); test("a clean site: no problems, and a clip no longer cited leaves the cache", async () => { const site = path.join(paths.sitesDir, SITE, "site.json"); writeJson(site, { title: "Demo", channels: [{ slug: CH }, { slug: X }], reports: ["r2"] }); const index = await prepareReportMedia({ siteId: SITE, settings: PUBLIC }); assert.deepEqual(index.problems, []); assert.deepEqual(Object.keys(index.moments), [VIDEO_KEY]); const files = readdirSync(reportMediaDir(paths, SITE)).sort(); const clip = index.moments[VIDEO_KEY].file; assert.deepEqual(files, [clip.replace(/\.mp4$/, ".json"), clip, "index.json"].sort()); }); test("a clean prepare ends by exporting the reports; a host with no browser skips the PDF with a note", async () => { // The record behind r2's citation, so the export can resolve it as compose does. writeJson(path.join(videoDir(CH, "abc123"), "metadata.info.json"), { id: "abc123", title: "Demo stream", upload_date: "20260110", webpage_url: "https://www.youtube.com/watch?v=abc123", extractor_key: "Youtube", }); writeFileSync( path.join(videoDir(CH, "abc123"), "transcript.en.vtt"), "WEBVTT\n\n00:00:00.000 --> 00:00:06.000 align:start position:0%\none<00:00:00.000>\n", ); const out: string[] = []; const err: string[] = []; const code = await prepareMain( { siteId: SITE, exportOptions: { openPdfPrinter: async () => ({ missing: "no browser here" }) } }, { log: (l) => out.push(l), error: (l) => err.push(l) }, ); assert.equal(code, 0, err.join("\n")); const dir = path.join(paths.exportSitesIndexDir, SITE, "report-exports", "r2"); assert.deepEqual(readdirSync(dir).sort(), ["evidence-pack.zip", "export.json", "report.html", "report.md", "slides.html"]); const manifest = JSON.parse(readFileSync(path.join(dir, "export.json"), "utf8")); assert.deepEqual(manifest.notes, ["report.pdf skipped: no browser here", "slides.pdf skipped: no browser here"]); assert.ok(out.some((l) => l.includes("report.pdf skipped: no browser here"))); });