commit 19e0f6b76ec44d4e137752be52f3c31a9f61f0bb
parent 762642b24ae0f1efcbd96b6b4824412745bd062a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 9 Oct 2026 12:03:42 -0400
reports: `reports check`, `reports verify-quotes` and `reports attach-video` on the CLI
Three report jobs that meant a build, or a by-hand script, before:
- `archilyzer reports check <site> [--reports a,b] [--allow-missing-media]`
is the reports stage of compose with nothing written — resolveSiteReports:
validation, every cited quote against its record, prepared media present
and current, the report video under the publish limit — and exits 1 with
compose's own problem lines. --reports checks drafts not yet in site.json.
- `archilyzer reports verify-quotes <report.json> [--json]` runs compose's
quote check over every video, audio and post quote of one report: the best
score and track, and the `en-orig` track's score where the record has one.
A quote that matches a served `en` rewrite but not en-orig (the words as
spoken) is `en-orig-drift`, though compose, which takes the best track,
would pass it. The span check is now ONE function, checkSpanQuote, which
compose calls too (it was inline in resolveSiteReports); its drift
sentence likewise (quoteDriftMessage).
- `archilyzer reports attach-video <report.json> <video> [--poster]
[--caption]` (publish/reportVideo.ts): an H.264 mp4 under 24 MiB is
remuxed (+faststart); anything else is encoded (H.264/AAC, bitrate from
the duration inside 96% of the limit, narrower as it falls) and encoded
again smaller if a pass overshoots; too long to be watchable is refused
with its length. video.mp4 and a poster (given, kept, or a frame) land
beside report.json, then `video` is set — only if the video's own fields
validate.
Tests: bin/reports-check.test.ts, 10 (a temp corpus with a rewritten `en`
track; two of them encode real 3 s clips with ffmpeg, skipped without it).
composeReports' three test files unchanged and green (19).
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
7 files changed, 962 insertions(+), 13 deletions(-)
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -284,6 +284,61 @@ export const COMMANDS: Command[] = [
},
},
{
+ path: ["reports", "check"],
+ usage:
+ "<id> [--reports <a,b>] [--allow-missing-media] what compose would say about the site's reports, with no build: each report validated, every cited quote checked against its record, every cited moment's prepared media present and current, the report video under the publish limit (exit 1 with the list; --reports checks those, drafts included; default id: SITE_ID)",
+ flags: { reports: "string", "allow-missing-media": "boolean" },
+ maxPositionals: 1,
+ run: async ({ positionals, flags, env }) => {
+ const siteId = siteIdFrom(positionals, env, "reports check");
+ if (!siteId) return 2;
+ const reports =
+ typeof flags.reports === "string"
+ ? flags.reports.split(",").map((r) => r.trim()).filter(Boolean)
+ : undefined;
+ return (await import("./reports-check")).checkMain({
+ siteId,
+ ...(reports ? { reports } : {}),
+ allowMissingMedia: flags["allow-missing-media"] === true,
+ });
+ },
+ },
+ {
+ path: ["reports", "verify-quotes"],
+ usage:
+ "<report.json> [--json] every video, audio and post quote of one report against its record with compose's own check: the best score and track, and the en-orig track's score where the record has one — a quote that matches a served `en` rewrite and not en-orig is reported (exit 1 when any quote drifted)",
+ flags: { json: "boolean" },
+ maxPositionals: 1,
+ run: async ({ positionals, flags }) => {
+ const [file] = positionals;
+ if (!file) {
+ console.error("reports verify-quotes: give <report.json>");
+ return 2;
+ }
+ return (await import("./reports-check")).verifyQuotesMain({ file, json: flags.json === true });
+ },
+ },
+ {
+ path: ["reports", "attach-video"],
+ usage:
+ "<report.json> <video> [--poster <image>] [--caption <line>] the report's video: remuxed (an H.264 mp4 under the limit) or encoded to fit the 24 MiB publish limit, written beside report.json as video.mp4 with a poster (given, kept, or a frame of the video), and `video` set in report.json",
+ flags: { poster: "string", caption: "string" },
+ maxPositionals: 2,
+ run: async ({ positionals, flags }) => {
+ const [reportFile, video] = positionals;
+ if (!reportFile || !video) {
+ console.error("reports attach-video: give <report.json> <video>");
+ return 2;
+ }
+ return (await import("./reports-attach-video")).attachVideoMain({
+ reportFile,
+ video,
+ ...(typeof flags.poster === "string" ? { poster: flags.poster } : {}),
+ ...(typeof flags.caption === "string" ? { caption: flags.caption } : {}),
+ });
+ },
+ },
+ {
path: ["reports", "convert"],
usage:
"<sweep|ask|manifest> <in> --out <report.json> [--channels-dir <dir>] [--id <id>] [--title <title>] a /sweep report (markdown), an /ask answer or a report-to-video manifest as a report.json, written only when it validates (--channels-dir: widen spans from the cues, find posts' channels)",
diff --git a/common/bin/reports-attach-video.ts b/common/bin/reports-attach-video.ts
@@ -0,0 +1,49 @@
+// `archilyzer reports attach-video <report.json> <video> [--poster <image>]
+// [--caption <line>]` — the video as the report's: encoded or remuxed to fit
+// the publish limit, written beside report.json as video.mp4 with a poster,
+// and `video` set in report.json (publish/reportVideo.ts).
+//
+// Exit 0 when attached, 1 when it could not be (the reason printed: too long
+// to fit, not a report, ffmpeg's error), 2 for usage.
+
+import { getPaths } from "../lib/paths";
+import { AttachVideoError, attachReportVideo } from "../publish/reportVideo";
+
+type Out = { log: (s: string) => void; error: (s: string) => void };
+
+export async function attachVideoMain(
+ opts: {
+ reportFile: string;
+ video: string;
+ poster?: string;
+ caption?: string;
+ limitBytes?: number;
+ ffmpegBin?: string;
+ ffprobeBin?: string;
+ },
+ out: Out = console,
+): Promise<number> {
+ const paths = getPaths();
+ try {
+ const done = await attachReportVideo({
+ reportFile: opts.reportFile,
+ video: opts.video,
+ ...(opts.poster !== undefined ? { poster: opts.poster } : {}),
+ ...(opts.caption !== undefined ? { caption: opts.caption } : {}),
+ ...(opts.limitBytes !== undefined ? { limitBytes: opts.limitBytes } : {}),
+ ffmpegBin: opts.ffmpegBin ?? paths.ffmpegBin,
+ ffprobeBin: opts.ffprobeBin ?? paths.ffprobeBin,
+ onLog: out.log,
+ });
+ out.log(
+ `reports attach-video: ${opts.reportFile} video = ${JSON.stringify(done.video)} ` +
+ `(${done.mode === "remux" ? "remuxed" : `encoded, ${done.attempts} pass(es)`}, ${(done.bytes / (1024 * 1024)).toFixed(2)} MiB)`,
+ );
+ return 0;
+ } catch (err) {
+ const e = err as Error & { stderr?: string };
+ const why = err instanceof AttachVideoError ? e.message : (e.stderr || e.message).trim().split("\n").slice(-3).join(" ");
+ out.error(`reports attach-video: ${why}`);
+ return 1;
+ }
+}
diff --git a/common/bin/reports-check.test.ts b/common/bin/reports-check.test.ts
@@ -0,0 +1,286 @@
+// `archilyzer reports check`, `reports verify-quotes` and `reports
+// attach-video`, over a temp corpus.
+//
+// One channel with three records: one whose `en` track has the quote, one
+// whose served `en` track is a REWRITE (the quote matches it) while
+// `en-orig` has the words as spoken, and one the quote does not match at all.
+// A site publishing a report on the first; drafts on the others.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test bin/reports-check.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { execFileSync } from "node:child_process";
+import { mkdirSync, mkdtempSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "reports-check-"));
+Object.assign(process.env, {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+});
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { checkMain, verifyQuotesMain } = await import("./reports-check");
+const { attachVideoMain } = await import("./reports-attach-video");
+const { encodePlan, encodeArgs } = await import("../publish/reportVideo");
+
+const paths = getPaths();
+const CHAN = "chan";
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const writeText = (file: string, text: string) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, text);
+};
+const ts = (s: number) => new Date(s * 1000).toISOString().slice(11, 23);
+const vtt = (cues: [number, number, string][]) =>
+ "WEBVTT\nKind: captions\nLanguage: en\n\n" +
+ cues.map(([a, b, text]) => `${ts(a)} --> ${ts(b)} align:start position:0%\n${text}<${ts(a)}><c></c>\n`).join("\n");
+
+function record(id: string, tracks: Record<string, [number, number, string][]>) {
+ const dir = path.join(paths.channelsDir, CHAN, "data", id);
+ writeJson(path.join(dir, "metadata.info.json"), {
+ id,
+ title: `Stream ${id}`,
+ channel: CHAN,
+ upload_date: "20260110",
+ duration: 600,
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ for (const [track, cues] of Object.entries(tracks)) writeText(path.join(dir, `transcript.${track}.vtt`), vtt(cues));
+}
+
+const span = (id: string, quote: string) => ({ kind: "video", channel: CHAN, id, start: 10, end: 20, quote });
+
+function report(id: string, citations: Record<string, unknown>) {
+ return {
+ format: "archilyzer-report",
+ version: 1,
+ id,
+ kind: "sweep",
+ title: `Report ${id}`,
+ citations,
+ sections: [
+ {
+ id: "s",
+ title: "S",
+ body: Object.keys(citations)
+ .map((c) => `Said [this](cite:${c}).`)
+ .join(" "),
+ },
+ ],
+ };
+}
+
+const reportFile = (siteId: string, id: string) => path.join(paths.sitesDir, siteId, "reports", id, "report.json");
+
+writeJson(paths.settingsFile, {});
+writeJson(path.join(paths.channelsDir, CHAN, "config.json"), {
+ handling: "youtube",
+ name: "Chan",
+ url: "https://www.youtube.com/@chan/videos",
+});
+record("good1", { en: [[10, 20, "The bridge opened in the spring, I was there for it."]] });
+record("rewr1", {
+ en: [[10, 20, "We will never agree to the deal on the table."]],
+ "en-orig": [[10, 20, "Honestly we might take whatever they offer us now."]],
+});
+record("miss1", { en: [[10, 20, "Something else entirely about the weather today."]] });
+writeJson(path.join(paths.sitesDir, "demo", "site.json"), {
+ siteId: "demo",
+ siteTitle: "Demo",
+ siteDescription: "fixture",
+ headerTitle: "demo",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHAN, groupId: "default" }],
+ siteUrl: "https://demo.example.test",
+ archives: false,
+ reports: ["pub1"],
+});
+writeJson(reportFile("demo", "pub1"), report("pub1", { c1: span("good1", "The bridge opened in the spring, I was there for it.") }));
+writeJson(
+ reportFile("demo", "draft1"),
+ report("draft1", { c1: span("miss1", "The bridge opened in the spring, I was there for it.") }),
+);
+writeJson(
+ reportFile("demo", "mixed"),
+ report("mixed", {
+ c1: span("good1", "The bridge opened in the spring, I was there for it."),
+ c2: span("rewr1", "We will never agree to the deal on the table."),
+ c3: span("miss1", "The bridge opened in the spring, I was there for it."),
+ c4: span("ghost1", "nothing here"),
+ }),
+);
+
+function capture() {
+ const lines: string[] = [];
+ return {
+ lines,
+ text: () => lines.join("\n"),
+ out: { log: (s: string) => lines.push(s), error: (s: string) => lines.push(s) },
+ };
+}
+
+test("check: compose's problems with no build — the published report lacks its prepared media", async () => {
+ const c = capture();
+ assert.equal(await checkMain({ siteId: "demo", paths }, c.out), 1);
+ assert.match(c.text(), /compose would fail/);
+ assert.match(c.text(), /missing-media: chan\/good1\/10\.00-20\.00: no prepared media/);
+ // Nothing was composed.
+ assert.throws(() => statSync(path.join(paths.exportPublicDir, "reports")));
+});
+
+test("check --allow-missing-media passes the text, and says what it let through", async () => {
+ const c = capture();
+ assert.equal(await checkMain({ siteId: "demo", paths, allowMissingMedia: true }, c.out), 0);
+ assert.match(c.text(), /allowed \(--allow-missing-media\): missing-media/);
+ assert.match(c.text(), /pub1: 1 quote\(s\) checked, lowest 1\.00/);
+ assert.match(c.text(), /compose would pass/);
+});
+
+test("check --reports checks a draft not in site.json: a drifted quote fails it", async () => {
+ const c = capture();
+ assert.equal(await checkMain({ siteId: "demo", paths, reports: ["draft1"], allowMissingMedia: true }, c.out), 1);
+ assert.match(c.text(), /quote-drift: draft1#c1: the quote matches \d+% of what the record says there/);
+});
+
+test("check: an unknown site is a usage error", async () => {
+ const c = capture();
+ assert.equal(await checkMain({ siteId: "nope", paths }, c.out), 2);
+});
+
+test("verify-quotes: each quote's best track, en-orig where there is one, and what failed", async () => {
+ const c = capture();
+ const code = await verifyQuotesMain({ file: reportFile("demo", "mixed"), json: true, paths, now: "2026-10-09T00:00:00Z" }, c.out);
+ assert.equal(code, 1);
+ const { results } = JSON.parse(c.text()) as {
+ results: { citation: string; status: string; score?: number; track?: string; enOrig?: number }[];
+ };
+ const by = Object.fromEntries(results.map((r) => [r.citation, r]));
+ assert.equal(by.c1.status, "ok");
+ assert.equal(by.c1.score, 1);
+ assert.equal(by.c1.track, "transcript.en.vtt");
+ // Compose would pass c2 (its best track, the served `en`, matches) — but
+ // the words as spoken do not.
+ assert.equal(by.c2.status, "en-orig-drift");
+ assert.equal(by.c2.score, 1);
+ assert.equal(by.c2.track, "transcript.en.vtt");
+ assert.ok(by.c2.enOrig !== undefined && by.c2.enOrig < 0.6, String(by.c2.enOrig));
+ assert.equal(by.c3.status, "drift");
+ assert.equal(by.c4.status, "missing-record");
+});
+
+test("verify-quotes: a clean report exits 0, in words", async () => {
+ const c = capture();
+ assert.equal(await verifyQuotesMain({ file: reportFile("demo", "pub1"), paths }, c.out), 0);
+ assert.match(c.text(), /ok\s+c1 {2}video chan\/good1 10–20 s {2}1\.00 transcript\.en\.vtt/);
+ assert.match(c.text(), /1 quote\(s\): 1 ok/);
+});
+
+// ─── attach-video ───
+
+test("encodePlan: remux an H.264 mp4 under the limit; encode the rest to fit; refuse what cannot", () => {
+ const mp4 = { durationSec: 120, formatName: "mov,mp4,m4a,3gp,3g2,mj2", videoCodec: "h264", audioCodec: "aac", width: 1920 };
+ const MiB = 1024 * 1024;
+ assert.deepEqual(encodePlan(mp4, 10 * MiB), { mode: "remux" });
+ // Over the limit: 24 MiB × 0.96 over 120 s ≈ 1610 kb/s in all.
+ const over = encodePlan(mp4, 80 * MiB);
+ assert.equal(over.mode, "encode");
+ if (over.mode === "encode") {
+ assert.equal(over.audioKbps, 128);
+ assert.ok(over.videoKbps > 1300 && over.videoKbps < 1500, String(over.videoKbps));
+ assert.equal(over.maxWidth, 1280);
+ }
+ // A VP9 webm is encoded whatever its size.
+ assert.equal(encodePlan({ ...mp4, formatName: "matroska,webm", videoCodec: "vp9", audioCodec: "opus" }, MiB).mode, "encode");
+ // Two hours do not fit 24 MiB watchably.
+ const long = encodePlan({ ...mp4, durationSec: 7200 }, 900 * MiB);
+ assert.equal(long.mode, "refuse");
+ if (long.mode === "refuse") assert.match(long.reason, /120\.0 min .* trim it/);
+ assert.equal(encodePlan({ ...mp4, videoCodec: null }, MiB).mode, "refuse");
+ const args = encodeArgs("in.webm", "out.mp4", { mode: "encode", videoKbps: 800, audioKbps: 64, maxWidth: 854 });
+ assert.deepEqual(args.slice(args.indexOf("-b:v"), args.indexOf("-b:v") + 6), ["-b:v", "800k", "-maxrate", "1200k", "-bufsize", "1600k"]);
+ assert.ok(args.includes("scale='min(854,iw)':-2"));
+ assert.equal(args.at(-2), "+faststart");
+});
+
+const hasFfmpeg = (() => {
+ try {
+ execFileSync("ffmpeg", ["-version"], { stdio: "ignore" });
+ execFileSync("ffprobe", ["-version"], { stdio: "ignore" });
+ return true;
+ } catch {
+ return false;
+ }
+})();
+
+// A 3 s 640×360 clip with a tone: an H.264/AAC mp4, or an MPEG-4 Part 2 MKV.
+function makeClip(file: string, container: "mp4" | "mkv") {
+ execFileSync("ffmpeg", [
+ "-v", "error", "-y",
+ "-f", "lavfi", "-i", "testsrc=size=640x360:rate=25",
+ "-f", "lavfi", "-i", "sine=frequency=440:sample_rate=44100",
+ "-t", "3",
+ ...(container === "mp4"
+ ? ["-c:v", "libx264", "-preset", "ultrafast", "-crf", "8", "-c:a", "aac"]
+ : ["-c:v", "mpeg4", "-q:v", "2", "-c:a", "aac"]),
+ file,
+ ]);
+}
+
+test("attach-video: an H.264 mp4 under the limit is remuxed, a poster drawn, report.json gets `video`", { skip: !hasFfmpeg && "no ffmpeg" }, async () => {
+ const file = reportFile("demo", "pub1");
+ const src = path.join(ROOT, "clip.mp4");
+ makeClip(src, "mp4");
+ const c = capture();
+ assert.equal(await attachVideoMain({ reportFile: file, video: src, caption: "The stream, cut." }, c.out), 0, c.text());
+ const dir = path.dirname(file);
+ const doc = JSON.parse(readFileSync(file, "utf8")) as { video: unknown; citations: unknown };
+ assert.deepEqual(doc.video, { src: "video.mp4", poster: "poster.jpg", caption: "The stream, cut." });
+ assert.ok(statSync(path.join(dir, "video.mp4")).size > 0);
+ assert.ok(statSync(path.join(dir, "poster.jpg")).size > 0);
+ assert.match(c.text(), /remuxed/);
+ // The rest of the report is untouched.
+ assert.deepEqual(Object.keys(doc.citations as object), ["c1"]);
+});
+
+test("attach-video: anything else is encoded under the limit (re-encoded smaller on an overshoot); the poster and caption are kept", { skip: !hasFfmpeg && "no ffmpeg" }, async () => {
+ const file = reportFile("demo", "pub1");
+ const src = path.join(ROOT, "clip.mkv");
+ makeClip(src, "mkv");
+ const limit = 200_000;
+ assert.ok(statSync(src).size > limit, "the fixture must start over the limit");
+ const c = capture();
+ assert.equal(await attachVideoMain({ reportFile: file, video: src, limitBytes: limit }, c.out), 0, c.text());
+ const dir = path.dirname(file);
+ const size = statSync(path.join(dir, "video.mp4")).size;
+ assert.ok(size > 0 && size <= limit, `video.mp4 is ${size} bytes, over ${limit}`);
+ const probe = execFileSync("ffprobe", ["-v", "error", "-show_entries", "stream=codec_name", "-of", "csv=p=0", path.join(dir, "video.mp4")], { encoding: "utf8" });
+ assert.deepEqual(probe.trim().split("\n").sort(), ["aac", "h264"]);
+ const doc = JSON.parse(readFileSync(file, "utf8")) as { video: unknown };
+ assert.deepEqual(doc.video, { src: "video.mp4", poster: "poster.jpg", caption: "The stream, cut." });
+ assert.match(c.text(), /encoded/);
+});
+
+test("attach-video refuses a file that is not a report, and writes nothing", { skip: !hasFfmpeg && "no ffmpeg" }, async () => {
+ const notReport = path.join(ROOT, "nope", "report.json");
+ writeJson(notReport, { hello: "world" });
+ const c = capture();
+ assert.equal(await attachVideoMain({ reportFile: notReport, video: path.join(ROOT, "clip.mp4") }, c.out), 1);
+ assert.match(c.text(), /is not a report/);
+ assert.throws(() => statSync(path.join(ROOT, "nope", "video.mp4")));
+});
diff --git a/common/bin/reports-check.ts b/common/bin/reports-check.ts
@@ -0,0 +1,276 @@
+// `archilyzer reports check <site>` and `archilyzer reports verify-quotes
+// <report.json>` — what compose would say about a site's reports, without a
+// build; and one report's quotes against the transcripts, track by track.
+//
+// CHECK is the reports stage of compose with nothing written:
+// resolveSiteReports (publish/composeReports.ts) — every report parsed and
+// validated, every cited quote verified against its record, every cited
+// moment's prepared media present and current, the report's video on disk and
+// under the publish limit. Its problems are compose's, word for word. With
+// `--reports a,b` it checks those reports (drafts included: a report need not
+// be in site.json yet); `--allow-missing-media` checks the text before
+// `reports prepare` has cut anything.
+//
+// VERIFY-QUOTES runs compose's own quote check (checkSpanQuote, the post
+// check) over every video, audio and post citation of ONE report.json,
+// published or not, and prints each one's best score and track — and, where
+// the record has an `en-orig` track, that track's score. A served `en` track
+// can be a rewrite of what was said; a quote that matches it and not
+// `en-orig` is not what the speaker said, and is reported as such, even
+// though compose (which takes the best track) would pass it.
+//
+// Exit 0 when there is nothing to report, 1 with the list, 2 for usage (an
+// unknown site, an unreadable file).
+
+import { readFile } from "node:fs/promises";
+import path from "node:path";
+import { getPaths, type Paths } from "../lib/paths";
+import { getSite, listSiteIds } from "../lib/site";
+import { readChannelConfig } from "../controller/channels";
+import { assertChannelTextReadable } from "../lib/channelMedia";
+import { readAllPosts } from "../lib/posts-server";
+import { parseReport } from "../lib/report/validate";
+import type { Report } from "../lib/report/schema";
+import { reportCitationNumbers } from "../lib/report/uses";
+import { QUOTE_DRIFT_THRESHOLD, quoteDrifted, quoteVerification, roundScore } from "../lib/citations/verify";
+import {
+ ComposeReportsError,
+ checkSpanQuote,
+ formatComposeReportsProblems,
+ quoteDriftMessage,
+ readCitedRecord,
+ resolveSiteReports,
+} from "../publish/composeReports";
+
+type Out = { log: (s: string) => void; error: (s: string) => void };
+
+// ─── reports check ───
+
+export async function checkMain(
+ opts: {
+ siteId: string;
+ reports?: string[];
+ allowMissingMedia?: boolean;
+ paths?: Paths;
+ settings?: { social?: { x?: { visibility?: unknown } } };
+ },
+ out: Out = console,
+): Promise<number> {
+ const paths = opts.paths ?? getPaths();
+ if (!listSiteIds(paths).includes(opts.siteId)) {
+ out.error(`reports check: no site "${opts.siteId}" (sites/${opts.siteId}/site.json)`);
+ return 2;
+ }
+ const site = getSite(opts.siteId, paths);
+ const ids = opts.reports ?? site.reports ?? [];
+ if (ids.length === 0) {
+ out.log(`reports check ${opts.siteId}: the site publishes no reports — nothing to check.`);
+ return 0;
+ }
+ try {
+ const resolved = await resolveSiteReports({
+ paths,
+ site: { ...site, reports: ids },
+ allowMissingMedia: opts.allowMissingMedia === true,
+ ...(opts.settings ? { settings: opts.settings } : {}),
+ });
+ for (const line of formatComposeReportsProblems(resolved.allowed)) {
+ out.log(` allowed (--allow-missing-media): ${line}`);
+ }
+ for (const r of resolved.reports) {
+ const scores = Object.values(r.citations ?? {})
+ .map((c) => (c.kind === "video" || c.kind === "audio" || c.kind === "post" ? c.verification?.quoteScore : undefined))
+ .filter((s): s is number => typeof s === "number");
+ const low = scores.length ? Math.min(...scores) : null;
+ out.log(
+ ` ${r.id}: ${scores.length} quote(s) checked` + (low === null ? "" : `, lowest ${low.toFixed(2)}`),
+ );
+ }
+ out.log(
+ `reports check ${opts.siteId}: ${resolved.reports.length} report(s), ${resolved.moments.length} moment(s) — compose would pass.`,
+ );
+ return 0;
+ } catch (err) {
+ if (!(err instanceof ComposeReportsError)) {
+ out.error(`reports check ${opts.siteId}: ${(err as Error).message}`);
+ return 1;
+ }
+ out.error(`reports check ${opts.siteId}: ${err.problems.length} problem(s) — compose would fail:`);
+ for (const line of formatComposeReportsProblems(err.problems)) out.error(` ${line}`);
+ return 1;
+ }
+}
+
+// ─── reports verify-quotes ───
+
+export type QuoteStatus =
+ | "ok"
+ | "drift"
+ | "en-orig-drift"
+ | "missing-record"
+ | "no-cues"
+ | "missing-post"
+ | "unreadable";
+
+export type QuoteResult = {
+ citation: string;
+ kind: "video" | "audio" | "post";
+ channel: string;
+ id: string;
+ start?: number;
+ end?: number;
+ cited: boolean;
+ status: QuoteStatus;
+ // The best score and the track it came from (a post: its text).
+ score?: number;
+ track?: string;
+ // Every track's score; and the en-orig track's, when the record has one.
+ tracks?: { name: string; score: number }[];
+ enOrig?: number;
+ message?: string;
+};
+
+export const EN_ORIG_TRACK = "transcript.en-orig.vtt";
+
+// Every video, audio and post citation of a report, checked against the
+// corpus at `paths.channelsDir` — cited or not (an uncited one is marked).
+export async function verifyReportQuotes(
+ report: Report,
+ opts: { paths: Paths; now?: string },
+): Promise<QuoteResult[]> {
+ const { paths } = opts;
+ const now = opts.now ?? new Date().toISOString();
+ const used = new Set(reportCitationNumbers(report).keys());
+ const results: QuoteResult[] = [];
+ const unreadable = new Map<string, string | null>();
+ const textProblem = async (slug: string) => {
+ if (!unreadable.has(slug)) {
+ try {
+ await assertChannelTextReadable(paths, slug, await readChannelConfig(paths, slug).catch(() => null));
+ unreadable.set(slug, null);
+ } catch (e) {
+ unreadable.set(slug, (e as Error).message);
+ }
+ }
+ return unreadable.get(slug) ?? null;
+ };
+ const posts = new Map<string, Map<string, string>>();
+ for (const [cid, c] of Object.entries(report.citations ?? {})) {
+ if (c.kind !== "video" && c.kind !== "audio" && c.kind !== "post") continue;
+ const base = {
+ citation: cid,
+ kind: c.kind,
+ channel: c.channel,
+ id: c.id,
+ cited: used.has(cid),
+ ...(c.kind === "post" ? {} : { start: c.start, end: c.end }),
+ };
+ const text = await textProblem(c.channel);
+ if (text) {
+ results.push({ ...base, status: "unreadable", message: text });
+ continue;
+ }
+ if (c.kind === "post") {
+ if (!posts.has(c.channel)) {
+ const all = await readAllPosts(path.join(paths.channelsDir, c.channel)).catch(() => []);
+ posts.set(c.channel, new Map(all.map((p) => [p.id, p.text])));
+ }
+ const postText = posts.get(c.channel)!.get(c.id);
+ if (postText === undefined) {
+ results.push({ ...base, status: "missing-post", message: `no post ${c.id} in the posts archive of "${c.channel}"` });
+ continue;
+ }
+ const v = quoteVerification(c.quote, postText, now);
+ results.push({
+ ...base,
+ score: v.quoteScore,
+ track: "post",
+ status: quoteDrifted(v) ? "drift" : "ok",
+ ...(quoteDrifted(v) ? { message: quoteDriftMessage(v.quoteScore) } : {}),
+ });
+ continue;
+ }
+ const record = await readCitedRecord(paths.channelsDir, c.channel, c.id);
+ if (!record) {
+ results.push({ ...base, status: "missing-record", message: `no record ${c.channel}/${c.id} (no metadata or transcript in its data dir)` });
+ continue;
+ }
+ if (record.cues.length === 0) {
+ results.push({ ...base, status: "no-cues", message: `${c.channel}/${c.id} has no transcript cues to check the quote against` });
+ continue;
+ }
+ const checked = checkSpanQuote(record, c, now);
+ const score = checked.verification.quoteScore ?? 0;
+ const best = checked.tracks.reduce<{ name: string; score: number } | null>(
+ (b, t) => (!b || t.score > b.score ? t : b),
+ null,
+ );
+ const enOrig = checked.tracks.find((t) => t.name === EN_ORIG_TRACK)?.score;
+ let status: QuoteStatus = "ok";
+ let message: string | undefined;
+ if (quoteDrifted(checked.verification)) {
+ status = "drift";
+ message = quoteDriftMessage(score);
+ } else if (enOrig !== undefined && enOrig < QUOTE_DRIFT_THRESHOLD) {
+ status = "en-orig-drift";
+ message =
+ `the quote matches ${best?.name ?? "a track"} (${score.toFixed(2)}) but only ${Math.round(enOrig * 100)}% of ` +
+ `the en-orig track, the words as spoken — a served \`en\` track can rewrite them: quote en-orig, or check the audio`;
+ }
+ results.push({
+ ...base,
+ score: roundScore(score),
+ ...(best ? { track: best.name } : {}),
+ tracks: checked.tracks,
+ ...(enOrig !== undefined ? { enOrig } : {}),
+ status,
+ ...(message ? { message } : {}),
+ });
+ }
+ return results;
+}
+
+function describe(r: QuoteResult): string {
+ const where = r.kind === "post" ? `${r.channel}/${r.id}` : `${r.channel}/${r.id} ${r.start}–${r.end} s`;
+ const score = r.score === undefined ? "" : ` ${r.score.toFixed(2)}${r.track ? ` ${r.track}` : ""}`;
+ const orig = r.enOrig === undefined || r.track === EN_ORIG_TRACK ? "" : ` · en-orig ${r.enOrig.toFixed(2)}`;
+ const label = r.status === "ok" ? "ok" : r.status.toUpperCase();
+ return ` ${label.padEnd(14)} ${r.citation}${r.cited ? "" : " (not cited)"} ${r.kind} ${where}${score}${orig}` +
+ (r.message ? `\n${" ".repeat(17)}${r.message}` : "");
+}
+
+export async function verifyQuotesMain(
+ opts: { file: string; json?: boolean; paths?: Paths; now?: string },
+ out: Out = console,
+): Promise<number> {
+ const paths = opts.paths ?? getPaths();
+ let raw: unknown;
+ try {
+ raw = JSON.parse(await readFile(opts.file, "utf8"));
+ } catch (err) {
+ out.error(`reports verify-quotes: ${opts.file} is not readable JSON (${(err as Error).message})`);
+ return 2;
+ }
+ const parsed = parseReport(raw);
+ if (!parsed.ok) {
+ out.error(`reports verify-quotes: ${opts.file} is not a report:`);
+ for (const p of parsed.problems) out.error(` ${p.path}: ${p.message}`);
+ return 1;
+ }
+ const results = await verifyReportQuotes(parsed.value, { paths, ...(opts.now ? { now: opts.now } : {}) });
+ const bad = results.filter((r) => r.status !== "ok");
+ if (opts.json) {
+ out.log(JSON.stringify({ file: opts.file, problems: parsed.problems, results }, null, 2));
+ } else {
+ for (const p of parsed.problems) out.error(` invalid: ${p.path}: ${p.message}`);
+ for (const r of results) (r.status === "ok" ? out.log : out.error)(describe(r));
+ const counts = new Map<QuoteStatus, number>();
+ for (const r of results) counts.set(r.status, (counts.get(r.status) ?? 0) + 1);
+ out.log(
+ `reports verify-quotes: ${results.length} quote(s): ` +
+ [...counts].map(([s, n]) => `${n} ${s}`).join(", ") +
+ (parsed.problems.length ? `; ${parsed.problems.length} validation problem(s)` : ""),
+ );
+ }
+ return bad.length > 0 || parsed.problems.length > 0 ? 1 : 0;
+}
diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts
@@ -229,7 +229,7 @@ export type ComposedReports = {
// ─── Reading the corpus ───
-type CitedRecord = {
+export type CitedRecord = {
summary: Pick<TranscriptSummary, "id" | "slug" | "title" | "uploadDate" | "platform" | "webpageUrl" | "channel">;
cues: Cue[];
// Every transcript of the record that has cues, by file name: the
@@ -304,6 +304,45 @@ export async function readCitedRecord(
return { summary, cues, tracks, archiveOrg, wayback };
}
+// THE SPAN QUOTE CHECK — the one compose runs, and `archilyzer reports
+// verify-quotes` and `reports check` report: the quote against the cue window
+// of EVERY transcript the record has (lib/citations/verify.ts), the best match
+// is the verification (its method names the track). `tracks` is each track's
+// own score, so a caller can say what the `en-orig` track — the words as
+// spoken — makes of a quote a served `en` rewrite matched.
+export type SpanQuoteCheck = {
+ verification: ReturnType<typeof quoteVerification>;
+ // The best track's cues (what a moment page shows), or null when the record
+ // had no track and its default cues were used.
+ cues: Cue[] | null;
+ tracks: { name: string; score: number }[];
+};
+
+export function checkSpanQuote(
+ record: Pick<CitedRecord, "cues" | "tracks">,
+ c: { quote: string; start: number; end: number },
+ now: string,
+): SpanQuoteCheck {
+ let best: { name: string; cues: Cue[]; v: ReturnType<typeof quoteVerification> } | null = null;
+ const tracks: SpanQuoteCheck["tracks"] = [];
+ for (const t of record.tracks) {
+ const v = quoteVerification(c.quote, cueWindowText(t.cues, c.start, c.end), now);
+ tracks.push({ name: t.name, score: v.quoteScore ?? 0 });
+ if (!best || (v.quoteScore ?? 0) > (best.v.quoteScore ?? 0)) best = { name: t.name, cues: t.cues, v };
+ }
+ return best
+ ? { verification: { ...best.v, method: `${best.v.method}; text: ${best.name}` }, cues: best.cues, tracks }
+ : { verification: quoteVerification(c.quote, cueWindowText(record.cues, c.start, c.end), now), cues: null, tracks };
+}
+
+// The sentence compose fails a drifted quote with.
+export function quoteDriftMessage(score: number | undefined): string {
+ return (
+ `the quote matches ${Math.round((score ?? 0) * 100)}% of what the record says there ` +
+ `(at least ${Math.round(QUOTE_DRIFT_THRESHOLD * 100)}% is required): quote it verbatim, or fix the span`
+ );
+}
+
const isoDay = (uploadDate: string | undefined): string | undefined =>
uploadDate && /^\d{8}$/.test(uploadDate)
? `${uploadDate.slice(0, 4)}-${uploadDate.slice(4, 6)}-${uploadDate.slice(6, 8)}`
@@ -541,25 +580,17 @@ export async function resolveSiteReports(opts: ResolveSiteReportsOptions): Promi
problems.push({ kind: "no-cues", citation: ref, report: report.id, message: `${c.channel}/${c.id} has no transcript cues to check the quote against` });
continue;
}
- let best: { name: string; cues: Cue[]; v: ReturnType<typeof quoteVerification> } | null = null;
- for (const t of record.tracks) {
- const v = quoteVerification(c.quote, cueWindowText(t.cues, c.start, c.end), now);
- if (!best || (v.quoteScore ?? 0) > (best.v.quoteScore ?? 0)) best = { name: t.name, cues: t.cues, v };
- }
- c.verification = best
- ? { ...best.v, method: `${best.v.method}; text: ${best.name}` }
- : quoteVerification(c.quote, cueWindowText(record.cues, c.start, c.end), now);
+ const checked = checkSpanQuote(record, c, now);
+ c.verification = checked.verification;
const mk = momentKeyOf(c);
- if (best && mk && !momentCues.has(mk)) momentCues.set(mk, best.cues);
+ if (checked.cues && mk && !momentCues.has(mk)) momentCues.set(mk, checked.cues);
}
if (quoteDrifted(c.verification)) {
problems.push({
kind: "quote-drift",
citation: ref,
report: report.id,
- message:
- `the quote matches ${Math.round((c.verification.quoteScore ?? 0) * 100)}% of what the record says there ` +
- `(at least ${Math.round(QUOTE_DRIFT_THRESHOLD * 100)}% is required): quote it verbatim, or fix the span`,
+ message: quoteDriftMessage(c.verification.quoteScore),
});
}
}
diff --git a/common/publish/reportVideo.ts b/common/publish/reportVideo.ts
@@ -0,0 +1,251 @@
+// A REPORT'S VIDEO — `archilyzer reports attach-video <report.json> <video>`:
+// the video encoded (or remuxed) to fit the publish limit, written beside the
+// report as `video.mp4` with a poster, and `video` set in report.json.
+//
+// The limit is the one compose enforces on the report's video
+// (lib/builtExport.ts PUBLISH_MAX_FILE_BYTES, 24 MiB — Pages allows 25 per
+// file); a video over it fails compose ("report-video"). So:
+//
+// - an mp4 already H.264 (+ AAC, or no audio) and under the limit is
+// REMUXED, not re-encoded: streams copied, `+faststart` so it plays
+// before it has loaded;
+// - anything else is ENCODED: H.264 + AAC at an average bitrate the
+// duration allows inside 96% of the limit (capped at 2.5 Mb/s of video,
+// no wider than 1280 px, narrower as the bitrate falls). A single pass
+// can land over its target, so a result over the limit is encoded again
+// at a bitrate scaled down by the overshoot, up to three times;
+// - a video so long that it would get under 100 kb/s of picture is refused
+// with its length — trim it, the encoder cannot make it watchable.
+//
+// The poster is the one given (png, jpg or webp), else the report's existing
+// one when it is on disk, else a frame from 10% into the video. Nothing is
+// written into report.json unless the video (and poster) are in place and
+// its `video` validates.
+
+import { copyFile, readFile, rename, rm, stat, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { execa } from "execa";
+import { PUBLISH_MAX_FILE_BYTES, publishFileSizeProblem } from "../lib/builtExport";
+import { parseReport } from "../lib/report/validate";
+
+export const REPORT_VIDEO_NAME = "video.mp4";
+export const REPORT_POSTER_NAME = "poster.jpg";
+// The share of the limit an encode aims at: the rest is container overhead
+// and the encoder's own overshoot.
+export const ATTACH_TARGET_FRACTION = 0.96;
+export const MAX_VIDEO_KBPS = 2500;
+export const MIN_VIDEO_KBPS = 100;
+const MAX_ENCODE_ATTEMPTS = 3;
+const POSTER_EXTS = new Set([".png", ".jpg", ".jpeg", ".webp"]);
+
+export type VideoProbe = {
+ durationSec: number;
+ formatName: string;
+ videoCodec: string | null;
+ audioCodec: string | null;
+ width: number | null;
+};
+
+export type EncodePlan =
+ | { mode: "remux" }
+ | { mode: "encode"; videoKbps: number; audioKbps: number; maxWidth: number }
+ | { mode: "refuse"; reason: string };
+
+// What to do with a video of `bytes` described by `probe`, to fit `limitBytes`.
+export function encodePlan(probe: VideoProbe, bytes: number, limitBytes = PUBLISH_MAX_FILE_BYTES): EncodePlan {
+ if (!probe.videoCodec) return { mode: "refuse", reason: "it has no video stream" };
+ if (!(probe.durationSec > 0)) return { mode: "refuse", reason: "its duration cannot be read" };
+ const isMp4 = /(^|,)(mp4|mov)(,|$)/.test(probe.formatName);
+ if (
+ isMp4 &&
+ probe.videoCodec === "h264" &&
+ (probe.audioCodec === null || probe.audioCodec === "aac") &&
+ bytes <= limitBytes
+ ) {
+ return { mode: "remux" };
+ }
+ const totalKbps = Math.floor((limitBytes * ATTACH_TARGET_FRACTION * 8) / 1000 / probe.durationSec);
+ const audioKbps = probe.audioCodec === null ? 0 : totalKbps >= 1000 ? 128 : 64;
+ // ~2% for the container.
+ const videoKbps = Math.min(MAX_VIDEO_KBPS, Math.floor((totalKbps - audioKbps) * 0.98));
+ if (videoKbps < MIN_VIDEO_KBPS) {
+ const minutes = (probe.durationSec / 60).toFixed(1);
+ return {
+ mode: "refuse",
+ reason:
+ `at ${minutes} min it would get ${Math.max(0, videoKbps)} kb/s of picture inside ` +
+ `${(limitBytes / (1024 * 1024)).toFixed(0)} MiB (at least ${MIN_VIDEO_KBPS} is watchable) — trim it`,
+ };
+ }
+ const maxWidth = videoKbps >= 1200 ? 1280 : videoKbps >= 500 ? 854 : 640;
+ return { mode: "encode", videoKbps, audioKbps, maxWidth };
+}
+
+export function encodeArgs(src: string, dst: string, plan: Exclude<EncodePlan, { mode: "refuse" }>): string[] {
+ const head = ["-nostdin", "-hide_banner", "-v", "error", "-y", "-i", src];
+ if (plan.mode === "remux") return [...head, "-map", "0:v:0", "-map", "0:a:0?", "-c", "copy", "-movflags", "+faststart", dst];
+ const v = plan.videoKbps;
+ return [
+ ...head,
+ "-map", "0:v:0",
+ "-map", "0:a:0?",
+ "-vf", `scale='min(${plan.maxWidth},iw)':-2`,
+ "-c:v", "libx264",
+ "-preset", "medium",
+ "-b:v", `${v}k`,
+ "-maxrate", `${Math.round(v * 1.5)}k`,
+ "-bufsize", `${v * 2}k`,
+ "-pix_fmt", "yuv420p",
+ ...(plan.audioKbps > 0 ? ["-c:a", "aac", "-b:a", `${plan.audioKbps}k`, "-ac", "2"] : ["-an"]),
+ "-movflags", "+faststart",
+ dst,
+ ];
+}
+
+export function posterArgs(src: string, dst: string, atSec: number): string[] {
+ return [
+ "-nostdin", "-hide_banner", "-v", "error", "-y",
+ "-ss", String(Math.max(0, Math.round(atSec * 100) / 100)),
+ "-i", src,
+ "-frames:v", "1",
+ "-vf", "scale='min(1280,iw)':-2",
+ "-q:v", "3",
+ dst,
+ ];
+}
+
+export async function probeVideo(ffprobeBin: string, file: string): Promise<VideoProbe> {
+ const { stdout } = await execa(ffprobeBin, [
+ "-v", "error", "-print_format", "json", "-show_format", "-show_streams", file,
+ ]);
+ const doc = JSON.parse(stdout) as {
+ format?: { duration?: string; format_name?: string };
+ streams?: { codec_type?: string; codec_name?: string; width?: number; duration?: string }[];
+ };
+ const streams = doc.streams ?? [];
+ const video = streams.find((s) => s.codec_type === "video");
+ const audio = streams.find((s) => s.codec_type === "audio");
+ const duration = Number(doc.format?.duration ?? video?.duration ?? NaN);
+ return {
+ durationSec: Number.isFinite(duration) ? duration : 0,
+ formatName: doc.format?.format_name ?? "",
+ videoCodec: video?.codec_name ?? null,
+ audioCodec: audio?.codec_name ?? null,
+ width: video?.width ?? null,
+ };
+}
+
+export type AttachVideoOptions = {
+ reportFile: string;
+ video: string;
+ poster?: string;
+ caption?: string;
+ ffmpegBin: string;
+ ffprobeBin: string;
+ limitBytes?: number;
+ onLog?: (line: string) => void;
+};
+
+export type AttachedVideo = {
+ video: { src: string; poster?: string; caption?: string };
+ bytes: number;
+ mode: "remux" | "encode";
+ attempts: number;
+};
+
+export class AttachVideoError extends Error {}
+
+const sizeOf = async (p: string) => (await stat(p).catch(() => null))?.size ?? null;
+
+export async function attachReportVideo(opts: AttachVideoOptions): Promise<AttachedVideo> {
+ const log = opts.onLog ?? (() => {});
+ const limit = opts.limitBytes ?? PUBLISH_MAX_FILE_BYTES;
+ const dir = path.dirname(path.resolve(opts.reportFile));
+
+ // The report first: nothing is encoded for a file that is not one.
+ let raw: Record<string, unknown>;
+ try {
+ raw = JSON.parse(await readFile(opts.reportFile, "utf8")) as Record<string, unknown>;
+ } catch (err) {
+ throw new AttachVideoError(`${opts.reportFile} is not readable JSON (${(err as Error).message})`);
+ }
+ if (!parseReport(raw).ok) throw new AttachVideoError(`${opts.reportFile} is not a report (archilyzer reports check)`);
+
+ const srcBytes = await sizeOf(opts.video);
+ if (srcBytes === null) throw new AttachVideoError(`${opts.video} does not exist`);
+ if (opts.poster !== undefined) {
+ if (!POSTER_EXTS.has(path.extname(opts.poster).toLowerCase())) {
+ throw new AttachVideoError(`the poster must be a png, jpg or webp: ${opts.poster}`);
+ }
+ if ((await sizeOf(opts.poster)) === null) throw new AttachVideoError(`${opts.poster} does not exist`);
+ }
+
+ const probe = await probeVideo(opts.ffprobeBin, opts.video);
+ let plan = encodePlan(probe, srcBytes, limit);
+ if (plan.mode === "refuse") throw new AttachVideoError(`${opts.video} cannot be attached: ${plan.reason}`);
+ log(
+ plan.mode === "remux"
+ ? `${path.basename(opts.video)}: H.264 and under the limit — remuxing (+faststart)`
+ : `${path.basename(opts.video)}: ${probe.durationSec.toFixed(1)} s ${probe.videoCodec}/${probe.audioCodec ?? "no audio"} — encoding at ${plan.videoKbps} kb/s video, ${plan.audioKbps} kb/s audio, ≤${plan.maxWidth} px wide`,
+ );
+
+ const dst = path.join(dir, REPORT_VIDEO_NAME);
+ const tmp = path.join(dir, `.${REPORT_VIDEO_NAME}.${process.pid}.tmp.mp4`);
+ let bytes = 0;
+ let attempts = 0;
+ try {
+ for (;;) {
+ attempts++;
+ await execa(opts.ffmpegBin, encodeArgs(opts.video, tmp, plan));
+ bytes = (await sizeOf(tmp)) ?? 0;
+ if (bytes > 0 && bytes <= limit) break;
+ if (plan.mode !== "encode" || attempts >= MAX_ENCODE_ATTEMPTS) {
+ throw new AttachVideoError(
+ publishFileSizeProblem(REPORT_VIDEO_NAME, bytes) ?? `ffmpeg wrote nothing for ${opts.video}`,
+ );
+ }
+ const scaled = Math.floor(plan.videoKbps * ((limit * ATTACH_TARGET_FRACTION) / bytes) * 0.95);
+ if (scaled < MIN_VIDEO_KBPS) {
+ throw new AttachVideoError(`${opts.video} does not fit the limit above ${MIN_VIDEO_KBPS} kb/s of picture — trim it`);
+ }
+ log(` ${(bytes / (1024 * 1024)).toFixed(1)} MiB is over the limit — again at ${scaled} kb/s`);
+ plan = { ...plan, videoKbps: scaled };
+ }
+ await rename(tmp, dst);
+ } finally {
+ await rm(tmp, { force: true });
+ }
+ log(` ${REPORT_VIDEO_NAME}: ${(bytes / (1024 * 1024)).toFixed(2)} MiB (limit ${(limit / (1024 * 1024)).toFixed(0)} MiB)`);
+
+ // The poster: given, kept, or drawn from the video.
+ const existing = (raw.video as { poster?: unknown; caption?: unknown } | undefined) ?? undefined;
+ let poster: string | undefined;
+ if (opts.poster !== undefined) {
+ poster = `poster${path.extname(opts.poster).toLowerCase()}`;
+ if (path.resolve(opts.poster) !== path.join(dir, poster)) await copyFile(opts.poster, path.join(dir, poster));
+ } else if (typeof existing?.poster === "string" && (await sizeOf(path.join(dir, existing.poster))) !== null) {
+ poster = existing.poster;
+ } else {
+ poster = REPORT_POSTER_NAME;
+ await execa(opts.ffmpegBin, posterArgs(dst, path.join(dir, poster), probe.durationSec * 0.1));
+ }
+ const posterProblem = publishFileSizeProblem(poster, (await sizeOf(path.join(dir, poster))) ?? 0);
+ if (posterProblem) throw new AttachVideoError(posterProblem);
+
+ const caption = opts.caption ?? (typeof existing?.caption === "string" ? existing.caption : undefined);
+ const video = { src: REPORT_VIDEO_NAME, poster, ...(caption ? { caption } : {}) };
+ const next = { ...raw, video };
+ // Only the video's own problems refuse the write: a report with others (a
+ // draft's dangling link) gets its video all the same, and `reports check`
+ // says the rest.
+ const parsed = parseReport(next);
+ const videoProblems = parsed.problems.filter((p) => p.path === "video" || p.path.startsWith("video."));
+ if (!parsed.ok || videoProblems.length > 0) {
+ const first = videoProblems[0] ?? parsed.problems[0];
+ throw new AttachVideoError(`report.json would not validate with the video: ${first ? `${first.path}: ${first.message}` : "?"}`);
+ }
+ const tmpJson = `${opts.reportFile}.${process.pid}.tmp`;
+ await writeFile(tmpJson, `${JSON.stringify(next, null, 2)}\n`);
+ await rename(tmpJson, opts.reportFile);
+ return { video, bytes, mode: plan.mode, attempts };
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Reports can be checked, their quotes verified and their video attached from the command line.** `archilyzer reports check <site> [--reports a,b] [--allow-missing-media]` says what the site's compose would say about its reports — each report validated, every cited quote checked against its record, every cited moment's prepared media present and current, the report's video under the publish limit — without a build, and exits 1 with the list; `--reports` checks those reports, drafts not yet in site.json included. `archilyzer reports verify-quotes <report.json> [--json]` runs compose's own quote check over every video, audio and post quote of one report and prints each one's score and the transcript it matched best, and, where the record has an `en-orig` track, that track's score: a quote that matches a served `en` track but not `en-orig` — the words as spoken — is reported, though compose would pass it. `archilyzer reports attach-video <report.json> <video> [--poster <image>] [--caption <line>]` makes the video the report's: an H.264 mp4 under the 24 MiB limit is remuxed, anything else encoded to fit (refused, with its length, when it would be unwatchable), written beside report.json as `video.mp4` with a poster — the one given, the one it had, or a frame of the video — and `video` set in report.json.
- **A file transcription is timed from the engine, and a short one does not queue behind batches.** `pnpm ops transcribe`'s `durationMs` is now the engine's own time — from the moment a worker took the job — and the new `waitedMs` is how long it waited for a free worker; the job's log says both. A file of up to 15 minutes of audio (a window, usually) waits in the worker pool ahead of every parked transcription, a manual batch's next video included, not only ahead of the transcription lane; a longer file keeps its place behind batches queued before it. Nothing running is interrupted. Needs a restart of the editor.
- **Heavy work takes turns, above a memory floor.** A publish stage's `next build` — a site's, the hub's, the homepage's — now waits for the machine's one heavy slot, which every e2e run and any `pnpm heavy -- <cmd>` (a video render) take too, and then until at least 6000 MB is available; the stage's log says whom it waits behind ("waiting for the heavy slot — held by …") or how much memory there is ("waiting for memory — 4210 MB available, the floor is 6000 MB"). Cancel still stops it. `HEAVY_MIN_FREE_MB` moves the floor (`0` turns it off) and `HEAVY=0` skips the gate. Needs a restart of the editor.
- **A curated tag can exist on some sites only.** A tag's new **Sites** field on /tags (`sites` in `transcripts/tags.json`; `pnpm ops tags` takes it in a define) names the sites it exists on. Its rules then fire, and its pins apply, only to videos on those sites' channels, and every other site drops it from its records, its counts and its `/tags.json` — where **Hidden** only hid the chip. Empty is every site, as before. Setting it, or changing the channels of those sites, re-derives the corpus's tags once at the next index update. The Eva tags are what this is for: they belong on Anilyzer alone.