commit ca110d06051901cc1eab9488daa942e0ba9af149
parent 561dcff167cf4f22747d415693c308a0f59c1fdd
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 7 Oct 2026 21:10:02 -0400
Merge reports/report-video (a report can carry a video)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
11 files changed, 126 insertions(+), 4 deletions(-)
diff --git a/REPORT.md b/REPORT.md
@@ -6,7 +6,7 @@ One cited report, format `"archilyzer-report"`, version 1, persisted to `transcr
A **fact-check** (`"kind": "factcheck"`) is sections (chapters) of claims, each with a verdict and its findings. A **sweep** (`"kind": "sweep"`) is sections with no verdicts, or bodies that cite inline. Markdown fields (`summary`, a section's `body`, a claim's `findings`) cite with `[label](cite:<id>)`.
-`common/lib/report/validate.ts` reports every problem with its JSON path: an unknown key, a reference that names nothing (a listed citation, a claim's source sentence, a `cite:` link, the subject), a section or claim id used twice (they share one namespace: the report page's anchors), a sweep's claim with a verdict, `updated` before `published`, and every citation problem CITATIONS.md lists. Whether a still exists and whether a quote matches its cues are checked when the site is composed.
+`common/lib/report/validate.ts` reports every problem with its JSON path: an unknown key, a reference that names nothing (a listed citation, a claim's source sentence, a `cite:` link, the subject), a section or claim id used twice (they share one namespace: the report page's anchors), a sweep's claim with a verdict, `updated` before `published`, and every citation problem CITATIONS.md lists. Whether a still or the report's video exists, whether the video fits the publish limit, and whether a quote matches its cues are checked when the site is composed.
Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.
@@ -26,6 +26,7 @@ Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/f
| `published` | no | When the report was published: `YYYY-MM-DD` or an ISO 8601 date-time with a zone. |
| `updated` | no | When it was last changed, in the same form; not before `published`. |
| `subject` | no | The document under review, when the report reviews one: `{ "source": "<id>" }`, an id in `sources`. |
+| `video` | no | A video of the report, shown at the head of its page under the title: `{ "src": "video.mp4", "poster": "poster.jpg", "caption": "…" }`. `src` is an mp4 and `poster` an image (png, jpg or webp), both relative to the report's directory; the caption is one line. Absent = none. |
| `verdicts` | no | Overrides of the shared verdict vocabulary's labels and colours, by verdict (`CORROBORATED`, `PARTLY`, `CONTRADICTED`, `NOT_FOUND`, `UNTESTABLE`): `{ "label": "…", "color": "#rrggbb" }`, each key optional. Absent = the shared defaults. |
| `sources` | no | The documents the report's `source` citations quote, by id — see [CITATIONS.md](CITATIONS.md). Absent = none. |
| `citations` | no | The report's citations, by id — see [CITATIONS.md](CITATIONS.md). A citation is cited from markdown with `[label](cite:<id>)` and listed under the claims that rest on it. Absent = none. |
diff --git a/common/lib/report/docs.ts b/common/lib/report/docs.ts
@@ -48,8 +48,9 @@ export function renderReportMarkdown(): string {
"source sentence, a `cite:` link, the subject), a section or claim id used " +
"twice (they share one namespace: the report page's anchors), a sweep's claim " +
"with a verdict, `updated` before `published`, and every citation problem " +
- "CITATIONS.md lists. Whether a still exists and whether a quote matches its cues " +
- "are checked when the site is composed.",
+ "CITATIONS.md lists. Whether a still or the report's video exists, whether the " +
+ "video fits the publish limit, and whether a quote matches its cues are checked " +
+ "when the site is composed.",
);
out.push("");
out.push(REGENERATE_SCHEMA_DOC);
diff --git a/common/lib/report/report.test.ts b/common/lib/report/report.test.ts
@@ -256,3 +256,15 @@ test("a claim's flag: one line of at most 60 characters", () => {
assert.deepEqual(paths(validateReport(withFlag(" "))), ["sections[0].claims[0].flag"]);
assert.deepEqual(paths(validateReport(withFlag("two\nlines"))), ["sections[0].claims[0].flag"]);
});
+
+test("a report's video: an mp4 and an image poster, relative to the report, and a one-line caption", () => {
+ const withVideo = (video: unknown) => ({ ...structuredClone(fixture()), video });
+ assert.deepEqual(validateReport(withVideo({ src: "video.mp4" })), []);
+ assert.deepEqual(validateReport(withVideo({ src: "media/cut.mp4", poster: "media/cut.jpg", caption: "The cut" })), []);
+ assert.deepEqual(paths(validateReport(withVideo({ src: "video.webm" }))), ["video.src"]);
+ assert.deepEqual(paths(validateReport(withVideo({ src: "../video.mp4" }))), ["video.src"]);
+ assert.deepEqual(paths(validateReport(withVideo({ src: "https://x.test/v.mp4" }))), ["video.src"]);
+ assert.deepEqual(paths(validateReport(withVideo({ src: "video.mp4", poster: "poster.gif" }))), ["video.poster"]);
+ assert.deepEqual(paths(validateReport(withVideo({ src: "video.mp4", caption: "two\nlines" }))), ["video.caption"]);
+ assert.equal(parseReport(withVideo({ src: "video.mp4", autoplay: true })).ok, false);
+});
diff --git a/common/lib/report/schema.ts b/common/lib/report/schema.ts
@@ -76,6 +76,7 @@ export const reportSchema = z.strictObject({
published: text.optional(),
updated: text.optional(),
subject: z.strictObject({ source: text }).optional(),
+ video: z.strictObject({ src: text, poster: text.optional(), caption: text.optional() }).optional(),
verdicts: z.partialRecord(verdict, verdictOverride).optional(),
sources: z.record(text, sourceSchema).optional(),
citations: z.record(text, citationSchema).optional(),
@@ -111,6 +112,8 @@ export const REPORT_FIELD_DOCS: FieldDocs<Report> = {
updated: "When it was last changed, in the same form; not before `published`.",
subject:
"The document under review, when the report reviews one: `{ \"source\": \"<id>\" }`, an id in `sources`.",
+ video:
+ "A video of the report, shown at the head of its page under the title: `{ \"src\": \"video.mp4\", \"poster\": \"poster.jpg\", \"caption\": \"…\" }`. `src` is an mp4 and `poster` an image (png, jpg or webp), both relative to the report's directory; the caption is one line. Absent = none.",
verdicts: `Overrides of the shared verdict vocabulary's labels and colours, by verdict (${VERDICTS.map((v) => `\`${v}\``).join(", ")}): \`{ "label": "…", "color": "#rrggbb" }\`, each key optional. Absent = the shared defaults.`,
sources: "The documents the report's `source` citations quote, by id — see [CITATIONS.md](CITATIONS.md). Absent = none.",
citations:
diff --git a/common/lib/report/validate.ts b/common/lib/report/validate.ts
@@ -18,7 +18,7 @@
// claim), and every `[label](cite:<id>)` link in the summary, the bodies
// and the findings.
//
-// What this cannot check is the disk and the corpus — that a still exists, that
+// What this cannot check is the disk and the corpus — that a still or the video exists, that
// a quote matches its cues; compose does those, where both are at hand.
import {
@@ -26,6 +26,7 @@ import {
isDateTime,
jsonPath,
problem,
+ relativePathProblem,
zodProblems,
type Parsed,
type PathSegment,
@@ -66,6 +67,21 @@ function reportProblems(report: Report, opts: ReportValidateOptions): Problem[]
if (report.subtitle !== undefined && /[\r\n]/.test(report.subtitle)) {
out.push(problem(["subtitle"], "must be one line"));
}
+ if (report.video) {
+ const { src, poster, caption } = report.video;
+ const srcProblem = relativePathProblem(src);
+ if (srcProblem) out.push(problem(["video", "src"], srcProblem));
+ else if (!/\.mp4$/i.test(src)) out.push(problem(["video", "src"], "must be an mp4"));
+ if (poster !== undefined) {
+ const posterProblem = relativePathProblem(poster);
+ if (posterProblem) out.push(problem(["video", "poster"], posterProblem));
+ else if (!/\.(png|jpe?g|webp)$/i.test(poster)) out.push(problem(["video", "poster"], "must be a png, jpg or webp"));
+ }
+ if (caption !== undefined) {
+ if (blank(caption)) out.push(problem(["video", "caption"], "must not be blank"));
+ else if (/[\r\n]/.test(caption)) out.push(problem(["video", "caption"], "must be one line"));
+ }
+ }
for (const key of ["published", "updated"] as const) {
const v = report[key];
if (v !== undefined && !isReportDate(v)) {
diff --git a/common/lib/report/views.test.ts b/common/lib/report/views.test.ts
@@ -407,3 +407,11 @@ test("what the check found: ruled claims by verdict, changed first, confirmed la
long.sections[0].claims![0].gist = "x".repeat(241);
assert.ok(validateReport(long).some((p) => p.path === "sections[0].claims[0].gist"));
});
+
+test("a report's video rides its page view as site-root paths; a report without one carries none", () => {
+ const withVideo = { ...report, video: { src: "video.mp4", poster: "media/poster.jpg", caption: "The cut" } };
+ assert.deepEqual(validateReport(withVideo), []);
+ const v = buildReportPageView(withVideo, { record: (c) => record(c.channel, c.id) });
+ assert.deepEqual(v.video, { src: `/reports/${report.id}/video.mp4`, poster: `/reports/${report.id}/media/poster.jpg`, caption: "The cut" });
+ assert.equal("video" in view, false);
+});
diff --git a/common/lib/report/views.ts b/common/lib/report/views.ts
@@ -274,6 +274,8 @@ export type ReportPageView = {
series?: string;
title: string;
subtitle?: string;
+ // A video of the report, shown under the title: site-root paths.
+ video?: ReportVideoView;
summary?: string;
// How it was checked (markdown, cites nothing).
method?: string;
@@ -298,6 +300,8 @@ export type ReportPageView = {
history?: ReportHistoryRef;
};
+export type ReportVideoView = { src: string; poster?: string; caption?: string };
+
// What a report page offers to download: its exports (html, pdf, md, the
// evidence pack as zip) and its citations (json, csv). Each key is present
// only when the file is published.
@@ -528,6 +532,13 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver)
series: report.series,
title: report.title,
subtitle: report.subtitle,
+ video: report.video
+ ? defined({
+ src: reportAssetPath(report.id, report.video.src),
+ poster: report.video.poster ? reportAssetPath(report.id, report.video.poster) : undefined,
+ caption: report.video.caption,
+ })
+ : undefined,
summary: report.summary,
method: report.method,
published: report.published,
diff --git a/common/publish/composeReports.test.ts b/common/publish/composeReports.test.ts
@@ -507,6 +507,39 @@ test("a citation without prepared media fails compose with the list, unless --al
}
});
+test("a report's video: copied beside its page and named in the view; a missing or oversize one fails compose", async () => {
+ const dir = path.join(paths.sitesDir, "cited", "reports", REPORT);
+ const file = path.join(dir, "report.json");
+ writeJson(file, { ...report(), video: { src: "video.mp4", poster: "poster.jpg", caption: "The cut" } });
+ try {
+ await assert.rejects(compose("cited"), (e: unknown) => {
+ assert.ok(e instanceof ComposeReportsError);
+ assert.deepEqual(e.problems.map((p) => [p.kind, p.report]), [["report-video", REPORT], ["report-video", REPORT]]);
+ assert.match(e.message, /video\.mp4 does not exist/);
+ return true;
+ });
+ writeText(path.join(dir, "video.mp4"), "mp4-bytes");
+ writeText(path.join(dir, "poster.jpg"), "jpg-bytes");
+ await compose("cited");
+ assert.equal(readFileSync(pub("reports", REPORT, "video.mp4"), "utf8"), "mp4-bytes");
+ assert.equal(readFileSync(pub("reports", REPORT, "poster.jpg"), "utf8"), "jpg-bytes");
+ const view = readJson<{ video?: unknown }>(pub("reports", REPORT, "page.json"));
+ assert.deepEqual(view.video, { src: `/reports/${REPORT}/video.mp4`, poster: `/reports/${REPORT}/poster.jpg`, caption: "The cut" });
+ writeFileSync(path.join(dir, "video.mp4"), Buffer.alloc(25 * 1024 * 1024));
+ await assert.rejects(compose("cited"), (e: unknown) => {
+ assert.ok(e instanceof ComposeReportsError);
+ assert.deepEqual(e.problems.map((p) => p.kind), ["report-video"]);
+ assert.match(e.message, /over the publish limit/);
+ return true;
+ });
+ } finally {
+ writeJson(file, report());
+ rmSync(path.join(dir, "video.mp4"), { force: true });
+ rmSync(path.join(dir, "poster.jpg"), { force: true });
+ await compose("cited");
+ }
+});
+
test("a clip prepared for another span is stale: the reports changed since prepare", async () => {
const sidecar = path.join(reportMediaDir(paths, "cited"), `${CLIP}.json`);
const saved = readFileSync(sidecar, "utf8");
diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts
@@ -57,6 +57,7 @@
import { copyFile, mkdir, readdir, readFile, rm, stat, writeFile } from "node:fs/promises";
import path from "node:path";
+import { publishFileSizeProblem } from "../lib/builtExport";
import type { Paths } from "../lib/paths";
import { siteChannelSlugs, type Site } from "../lib/site";
import { isCitedSite } from "../lib/siteSchema";
@@ -167,6 +168,7 @@ export type ComposeReportsProblemKind =
| "quote-drift"
| "missing-post"
| "missing-still"
+ | "report-video"
| "missing-media"
| "stale-media";
@@ -481,6 +483,16 @@ export async function resolveSiteReports(opts: ResolveSiteReportsOptions): Promi
// record is read.
const refusedChannel = new Set<string>();
for (const report of reports) {
+ // The report's video and its poster: on disk, and small enough to publish.
+ if (report.video) {
+ const dir = siteReportDir(paths, site.siteId, report.id);
+ for (const rel of [report.video.src, report.video.poster]) {
+ if (!rel) continue;
+ const st = await stat(path.join(dir, rel)).catch(() => null);
+ const why = !st?.isFile() ? `the video file ${rel} does not exist` : publishFileSizeProblem(rel, st.size);
+ if (why) problems.push({ kind: "report-video", report: report.id, message: why });
+ }
+ }
const all = report.citations ?? {};
// Only what the report cites: a citation it defines but never cites is
// not in its view, and is neither checked nor published.
@@ -836,6 +848,13 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo
await writeOut(publicDir, reportViewPath(report.id), json(view));
await writeOut(publicDir, reportCitationsDownloadPath(report.id, "json"), json(citationSet(report, view)));
await writeOut(publicDir, reportCitationsDownloadPath(report.id, "csv"), citationsCsv(view));
+ if (report.video && view.video) {
+ const dir = siteReportDir(paths, site.siteId, report.id);
+ await copyOut(publicDir, path.join(dir, report.video.src), view.video.src);
+ if (report.video.poster && view.video.poster) {
+ await copyOut(publicDir, path.join(dir, report.video.poster), view.video.poster);
+ }
+ }
for (const c of orderedCitations(view)) {
if (c.kind !== "source" || !c.image) continue;
const rel = (report.citations![c.id] as { image: string }).image;
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **A report can carry a video.** `report.json` `video` (`{ "src": "video.mp4", "poster": "poster.jpg", "caption": "…" }`, files in the report's directory: an mp4, a png/jpg/webp poster, a one-line caption) plays at the head of the report's page, under its header. Composing the site refuses a report whose video or poster is missing, or whose video is over the 24 MiB publish limit. Needs a rebuild and deploy of the site.
- **Search reads every English track of a video, and the transcript switches tracks.** Where a video has another English caption track whose words differ from its transcript — the uploaded captions beside the original audio's, a regional or auto-translated track — a query matches it too: a hit only that track holds says so ("in uploaded captions") and opens the transcript on that track at that moment, and a word both say is found once, in the transcript. The transcript reader shows a small "Track:" switcher beside the mode buttons on such a video; the transcript stays the default, and the choice rides on the share link (`vt`). Downloads and Copy MD take the track on show. Needs an index build and a rebuild and deploy of each site.
- **A citation of a Wayback Machine copy links its original and the copy.** A cited record downloaded from a Wayback capture shows "Original (may be gone)", the original at the cited second where its platform takes one, and "Wayback Machine copy, <capture date>", the capture page, which plays. Its moment link is the capture: a capture URL never takes a time param.
- **Transcripts read the original-audio captions.** Where a video has both, its transcript is YouTube's `en-orig` track (the captions of what was said) rather than the served `en`, which can reword it; a track with no text falls through to the next. Videos whose only captions are in cue blocks (some livestream recordings) have their text.
diff --git a/export/app/components/reports/ReportArticle.tsx b/export/app/components/reports/ReportArticle.tsx
@@ -269,6 +269,23 @@ export default function ReportArticle({ view }: { view: ReportPageView }) {
{view.subtitle && <p className="text-lg text-muted-foreground">{view.subtitle}</p>}
</header>
+ {/* The report as a video, when it has one. */}
+ {view.video && (
+ <figure data-report-video="" className="flex flex-col gap-2">
+ <video
+ controls
+ playsInline
+ preload="metadata"
+ poster={view.video.poster}
+ src={view.video.src}
+ className="aspect-video w-full rounded-lg border border-border bg-black"
+ />
+ {view.video.caption && (
+ <figcaption className="text-sm text-muted-foreground">{view.video.caption}</figcaption>
+ )}
+ </figure>
+ )}
+
{/* The quick take: the tally and the summary, then where to go next. */}
<section data-report-tier="1" aria-label="In brief" className="flex flex-col gap-4">
<TierMarker depth={1} />