commit 8f30ffa23f2467b180d20546c2aa907718066f56
parent 9f1b0c07ef07d94eb3bc2784aa9decb2fed9ba65
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 02:28:16 -0400
common: the report document (lib/report/): schema, validation, reading order, cited-in index; REPORT.md + CITATIONS.md generated
report.json (archilyzer-report, version 1) is sections of claims over the
citation model's sources and citations. validate.ts reports every problem
with its JSON path: references that name nothing (listed citations, source
sentences, cite: links, the subject), duplicate anchors, a sweep with
verdicts, dates, verdict overrides, and every citation problem. uses.ts is
the one reading-order walk behind numbering and the cited-in index (moment
key to each report/section/claim citing it). The two key tables are
generated by file-schemas-docs.ts and pinned by a test.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
12 files changed, 1155 insertions(+), 6 deletions(-)
diff --git a/CITATIONS.md b/CITATIONS.md
@@ -0,0 +1,134 @@
+# Citations
+
+<!-- GENERATED by common/bin/file-schemas-docs.ts from the schemas and their *_FIELD_DOCS records — do not edit by hand. -->
+
+The one citation model, for every document that cites the corpus: a report (`report.json` — see [REPORT.md](REPORT.md)), a standalone citation set (`archilyzer-citations`, below), and whatever comes next. The schema is `common/lib/citations/schema.ts`; the value rules are `common/lib/citations/validate.ts`, which reports every problem with its JSON path instead of stopping at the first.
+
+A citation is a discriminated union on `kind`: `video` and `audio` (a span of a record's media), `post` (a post record), `source` (a sentence of a document under review) and `page` (a web page outside the corpus). Every kind carries the common fields. A container names its citations by id in a `citations` map, and its documents in a `sources` map; an id is letters, digits and `_ . : -`, starting with a letter or digit, at most 64, and no id names both a source and a citation. Unknown keys are refused, and so is an unknown kind: a new kind is a new version.
+
+**Citing inline.** In any markdown a document carries, `[label](cite:<id>)` cites the citation `<id>` (a link inside a code span or a fenced block is text, not a citation). Citations are numbered per document by first appearance in reading order — a citation cited again keeps its number — and each has the stable anchor `#c-<id>`.
+
+**Moments.** A `video` or `audio` citation opens the page `/m/<channel>/<id>/<start>-<end>/`, a `post` citation `/m/<channel>/<id>/`; `source` and `page` citations have no page of their own. `<start>` and `<end>` are written with two decimals, always (`12.50`, never `12.5`), so a moment has one URL; the pad is not part of it. `<channel>` and `<id>` are single safe path segments (letters, digits, `.`, `_`, `-`; a channel starts with a letter or digit, an id may start with `-`). Moment keys are `common/lib/citations/moments.ts`.
+
+**Spans.** `start` is at least 0, `end` is after it (by at least 0.01 s), each `pad` is at least 0, and the span with its pad is at most 120 s.
+
+**Paths.** A still (`image`) or a saved document (`saved`) is relative to the container's directory: `/`-separated, never absolute, with no empty, `.` or `..` segment.
+
+Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.
+
+## Every kind
+
+#### Common fields
+
+| Key | Required | Description |
+|---|---|---|
+| `quote` | yes | The cited words, VERBATIM — as the record says them (a span's cues, a post's text, the source's sentence, the page's text). Never a paraphrase: compose checks a span's quote against its cues and fails on drift. |
+| `speaker` | no | Who says the quote, when that is not the record's own channel (a guest, a co-host, a caller). |
+| `date` | no | When the quote was said or written: `YYYY`, `YYYY-MM`, `YYYY-MM-DD` or an ISO 8601 date-time. Absent = the record's own date. |
+| `label` | no | A short display name for the citation (one line), used where its number alone would be too little. |
+| `note` | no | An editorial note shown with the citation (plain text): context the quote needs. |
+| `verification` | no | COMPUTED, not authored: what checking this citation found, written by compose. A hand-typed block that could not have come from a check (a score outside 0–1, a score without its time) is a validation problem. |
+
+#### `verification`
+
+| Key | Required | Description |
+|---|---|---|
+| `quoteScore` | no | How closely `quote` matches what the record says at the cited place, from 0 (nothing alike) to 1 (verbatim). Written by the check, with `quoteCheckedAt`. |
+| `quoteCheckedAt` | no | When the quote was checked: an ISO 8601 date-time. Written with `quoteScore`. |
+| `voiceChecked` | no | True when the speaker's voice in the cited span was checked against `speaker`. Only a span (video, audio) has a voice to check. |
+| `method` | no | What did the checking, e.g. the cue-window comparison and its version. Free text, one line. |
+
+## The kinds
+
+#### `"video"`
+
+| Key | Required | Description |
+|---|---|---|
+| `kind` | yes | `"video"`. |
+| `channel` | yes | The record's channel slug (its directory under `transcripts/channels/`). A path segment of the moment page. |
+| `id` | yes | The record's video id (its directory under `data/`; may start with `-`). A path segment of the moment page. |
+| `start` | yes | Where the cited span starts, in seconds from the start of the record. |
+| `end` | yes | Where the cited span ends, in seconds; after `start`. |
+| `pad` | no | Context around the span in the evidence clip, in seconds. Absent = none. The span plus its pad is at most 120 s. |
+
+#### `"audio"`
+
+| Key | Required | Description |
+|---|---|---|
+| `kind` | yes | `"audio"`: a span of a record with no picture worth showing (a podcast). Rendered with a poster. |
+| `channel` | yes | The record's channel slug (its directory under `transcripts/channels/`). A path segment of the moment page. |
+| `id` | yes | The record's video id (its directory under `data/`; may start with `-`). A path segment of the moment page. |
+| `start` | yes | Where the cited span starts, in seconds from the start of the record. |
+| `end` | yes | Where the cited span ends, in seconds; after `start`. |
+| `pad` | no | Context around the span in the evidence clip, in seconds. Absent = none. The span plus its pad is at most 120 s. |
+
+#### `pad` (video, audio)
+
+| Key | Required | Description |
+|---|---|---|
+| `before` | no | Seconds of context before `start`; ≥ 0. Absent = 0. |
+| `after` | no | Seconds of context after `end`; ≥ 0. Absent = 0. |
+
+#### `"post"`
+
+| Key | Required | Description |
+|---|---|---|
+| `kind` | yes | `"post"`. |
+| `channel` | yes | The post's channel slug. A path segment of the moment page. |
+| `id` | yes | The post's id. A path segment of the moment page. |
+| `thread` | no | True to show the post with the thread it belongs to. Absent = the post alone. |
+
+#### `"source"`
+
+| Key | Required | Description |
+|---|---|---|
+| `kind` | yes | `"source"`: a sentence of a document under review. |
+| `source` | yes | The id of the document in the container's `sources`. Its archive links are shown with the citation. |
+| `image` | no | A still of the sentence as the document shows it: a path relative to the container's directory (e.g. `stills/a01.png`), never absolute, never leaving it. |
+
+#### `"page"`
+
+| Key | Required | Description |
+|---|---|---|
+| `kind` | yes | `"page"`: a web page outside the corpus. |
+| `url` | yes | The page's address: an http(s) URL. |
+| `title` | no | The page's title. |
+| `archiveUrl` | no | An archived copy of the page (an http(s) URL), shown beside the live link. |
+
+## Sources
+
+The documents a `source` citation quotes, in a container's `sources` map by id. A saved copy (`saved`) is the input stills are shot from and is never published.
+
+#### `sources.<id>`
+
+| Key | Required | Description |
+|---|---|---|
+| `kind` | yes | What the document is: `article`, `page`, `video`, `post`, `document`, `other`. |
+| `title` | yes | The document's title. |
+| `url` | no | Where the document lives: an http(s) URL. |
+| `publisher` | no | Who published it (the outlet, the site). |
+| `author` | no | Who wrote it. |
+| `date` | no | When it was published: `YYYY`, `YYYY-MM`, `YYYY-MM-DD` or an ISO 8601 date-time. |
+| `archives` | no | The document's archive links, in context — as the document had them. Shown with every citation of it. |
+| `saved` | no | A saved copy of the document, relative to the container's directory (e.g. `sources/s0/page.html`): the input the stills are shot from. NEVER published. |
+
+#### `sources.<id>.archives[]`
+
+| Key | Required | Description |
+|---|---|---|
+| `label` | yes | The link's text. |
+| `url` | yes | The archived copy: an http(s) URL. |
+| `context` | no | The words around the link in the document, so a reader sees what it was offered as. |
+
+## A citation set
+
+Citations outside a report — what a converter of a sweep, an answer or a video manifest emits — are one JSON document, format `"archilyzer-citations"`, version 1.
+
+#### The document
+
+| Key | Required | Description |
+|---|---|---|
+| `format` | yes | `"archilyzer-citations"`. |
+| `version` | yes | `1`. |
+| `sources` | no | The documents the `source` citations quote, by id. Absent = none. |
+| `citations` | yes | The citations, by id. An id is what a `[label](cite:<id>)` link names. |
diff --git a/REPORT.md b/REPORT.md
@@ -0,0 +1,62 @@
+# report.json keys
+
+<!-- GENERATED by common/bin/file-schemas-docs.ts from the schemas and their *_FIELD_DOCS records — do not edit by hand. -->
+
+One cited report, format `"archilyzer-report"`, version 1, persisted to `transcripts/sites/<siteId>/reports/<reportId>/report.json` beside its `stills/` and `sources/<sourceId>/`; a relative path in it is relative to that directory. A site's `reports` list in `site.json` is the published, ordered list — see [SITE.md](SITE.md); a report directory it does not name is a draft. The schema is `common/lib/report/schema.ts`; its citations and sources are the citation model's — see [CITATIONS.md](CITATIONS.md).
+
+A **fact-check** (`"kind": "factcheck"`) is sections (chapters) of claims, each with a verdict and its findings. A **sweep** (`"kind": "sweep"`) is sections with no verdicts, or bodies that cite inline. Markdown fields (`summary`, a section's `body`, a claim's `findings`) cite with `[label](cite:<id>)`.
+
+`common/lib/report/validate.ts` reports every problem with its JSON path: an unknown key, a reference that names nothing (a listed citation, a claim's source sentence, a `cite:` link, the subject), a section or claim id used twice (they share one namespace: the report page's anchors), a sweep's claim with a verdict, `updated` before `published`, and every citation problem CITATIONS.md lists. Whether a still exists and whether a quote matches its cues are checked when the site is composed.
+
+Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.
+
+## The document
+
+| Key | Required | Description |
+|---|---|---|
+| `format` | yes | `"archilyzer-report"`. |
+| `version` | yes | `1`. |
+| `id` | yes | The report's id: a lowercase slug (`[a-z0-9][a-z0-9-]*`, at most 64), its directory name under `reports/` and the last segment of its page, `/reports/<id>/`. Must match the directory. |
+| `kind` | yes | `"factcheck"` — sections of claims, each with a verdict — or `"sweep"` — sections with no verdicts, or bodies with inline citations. |
+| `title` | yes | The report's title. |
+| `subtitle` | no | A line under the title. |
+| `summary` | no | The report's summary, in markdown, shown before the sections. May cite inline: `[label](cite:<id>)`. |
+| `published` | no | When the report was published: `YYYY-MM-DD` or an ISO 8601 date-time with a zone. |
+| `updated` | no | When it was last changed, in the same form; not before `published`. |
+| `subject` | no | The document under review, when the report reviews one: `{ "source": "<id>" }`, an id in `sources`. |
+| `verdicts` | no | Overrides of the shared verdict vocabulary's labels and colours, by verdict (`CORROBORATED`, `PARTLY`, `CONTRADICTED`, `NOT_FOUND`, `UNTESTABLE`): `{ "label": "…", "color": "#rrggbb" }`, each key optional. Absent = the shared defaults. |
+| `sources` | no | The documents the report's `source` citations quote, by id — see [CITATIONS.md](CITATIONS.md). Absent = none. |
+| `citations` | no | The report's citations, by id — see [CITATIONS.md](CITATIONS.md). A citation is cited from markdown with `[label](cite:<id>)` and listed under the claims that rest on it. Absent = none. |
+| `sections` | yes | The report's sections, in order. |
+
+#### `sections[]`
+
+| Key | Required | Description |
+|---|---|---|
+| `id` | yes | The section's id (letters, digits, `_ . : -`; at most 64): its anchor on the report page. Unique among the report's section and claim ids. |
+| `title` | yes | The section's heading. |
+| `body` | no | Markdown under the heading. May cite inline. |
+| `claims` | no | The section's claims, in order. Absent = none. |
+
+#### `sections[].claims[]`
+
+| Key | Required | Description |
+|---|---|---|
+| `id` | yes | The claim's id (letters, digits, `_ . : -`; at most 64): its anchor on the report page. Unique among the report's section and claim ids. |
+| `text` | yes | The claim, as stated by the document under review (plain text). |
+| `verdict` | no | The ruling on the claim: `CORROBORATED`, `PARTLY`, `CONTRADICTED`, `NOT_FOUND`, `UNTESTABLE`. A fact-check's claim may leave it out (not yet ruled); a sweep's carries none. |
+| `sourceQuote` | no | The document's own sentence making the claim: `{ "citation": "<id>" }`, naming a `source` citation (its still is shown with the claim). |
+| `findings` | no | What the evidence shows, in markdown, citing inline: `[label](cite:<id>)`. |
+| `citations` | no | The citations the claim rests on, in the order they are listed under it. Each must exist; none twice. |
+
+## The verdicts
+
+The shared vocabulary (`common/lib/report/verdicts.mjs` — the one copy; report-to-video's fact-check stamps read it too), in the order a tally lists them. A report's `verdicts` overrides a label (one line, at most 24 characters) or a colour (`#rgb` or `#rrggbb`), each on its own.
+
+| Verdict | Default label | Default colour |
+|---|---|---|
+| `CORROBORATED` | Corroborated | `#3fbf7f` |
+| `PARTLY` | Partly true | `#e3b23c` |
+| `CONTRADICTED` | Contradicted | `#e5534b` |
+| `NOT_FOUND` | Not found | `#8b93a7` |
+| `UNTESTABLE` | Untestable | `#7d8fd6` |
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -296,7 +296,7 @@ export const COMMANDS: Command[] = [
},
{
path: ["docs", "files"],
- usage: "[--check] write SITE.md + CHANNEL.md from the file schemas",
+ usage: "[--check] write SITE.md + CHANNEL.md + REPORT.md + CITATIONS.md from the file schemas",
flags: { check: "boolean" },
run: async ({ flags }) =>
(await import("./file-schemas-docs")).main({ check: flags.check === true }),
diff --git a/common/bin/file-schemas-docs.ts b/common/bin/file-schemas-docs.ts
@@ -1,16 +1,17 @@
#!/usr/bin/env tsx
-// WRITE SITE.md AND CHANNEL.md FROM THE FILE SCHEMAS.
+// WRITE SITE.md, CHANNEL.md, REPORT.md AND CITATIONS.md FROM THE FILE SCHEMAS.
//
// Usage (from the repo root):
// pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts
// pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts --check
//
-// `--check` writes nothing and exits 1 if either committed file differs from
+// `--check` writes nothing and exits 1 if any committed file differs from
// what the schemas generate (the same claim common/lib/fileSchemaDocs.test.ts
-// makes). The sibling of settings-example.ts (SETTINGS.md).
+// and lib/report/docs.test.ts make). The sibling of settings-example.ts
+// (SETTINGS.md).
//
-// Reads no site.json and no config.json: both outputs are functions of the
-// schemas' docs records alone.
+// Reads no site.json, config.json or report.json: every output is a function of
+// the schemas' docs records alone.
import { readFile, writeFile } from "node:fs/promises";
import path from "node:path";
@@ -19,6 +20,8 @@ import {
renderChannelMarkdown,
renderSiteMarkdown,
} from "../lib/fileSchemaDocs";
+import { renderCitationsMarkdown } from "../lib/citations/docs";
+import { renderReportMarkdown } from "../lib/report/docs";
import { parseFlags } from "./_parseFlags";
import { runIfEntryPoint } from "./_cli";
@@ -27,6 +30,8 @@ const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", ".
const FILE_SCHEMA_OUTPUTS: ReadonlyArray<[string, () => string]> = [
["SITE.md", renderSiteMarkdown],
["CHANNEL.md", renderChannelMarkdown],
+ ["REPORT.md", renderReportMarkdown],
+ ["CITATIONS.md", renderCitationsMarkdown],
];
// `check` writes nothing and returns 1 when a committed file is stale.
diff --git a/common/lib/citations/docs.ts b/common/lib/citations/docs.ts
@@ -0,0 +1,167 @@
+// CITATIONS.md — the citation model's key tables, GENERATED from the schemas
+// (./schema.ts): the keys and their order from each member's zod shape,
+// required-or-not from the same shape, the prose from the *_FIELD_DOCS records.
+//
+// A pure renderer, the sibling of lib/fileSchemaDocs.ts (SITE.md, CHANNEL.md):
+// common/bin/file-schemas-docs.ts writes the file and citations/docs.test.ts
+// asserts the committed bytes are what this returns, so it cannot be edited by
+// hand.
+
+import type { z } from "zod";
+import { cell } from "../settingsDocs";
+import {
+ AUDIO_CITATION_FIELD_DOCS,
+ CITATION_COMMON_FIELD_DOCS,
+ CITATION_PAD_FIELD_DOCS,
+ CITATION_SET_FIELD_DOCS,
+ CITATION_SET_FORMAT,
+ CITATION_VERIFICATION_FIELD_DOCS,
+ CITATIONS_VERSION,
+ MAX_CITATION_SPAN_SECONDS,
+ PAGE_CITATION_FIELD_DOCS,
+ POST_CITATION_FIELD_DOCS,
+ SOURCE_ARCHIVE_FIELD_DOCS,
+ SOURCE_CITATION_FIELD_DOCS,
+ SOURCE_FIELD_DOCS,
+ VIDEO_CITATION_FIELD_DOCS,
+ audioCitationSchema,
+ citationSetSchema,
+ pageCitationSchema,
+ postCitationSchema,
+ sourceArchiveSchema,
+ sourceCitationSchema,
+ sourceSchema,
+ verificationSchema,
+ videoCitationSchema,
+} from "./schema";
+
+export const GENERATED_SCHEMA_DOC =
+ "<!-- GENERATED by common/bin/file-schemas-docs.ts from the schemas and their *_FIELD_DOCS records — do not edit by hand. -->";
+
+export const REGENERATE_SCHEMA_DOC =
+ "Regenerate this file with " +
+ "`pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.";
+
+type Shape = Record<string, z.ZodType>;
+
+// Whether a key of a zod object shape may be left out.
+export function isOptionalKey(shape: Shape, key: string): boolean {
+ return shape[key].safeParse(undefined).success;
+}
+
+// One key table, `Key | Required | Description`, in `docs` order; only the
+// keys `docs` names (a member's own keys, when the common ones are shared).
+export function renderKeyTable(
+ out: string[],
+ heading: string,
+ shape: Shape,
+ docs: Readonly<Record<string, string>>,
+): void {
+ out.push(heading);
+ out.push("");
+ out.push("| Key | Required | Description |");
+ out.push("|---|---|---|");
+ for (const [key, text] of Object.entries(docs)) {
+ if (!(key in shape)) throw new Error(`${heading}: ${key} is documented but not in the schema`);
+ out.push(`| \`${key}\` | ${isOptionalKey(shape, key) ? "no" : "yes"} | ${cell(text)} |`);
+ }
+ out.push("");
+}
+
+export function renderCitationsMarkdown(): string {
+ const out: string[] = [];
+ out.push("# Citations");
+ out.push("");
+ out.push(GENERATED_SCHEMA_DOC);
+ out.push("");
+ out.push(
+ "The one citation model, for every document that cites the corpus: a report " +
+ "(`report.json` — see [REPORT.md](REPORT.md)), a standalone citation set " +
+ `(\`${CITATION_SET_FORMAT}\`, below), and whatever comes next. The schema is ` +
+ "`common/lib/citations/schema.ts`; the value rules are " +
+ "`common/lib/citations/validate.ts`, which reports every problem with its JSON " +
+ "path instead of stopping at the first.",
+ );
+ out.push("");
+ out.push(
+ "A citation is a discriminated union on `kind`: `video` and `audio` (a span of a " +
+ "record's media), `post` (a post record), `source` (a sentence of a document under " +
+ "review) and `page` (a web page outside the corpus). Every kind carries the common " +
+ "fields. A container names its citations by id in a `citations` map, and its " +
+ "documents in a `sources` map; an id is letters, digits and `_ . : -`, starting " +
+ "with a letter or digit, at most 64, and no id names both a source and a citation. " +
+ "Unknown keys are refused, and so is an unknown kind: a new kind is a new version.",
+ );
+ out.push("");
+ out.push(
+ "**Citing inline.** In any markdown a document carries, `[label](cite:<id>)` cites " +
+ "the citation `<id>` (a link inside a code span or a fenced block is text, not a " +
+ "citation). Citations are numbered per document by first appearance in reading " +
+ "order — a citation cited again keeps its number — and each has the stable anchor " +
+ "`#c-<id>`.",
+ );
+ out.push("");
+ out.push(
+ "**Moments.** A `video` or `audio` citation opens the page " +
+ "`/m/<channel>/<id>/<start>-<end>/`, a `post` citation `/m/<channel>/<id>/`; " +
+ "`source` and `page` citations have no page of their own. `<start>` and `<end>` " +
+ "are written with two decimals, always (`12.50`, never `12.5`), so a moment has " +
+ "one URL; the pad is not part of it. `<channel>` and `<id>` are single safe path " +
+ "segments (letters, digits, `.`, `_`, `-`; a channel starts with a letter or " +
+ "digit, an id may start with `-`). Moment keys are `common/lib/citations/moments.ts`.",
+ );
+ out.push("");
+ out.push(
+ `**Spans.** \`start\` is at least 0, \`end\` is after it (by at least 0.01 s), each \`pad\` ` +
+ `is at least 0, and the span with its pad is at most ${MAX_CITATION_SPAN_SECONDS} s.`,
+ );
+ out.push("");
+ out.push(
+ "**Paths.** A still (`image`) or a saved document (`saved`) is relative to the " +
+ "container's directory: `/`-separated, never absolute, with no empty, `.` or `..` " +
+ "segment.",
+ );
+ out.push("");
+ out.push(REGENERATE_SCHEMA_DOC);
+ out.push("");
+
+ out.push("## Every kind");
+ out.push("");
+ renderKeyTable(out, "#### Common fields", videoCitationSchema.shape, CITATION_COMMON_FIELD_DOCS);
+ renderKeyTable(out, "#### `verification`", verificationSchema.shape, CITATION_VERIFICATION_FIELD_DOCS);
+
+ out.push("## The kinds");
+ out.push("");
+ renderKeyTable(out, '#### `"video"`', videoCitationSchema.shape, VIDEO_CITATION_FIELD_DOCS);
+ renderKeyTable(out, '#### `"audio"`', audioCitationSchema.shape, AUDIO_CITATION_FIELD_DOCS);
+ renderKeyTable(
+ out,
+ "#### `pad` (video, audio)",
+ videoCitationSchema.shape.pad.unwrap().shape,
+ CITATION_PAD_FIELD_DOCS,
+ );
+ renderKeyTable(out, '#### `"post"`', postCitationSchema.shape, POST_CITATION_FIELD_DOCS);
+ renderKeyTable(out, '#### `"source"`', sourceCitationSchema.shape, SOURCE_CITATION_FIELD_DOCS);
+ renderKeyTable(out, '#### `"page"`', pageCitationSchema.shape, PAGE_CITATION_FIELD_DOCS);
+
+ out.push("## Sources");
+ out.push("");
+ out.push(
+ "The documents a `source` citation quotes, in a container's `sources` map by id. " +
+ "A saved copy (`saved`) is the input stills are shot from and is never published.",
+ );
+ out.push("");
+ renderKeyTable(out, "#### `sources.<id>`", sourceSchema.shape, SOURCE_FIELD_DOCS);
+ renderKeyTable(out, "#### `sources.<id>.archives[]`", sourceArchiveSchema.shape, SOURCE_ARCHIVE_FIELD_DOCS);
+
+ out.push("## A citation set");
+ out.push("");
+ out.push(
+ `Citations outside a report — what a converter of a sweep, an answer or a video ` +
+ `manifest emits — are one JSON document, format \`"${CITATION_SET_FORMAT}"\`, ` +
+ `version ${CITATIONS_VERSION}.`,
+ );
+ out.push("");
+ renderKeyTable(out, "#### The document", citationSetSchema.shape, CITATION_SET_FIELD_DOCS);
+ return out.join("\n");
+}
diff --git a/common/lib/report/citedIn.ts b/common/lib/report/citedIn.ts
@@ -0,0 +1,50 @@
+// "CITED IN" — the back-link index a moment page reads: for every moment the
+// given reports cite, each place that cites it.
+//
+// moment key → [{ reportId, sectionId, claimId, citationId }]
+//
+// Built at compose from the site's published reports, in the order given, each
+// report in reading order (./uses.ts). A place is listed once however many
+// times it cites the moment there (a claim that cites a span in its findings
+// and lists it under itself is one back-link); two citations of the same span
+// in one claim are two. Citations without a moment (`source`, `page`), and
+// references to citations the report does not define, are not indexed.
+//
+// Pure: the input is parsed reports, the output plain JSON.
+
+import { momentKeyOf } from "../citations/moments";
+import type { Report } from "./schema";
+import { reportCitationUses } from "./uses";
+
+export type CitedIn = {
+ reportId: string;
+ // null when cited in the report's summary.
+ sectionId: string | null;
+ // null when cited in the summary or a section's body.
+ claimId: string | null;
+ citationId: string;
+};
+
+export function buildCitedIn(reports: readonly Report[]): Record<string, CitedIn[]> {
+ const out: Record<string, CitedIn[]> = {};
+ const seen = new Set<string>();
+ for (const report of reports) {
+ const citations = report.citations ?? {};
+ for (const use of reportCitationUses(report)) {
+ if (!Object.hasOwn(citations, use.citationId)) continue;
+ const key = momentKeyOf(citations[use.citationId]);
+ if (!key) continue;
+ const entry: CitedIn = {
+ reportId: report.id,
+ sectionId: use.sectionId,
+ claimId: use.claimId,
+ citationId: use.citationId,
+ };
+ const dedup = JSON.stringify([key, entry.reportId, entry.sectionId, entry.claimId, entry.citationId]);
+ if (seen.has(dedup)) continue;
+ seen.add(dedup);
+ (out[key] ??= []).push(entry);
+ }
+ }
+ return out;
+}
diff --git a/common/lib/report/docs.test.ts b/common/lib/report/docs.test.ts
@@ -0,0 +1,46 @@
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { renderReportMarkdown } from "./docs";
+import { renderCitationsMarkdown } from "../citations/docs";
+import { claimSchema, reportSchema, sectionSchema } from "./schema";
+import { CITATION_KINDS } from "../citations/schema";
+import { VERDICTS } from "./verdicts";
+
+// REPORT.md and CITATIONS.md are GENERATED from the schemas
+// (common/bin/file-schemas-docs.ts). A hand edit to either, or a schema change
+// without a regenerate, fails here.
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..", "..");
+
+for (const [name, render] of [
+ ["REPORT.md", renderReportMarkdown],
+ ["CITATIONS.md", renderCitationsMarkdown],
+] as const) {
+ test(`${name} is what the schema generates`, () => {
+ const committed = readFileSync(path.join(REPO, name), "utf8");
+ assert.equal(
+ committed,
+ render(),
+ `${name} is stale: run pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`,
+ );
+ });
+}
+
+test("REPORT.md has a row for every key of the document, a section and a claim, and every verdict", () => {
+ const md = renderReportMarkdown();
+ for (const shape of [reportSchema.shape, sectionSchema.shape, claimSchema.shape]) {
+ for (const key of Object.keys(shape)) assert.ok(md.includes(`| \`${key}\` |`), key);
+ }
+ for (const v of VERDICTS) assert.ok(md.includes(`| \`${v}\` |`), v);
+});
+
+test("CITATIONS.md documents every key of every kind", () => {
+ const md = renderCitationsMarkdown();
+ for (const member of CITATION_KINDS) {
+ assert.ok(md.includes(`#### \`"${member.shape.kind.value}"\``), member.shape.kind.value);
+ for (const key of Object.keys(member.shape)) assert.ok(md.includes(`| \`${key}\` |`), key);
+ }
+});
diff --git a/common/lib/report/docs.ts b/common/lib/report/docs.ts
@@ -0,0 +1,78 @@
+// REPORT.md — `report.json`'s key tables, GENERATED from the schema
+// (./schema.ts) and the verdict vocabulary (./verdicts.mjs). The citation keys
+// are CITATIONS.md's (lib/citations/docs.ts), linked rather than repeated.
+//
+// A pure renderer: common/bin/file-schemas-docs.ts writes the file and
+// report/docs.test.ts asserts the committed bytes are what this returns.
+
+import { cell } from "../settingsDocs";
+import { GENERATED_SCHEMA_DOC, REGENERATE_SCHEMA_DOC, renderKeyTable } from "../citations/docs";
+import {
+ CLAIM_FIELD_DOCS,
+ REPORT_FIELD_DOCS,
+ REPORT_FORMAT,
+ REPORT_VERSION,
+ SECTION_FIELD_DOCS,
+ claimSchema,
+ reportSchema,
+ sectionSchema,
+} from "./schema";
+import { VERDICT_DEFAULTS, VERDICT_LABEL_MAX, VERDICTS } from "./verdicts";
+
+export function renderReportMarkdown(): string {
+ const out: string[] = [];
+ out.push("# report.json keys");
+ out.push("");
+ out.push(GENERATED_SCHEMA_DOC);
+ out.push("");
+ out.push(
+ `One cited report, format \`"${REPORT_FORMAT}"\`, version ${REPORT_VERSION}, ` +
+ "persisted to `transcripts/sites/<siteId>/reports/<reportId>/report.json` beside " +
+ "its `stills/` and `sources/<sourceId>/`; a relative path in it is relative to " +
+ "that directory. A site's `reports` list in `site.json` is the published, " +
+ "ordered list — see [SITE.md](SITE.md); a report directory it does not name is a " +
+ "draft. The schema is `common/lib/report/schema.ts`; its citations and sources " +
+ "are the citation model's — see [CITATIONS.md](CITATIONS.md).",
+ );
+ out.push("");
+ out.push(
+ 'A **fact-check** (`"kind": "factcheck"`) is sections (chapters) of claims, each ' +
+ 'with a verdict and its findings. A **sweep** (`"kind": "sweep"`) is sections ' +
+ "with no verdicts, or bodies that cite inline. Markdown fields (`summary`, a " +
+ "section's `body`, a claim's `findings`) cite with `[label](cite:<id>)`.",
+ );
+ out.push("");
+ out.push(
+ "`common/lib/report/validate.ts` reports every problem with its JSON path: an " +
+ "unknown key, a reference that names nothing (a listed citation, a claim's " +
+ "source sentence, a `cite:` link, the subject), a section or claim id used " +
+ "twice (they share one namespace: the report page's anchors), a sweep's claim " +
+ "with a verdict, `updated` before `published`, and every citation problem " +
+ "CITATIONS.md lists. Whether a still exists and whether a quote matches its cues " +
+ "are checked when the site is composed.",
+ );
+ out.push("");
+ out.push(REGENERATE_SCHEMA_DOC);
+ out.push("");
+ renderKeyTable(out, "## The document", reportSchema.shape, REPORT_FIELD_DOCS);
+ renderKeyTable(out, "#### `sections[]`", sectionSchema.shape, SECTION_FIELD_DOCS);
+ renderKeyTable(out, "#### `sections[].claims[]`", claimSchema.shape, CLAIM_FIELD_DOCS);
+
+ out.push("## The verdicts");
+ out.push("");
+ out.push(
+ "The shared vocabulary (`common/lib/report/verdicts.mjs` — the one copy; " +
+ "report-to-video's fact-check stamps read it too), in the order a tally lists " +
+ `them. A report's \`verdicts\` overrides a label (one line, at most ${VERDICT_LABEL_MAX} ` +
+ "characters) or a colour (`#rgb` or `#rrggbb`), each on its own.",
+ );
+ out.push("");
+ out.push("| Verdict | Default label | Default colour |");
+ out.push("|---|---|---|");
+ for (const v of VERDICTS) {
+ const d = VERDICT_DEFAULTS[v];
+ out.push(`| \`${v}\` | ${cell(d.label)} | \`${d.color}\` |`);
+ }
+ out.push("");
+ return out.join("\n");
+}
diff --git a/common/lib/report/report.test.ts b/common/lib/report/report.test.ts
@@ -0,0 +1,241 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { parseReport, validateReport } from "./validate";
+import { reportCitationNumbers, reportCitationUses } from "./uses";
+import { buildCitedIn } from "./citedIn";
+import { VERDICTS, VERDICT_DEFAULTS, isVerdict, resolveVerdicts } from "./verdicts";
+import type { Report } from "./schema";
+import type { Problem } from "../citations/validate";
+
+const paths = (ps: Problem[]) => ps.map((p) => p.path);
+
+function fixture(): Record<string, unknown> & Report {
+ return {
+ format: "archilyzer-report",
+ version: 1,
+ id: "demo-report",
+ kind: "factcheck",
+ title: "A demo fact-check",
+ subtitle: "What the article says, against the record",
+ summary: "The article gets one thing [right](cite:c02) and one [wrong](cite:c01).",
+ published: "2026-10-01",
+ updated: "2026-10-04T09:30:00Z",
+ subject: { source: "s0" },
+ verdicts: { PARTLY: { label: "Half true" } },
+ sources: {
+ s0: {
+ kind: "article",
+ title: "A demo article",
+ url: "https://example.test/article",
+ archives: [{ label: "archived", url: "https://archive.example.test/a" }],
+ saved: "sources/s0/page.html",
+ },
+ },
+ citations: {
+ c01: { kind: "video", channel: "demo-channel", id: "-abc123", start: 10, end: 20, quote: "q1", pad: { before: 5, after: 5 } },
+ c02: { kind: "video", channel: "demo-channel", id: "def456", start: 30, end: 42.5, quote: "q2" },
+ p01: { kind: "post", channel: "demo-channel", id: "1234567890", quote: "q3" },
+ a01: { kind: "source", source: "s0", quote: "the claim", image: "stills/a01.png" },
+ a02: { kind: "source", source: "s0", quote: "another claim" },
+ w01: { kind: "page", url: "https://example.test/page", quote: "q4" },
+ },
+ sections: [
+ {
+ id: "ch1",
+ title: "Chapter one",
+ body: "Background, see [this](cite:w01).",
+ claims: [
+ {
+ id: "k1",
+ text: "The first claim.",
+ verdict: "CONTRADICTED",
+ sourceQuote: { citation: "a01" },
+ findings: "The record says otherwise: [at 0:10](cite:c01), and [the post](cite:p01).",
+ citations: ["c01", "p01"],
+ },
+ {
+ id: "k2",
+ text: "The second claim.",
+ verdict: "PARTLY",
+ sourceQuote: { citation: "a02" },
+ findings: "Partly: [here](cite:c02) and [here again](cite:c01).",
+ citations: ["c02"],
+ },
+ ],
+ },
+ { id: "ch2", title: "Chapter two", claims: [{ id: "k3", text: "Not yet ruled." }] },
+ ],
+ };
+}
+
+test("a sound fact-check parses with no problems", () => {
+ const r = parseReport(fixture(), { id: "demo-report" });
+ assert.ok(r.ok);
+ assert.deepEqual(r.problems, []);
+});
+
+test("schema rejects: wrong format/version, unknown keys, an unknown verdict, a missing title", () => {
+ const f = fixture() as Record<string, unknown>;
+ const bad = { ...f, format: "other", version: 2, extra: true, title: undefined, verdicts: { MAYBE: {} } };
+ const r = parseReport(bad);
+ assert.equal(r.ok, false);
+ const ps = paths(r.problems);
+ for (const want of ["format", "version", "title", "verdicts.MAYBE"]) assert.ok(ps.includes(want), `${want} in ${ps}`);
+ assert.ok(ps.includes(""), `the unknown key at the root in ${ps}`);
+ const claim = structuredClone(fixture());
+ (claim.sections[0].claims![0] as Record<string, unknown>).verdict = "MOSTLY";
+ assert.deepEqual(paths(validateReport(claim)), ["sections[0].claims[0].verdict"]);
+});
+
+test("every reference must resolve: listed citations, source sentences, cite links, the subject", () => {
+ const f = fixture();
+ f.subject = { source: "s9" };
+ const k1 = f.sections[0].claims![0];
+ k1.citations = ["c01", "c99", "c01"];
+ k1.sourceQuote = { citation: "c01" }; // not a source citation
+ k1.findings = "See [gone](cite:c98) and [blank](cite:).";
+ f.summary = "Also [missing](cite:zz).";
+ const ps = validateReport(f);
+ assert.deepEqual(paths(ps).sort(), [
+ "sections[0].claims[0].citations[1]",
+ "sections[0].claims[0].citations[2]",
+ "sections[0].claims[0].findings",
+ "sections[0].claims[0].findings",
+ "sections[0].claims[0].sourceQuote.citation",
+ "subject.source",
+ "summary",
+ ]);
+ assert.ok(ps.some((p) => /\[gone\]\(cite:c98\) names no citation/.test(p.message)));
+ assert.ok(ps.some((p) => /lists "c01" twice/.test(p.message)));
+ assert.ok(ps.some((p) => /names a video citation/.test(p.message)));
+});
+
+test("citation problems surface under the report's paths", () => {
+ const f = fixture();
+ f.citations!.c01 = { ...(f.citations!.c01 as object), end: 200 } as never;
+ f.citations!.a01 = { ...(f.citations!.a01 as object), image: "../../etc/passwd" } as never;
+ assert.deepEqual(paths(validateReport(f)).sort(), ["citations.a01.image", "citations.c01"]);
+});
+
+test("ids: report id is a slug and matches its directory; section and claim ids unique together", () => {
+ const f = fixture();
+ f.sections[1].id = "k1";
+ f.sections[1].claims![0].id = "bad id";
+ const ps = validateReport(f, { id: "other-report" });
+ assert.deepEqual(paths(ps).sort(), ["id", "sections[1].claims[0].id", "sections[1].id"]);
+ assert.ok(ps.some((p) => /already the id of sections\[0\]\.claims\[0\]/.test(p.message)));
+ assert.deepEqual(paths(validateReport({ ...fixture(), id: "Demo_Report" })), ["id"]);
+});
+
+test("dates, verdict overrides, a sweep with a verdict", () => {
+ const f = fixture();
+ f.published = "2026-10-05";
+ f.updated = "2026-10-04";
+ f.verdicts = { PARTLY: { label: "x".repeat(25), color: "orange" } };
+ const ps = validateReport(f);
+ assert.deepEqual(paths(ps).sort(), ["updated", "verdicts.PARTLY.color", "verdicts.PARTLY.label"]);
+ assert.deepEqual(paths(validateReport({ ...fixture(), published: "1 Oct 2026" })), ["published"]);
+ assert.deepEqual(paths(validateReport({ ...fixture(), kind: "sweep" })), [
+ "sections[0].claims[0].verdict",
+ "sections[0].claims[1].verdict",
+ ]);
+});
+
+test("a sweep with inline citations in its bodies and no claims is a report", () => {
+ const sweep = {
+ format: "archilyzer-report",
+ version: 1,
+ id: "demo-sweep",
+ kind: "sweep",
+ title: "A demo sweep",
+ citations: { c01: { kind: "video", channel: "demo-channel", id: "abc123", start: 1, end: 2, quote: "q" } },
+ sections: [{ id: "s1", title: "Found", body: "It came up [once](cite:c01)." }],
+ };
+ assert.deepEqual(validateReport(sweep), []);
+});
+
+test("uses: reading order — summary, body, then each claim's source sentence, findings, list", () => {
+ const r = parseReport(fixture());
+ assert.ok(r.ok);
+ const uses = reportCitationUses(r.value).map((u) => `${u.field}:${u.citationId}`);
+ assert.deepEqual(uses, [
+ "summary:c02",
+ "summary:c01",
+ "body:w01",
+ "sourceQuote:a01",
+ "findings:c01",
+ "findings:p01",
+ "citations:c01",
+ "citations:p01",
+ "sourceQuote:a02",
+ "findings:c02",
+ "findings:c01",
+ "citations:c02",
+ ]);
+ assert.deepEqual([...reportCitationNumbers(r.value)], [
+ ["c02", 1],
+ ["c01", 2],
+ ["w01", 3],
+ ["a01", 4],
+ ["p01", 5],
+ ["a02", 6],
+ ]);
+});
+
+test("numbers skip dangling references", () => {
+ const f = fixture();
+ f.summary = "[x](cite:nope) then [y](cite:c01)";
+ const r = parseReport(f);
+ assert.ok(r.ok);
+ assert.equal(reportCitationNumbers(r.value).get("c01"), 1);
+ assert.equal(reportCitationNumbers(r.value).has("nope"), false);
+});
+
+test("cited-in: moment key → each place that cites it, across reports, deduplicated", () => {
+ const a = parseReport(fixture());
+ const second = fixture();
+ second.id = "second-report";
+ second.summary = undefined;
+ second.sections = [{ id: "only", title: "Only", body: "Again [here](cite:c01)." }];
+ const b = parseReport(second);
+ assert.ok(a.ok && b.ok);
+ const index = buildCitedIn([a.value, b.value]);
+ assert.deepEqual(Object.keys(index), [
+ "demo-channel/def456/30.00-42.50",
+ "demo-channel/-abc123/10.00-20.00",
+ "demo-channel/1234567890",
+ ]);
+ assert.deepEqual(index["demo-channel/-abc123/10.00-20.00"], [
+ { reportId: "demo-report", sectionId: null, claimId: null, citationId: "c01" },
+ { reportId: "demo-report", sectionId: "ch1", claimId: "k1", citationId: "c01" },
+ { reportId: "demo-report", sectionId: "ch1", claimId: "k2", citationId: "c01" },
+ { reportId: "second-report", sectionId: "only", claimId: null, citationId: "c01" },
+ ]);
+ assert.deepEqual(index["demo-channel/1234567890"], [
+ { reportId: "demo-report", sectionId: "ch1", claimId: "k1", citationId: "p01" },
+ ]);
+});
+
+test("cited-in: two citations of one span in one place are two back-links; source and page citations have none", () => {
+ const f = fixture();
+ f.citations!.c03 = { ...(f.citations!.c01 as object), quote: "the same span, quoted again" } as never;
+ f.sections[0].claims![0].citations = ["c01", "c03"];
+ const r = parseReport(f);
+ assert.ok(r.ok);
+ const entries = buildCitedIn([r.value])["demo-channel/-abc123/10.00-20.00"];
+ assert.deepEqual(
+ entries.filter((e) => e.claimId === "k1").map((e) => e.citationId),
+ ["c01", "c03"],
+ );
+ const keys = Object.keys(buildCitedIn([r.value]));
+ assert.ok(!keys.some((k) => k.includes("example.test")));
+});
+
+test("the verdict vocabulary: five verdicts, a default for each, overrides laid over one key at a time", () => {
+ assert.deepEqual([...VERDICTS], ["CORROBORATED", "PARTLY", "CONTRADICTED", "NOT_FOUND", "UNTESTABLE"]);
+ assert.deepEqual(Object.keys(VERDICT_DEFAULTS), [...VERDICTS]);
+ assert.ok(isVerdict("PARTLY") && !isVerdict("partly"));
+ const v = resolveVerdicts({ PARTLY: { label: "Half true" }, MAYBE: { label: "x" } });
+ assert.deepEqual(v.PARTLY, { label: "Half true", color: VERDICT_DEFAULTS.PARTLY.color });
+ assert.deepEqual(Object.keys(v), [...VERDICTS]);
+});
diff --git a/common/lib/report/schema.ts b/common/lib/report/schema.ts
@@ -0,0 +1,123 @@
+// THE REPORT DOCUMENT — one definition of `report.json` (format
+// `archilyzer-report`), used by the validator, the compose stage, the export
+// site's report pages and REPORT.md.
+//
+// A report is a cited document: a summary, then sections, each with an
+// optional markdown body and claims. A FACT-CHECK (`kind: "factcheck"`) is
+// sections (chapters) of claims, each with a verdict from the shared
+// vocabulary (./verdicts.mjs) and its findings; a SWEEP (`kind: "sweep"`) is
+// sections with no verdicts, or bodies with inline citations.
+//
+// ITS CITATIONS ARE THE CITATION MODEL'S (lib/citations/): the `sources` and
+// `citations` maps are that model's schemas, cited inline with
+// `[label](cite:<id>)` and listed per claim. Nothing about a citation is
+// defined here.
+//
+// Lives at `transcripts/sites/<siteId>/reports/<reportId>/report.json`, beside
+// its `stills/` and `sources/<sourceId>/`; a relative path in it (a still, a
+// saved source) is relative to that directory.
+//
+// As in lib/citations/schema.ts, zod checks SHAPE only; every rule about
+// values — ids, references, `cite:` links, the verdict overrides, dates — is
+// ./validate.ts's, which reports every problem with its JSON path.
+//
+// SERVER-ONLY: zod. A client importer takes the types with `import type`.
+
+import { z } from "zod";
+import type { FieldDocs } from "../fieldDocs";
+import { citationSchema, sourceSchema } from "../citations/schema";
+import { VERDICTS, type Verdict } from "./verdicts";
+
+export const REPORT_FORMAT = "archilyzer-report";
+export const REPORT_VERSION = 1;
+
+export const REPORT_KINDS = ["factcheck", "sweep"] as const;
+export type ReportKind = (typeof REPORT_KINDS)[number];
+
+const text = z.string();
+const verdict = z.enum(VERDICTS as unknown as [Verdict, ...Verdict[]]);
+
+export const claimSchema = z.strictObject({
+ id: text,
+ text,
+ verdict: verdict.optional(),
+ sourceQuote: z.strictObject({ citation: text }).optional(),
+ findings: text.optional(),
+ citations: z.array(text).optional(),
+});
+
+export const sectionSchema = z.strictObject({
+ id: text,
+ title: text,
+ body: text.optional(),
+ claims: z.array(claimSchema).optional(),
+});
+
+const verdictOverride = z.strictObject({ label: text.optional(), color: text.optional() });
+
+export const reportSchema = z.strictObject({
+ format: z.literal(REPORT_FORMAT),
+ version: z.literal(REPORT_VERSION),
+ id: text,
+ kind: z.enum(REPORT_KINDS),
+ title: text,
+ subtitle: text.optional(),
+ summary: text.optional(),
+ published: text.optional(),
+ updated: text.optional(),
+ subject: z.strictObject({ source: text }).optional(),
+ verdicts: z.partialRecord(verdict, verdictOverride).optional(),
+ sources: z.record(text, sourceSchema).optional(),
+ citations: z.record(text, citationSchema).optional(),
+ sections: z.array(sectionSchema),
+});
+
+export type Claim = z.infer<typeof claimSchema>;
+export type Section = z.infer<typeof sectionSchema>;
+export type Report = z.infer<typeof reportSchema>;
+export type ReportSubject = NonNullable<Report["subject"]>;
+export type ClaimSourceQuote = NonNullable<Claim["sourceQuote"]>;
+
+// A report id is its directory name under `reports/` and a path segment of its
+// page (`/reports/<id>/`): a lowercase slug, like a site id.
+export const REPORT_ID_RE = /^[a-z0-9][a-z0-9-]{0,63}$/;
+
+export function isReportId(v: unknown): v is string {
+ return typeof v === "string" && REPORT_ID_RE.test(v);
+}
+
+export const REPORT_FIELD_DOCS: FieldDocs<Report> = {
+ format: `\`"${REPORT_FORMAT}"\`.`,
+ version: `\`${REPORT_VERSION}\`.`,
+ id: "The report's id: a lowercase slug (`[a-z0-9][a-z0-9-]*`, at most 64), its directory name under `reports/` and the last segment of its page, `/reports/<id>/`. Must match the directory.",
+ kind: '`"factcheck"` — sections of claims, each with a verdict — or `"sweep"` — sections with no verdicts, or bodies with inline citations.',
+ title: "The report's title.",
+ subtitle: "A line under the title.",
+ summary: "The report's summary, in markdown, shown before the sections. May cite inline: `[label](cite:<id>)`.",
+ published: "When the report was published: `YYYY-MM-DD` or an ISO 8601 date-time with a zone.",
+ updated: "When it was last changed, in the same form; not before `published`.",
+ subject:
+ "The document under review, when the report reviews one: `{ \"source\": \"<id>\" }`, an id in `sources`.",
+ verdicts: `Overrides of the shared verdict vocabulary's labels and colours, by verdict (${VERDICTS.map((v) => `\`${v}\``).join(", ")}): \`{ "label": "…", "color": "#rrggbb" }\`, each key optional. Absent = the shared defaults.`,
+ sources: "The documents the report's `source` citations quote, by id — see [CITATIONS.md](CITATIONS.md). Absent = none.",
+ citations:
+ "The report's citations, by id — see [CITATIONS.md](CITATIONS.md). A citation is cited from markdown with `[label](cite:<id>)` and listed under the claims that rest on it. Absent = none.",
+ sections: "The report's sections, in order.",
+};
+
+export const SECTION_FIELD_DOCS: FieldDocs<Section> = {
+ id: "The section's id (letters, digits, `_ . : -`; at most 64): its anchor on the report page. Unique among the report's section and claim ids.",
+ title: "The section's heading.",
+ body: "Markdown under the heading. May cite inline.",
+ claims: "The section's claims, in order. Absent = none.",
+};
+
+export const CLAIM_FIELD_DOCS: FieldDocs<Claim> = {
+ id: "The claim's id (letters, digits, `_ . : -`; at most 64): its anchor on the report page. Unique among the report's section and claim ids.",
+ text: "The claim, as stated by the document under review (plain text).",
+ verdict: `The ruling on the claim: ${VERDICTS.map((v) => `\`${v}\``).join(", ")}. A fact-check's claim may leave it out (not yet ruled); a sweep's carries none.`,
+ sourceQuote:
+ "The document's own sentence making the claim: `{ \"citation\": \"<id>\" }`, naming a `source` citation (its still is shown with the claim).",
+ findings: "What the evidence shows, in markdown, citing inline: `[label](cite:<id>)`.",
+ citations: "The citations the claim rests on, in the order they are listed under it. Each must exist; none twice.",
+};
diff --git a/common/lib/report/uses.ts b/common/lib/report/uses.ts
@@ -0,0 +1,85 @@
+// EVERY PLACE A REPORT CITES, in reading order — the one walk the validator,
+// the numbering and the back-link index share, so they cannot disagree about
+// what a report cites or in what order.
+//
+// Reading order: the summary; then each section's body, and each of its
+// claims in turn — the claim's source sentence, its findings, then the
+// citations listed under it. Within a markdown field, `cite:` links in the
+// order they are written (lib/citations/inline.ts).
+//
+// Pure, no imports but types and the inline parser: the export site can use it.
+
+import { extractCiteRefs, numberCitations } from "../citations/inline";
+import type { PathSegment } from "../citations/validate";
+import type { Report } from "./schema";
+
+export type CitationUseField = "summary" | "body" | "sourceQuote" | "findings" | "citations";
+
+export type CitationUse = {
+ citationId: string;
+ // The section and claim the use is in; null in the summary (both) or a
+ // section's body (the claim).
+ sectionId: string | null;
+ claimId: string | null;
+ field: CitationUseField;
+ // The JSON path of the field (with the list index for `citations`).
+ path: PathSegment[];
+ // An inline link's label; absent for a listed citation.
+ label?: string;
+};
+
+export function reportCitationUses(report: Report): CitationUse[] {
+ const out: CitationUse[] = [];
+ const inline = (
+ md: string | undefined,
+ field: CitationUseField,
+ path: PathSegment[],
+ sectionId: string | null,
+ claimId: string | null,
+ ) => {
+ for (const ref of extractCiteRefs(md)) {
+ out.push({ citationId: ref.id, sectionId, claimId, field, path, label: ref.label });
+ }
+ };
+ inline(report.summary, "summary", ["summary"], null, null);
+ report.sections.forEach((section, si) => {
+ const sp: PathSegment[] = ["sections", si];
+ inline(section.body, "body", [...sp, "body"], section.id, null);
+ (section.claims ?? []).forEach((claim, ci) => {
+ const cp: PathSegment[] = [...sp, "claims", ci];
+ if (claim.sourceQuote) {
+ out.push({
+ citationId: claim.sourceQuote.citation,
+ sectionId: section.id,
+ claimId: claim.id,
+ field: "sourceQuote",
+ path: [...cp, "sourceQuote", "citation"],
+ });
+ }
+ inline(claim.findings, "findings", [...cp, "findings"], section.id, claim.id);
+ (claim.citations ?? []).forEach((id, i) => {
+ out.push({
+ citationId: id,
+ sectionId: section.id,
+ claimId: claim.id,
+ field: "citations",
+ path: [...cp, "citations", i],
+ });
+ });
+ });
+ });
+ return out;
+}
+
+// Each citation's number in the report, from 1, by first appearance in reading
+// order. Only citations the report defines are numbered (a dangling reference
+// is a validation problem, not a number); a citation the report defines but
+// never cites has none.
+export function reportCitationNumbers(report: Report): Map<string, number> {
+ const defined = report.citations ?? {};
+ return numberCitations(
+ reportCitationUses(report)
+ .map((u) => u.citationId)
+ .filter((id) => Object.hasOwn(defined, id)),
+ );
+}
diff --git a/common/lib/report/validate.ts b/common/lib/report/validate.ts
@@ -0,0 +1,158 @@
+// REPORT VALIDATION — every problem with a `report.json`, as a list with JSON
+// paths. Never throws.
+//
+// Shape first (./schema.ts, zod): a report that does not parse gets zod's
+// problems and nothing else, since the rest cannot be walked. A report that
+// parses gets every value problem at once:
+//
+// - its id is a report id (and, when asked, its directory's name);
+// - the dates are dates, `updated` not before `published`;
+// - the verdict overrides follow the shared rule (./verdicts.mjs);
+// - section and claim ids are reference ids, unique together (they are the
+// report page's anchors);
+// - a sweep's claims carry no verdict;
+// - every citation and source is sound (lib/citations/validate.ts: ids,
+// spans, safe paths, URLs, verification);
+// - every reference resolves: `subject.source`, each claim's `sourceQuote`
+// (to a `source` citation), each listed citation (none twice in one
+// claim), and every `[label](cite:<id>)` link in the summary, the bodies
+// and the findings.
+//
+// What this cannot check is the disk and the corpus — that a still exists, that
+// a quote matches its cues; compose does those, where both are at hand.
+
+import {
+ citationMapProblems,
+ isDateTime,
+ jsonPath,
+ problem,
+ zodProblems,
+ type Parsed,
+ type PathSegment,
+ type Problem,
+} from "../citations/validate";
+import { isRefId } from "../citations/schema";
+import { reportSchema, isReportId, type Report } from "./schema";
+import { reportCitationUses } from "./uses";
+import { verdictOverrideProblems } from "./verdicts";
+
+const DATE_RE = /^\d{4}-\d{2}-\d{2}$/;
+
+function isReportDate(v: string): boolean {
+ if (DATE_RE.test(v)) return isDateTime(`${v}T00:00Z`);
+ return isDateTime(v);
+}
+
+const blank = (s: string) => !/\S/.test(s);
+
+export type ReportValidateOptions = {
+ // The report's directory name: its `id` must be this.
+ id?: string;
+};
+
+function reportProblems(report: Report, opts: ReportValidateOptions): Problem[] {
+ const out: Problem[] = [];
+ const citations = report.citations ?? {};
+ const sources = report.sources ?? {};
+
+ if (!isReportId(report.id)) {
+ out.push(problem(["id"], "must be a lowercase slug (`[a-z0-9][a-z0-9-]*`, at most 64)"));
+ } else if (opts.id !== undefined && opts.id !== report.id) {
+ out.push(problem(["id"], `is ${JSON.stringify(report.id)} but the report's directory is ${JSON.stringify(opts.id)}`));
+ }
+ if (blank(report.title)) out.push(problem(["title"], "must not be blank"));
+ else if (/[\r\n]/.test(report.title)) out.push(problem(["title"], "must be one line"));
+ if (report.subtitle !== undefined && /[\r\n]/.test(report.subtitle)) {
+ out.push(problem(["subtitle"], "must be one line"));
+ }
+ for (const key of ["published", "updated"] as const) {
+ const v = report[key];
+ if (v !== undefined && !isReportDate(v)) {
+ out.push(problem([key], "must be YYYY-MM-DD or an ISO 8601 date-time with a zone"));
+ }
+ }
+ if (
+ report.published !== undefined &&
+ report.updated !== undefined &&
+ isReportDate(report.published) &&
+ isReportDate(report.updated)
+ ) {
+ const p = Date.parse(DATE_RE.test(report.published) ? `${report.published}T00:00Z` : report.published);
+ const u = Date.parse(DATE_RE.test(report.updated) ? `${report.updated}T00:00Z` : report.updated);
+ if (u < p) out.push(problem(["updated"], "is before published"));
+ }
+
+ for (const [v, override] of Object.entries(report.verdicts ?? {})) {
+ for (const message of verdictOverrideProblems(override, "")) {
+ // The shared rule names its own sub-path (`.label`); split it back off.
+ const m = /^\.(\w+) (.*)$/s.exec(message);
+ out.push(m ? problem(["verdicts", v, m[1]], m[2]) : problem(["verdicts", v], message.trim()));
+ }
+ }
+
+ if (report.subject && !Object.hasOwn(sources, report.subject.source)) {
+ out.push(problem(["subject", "source"], `names no source (${JSON.stringify(report.subject.source)} is not in sources)`));
+ }
+
+ out.push(...citationMapProblems(citations, report.sources));
+
+ const anchors = new Map<string, string>();
+ const anchor = (id: string, at: PathSegment[]) => {
+ if (!isRefId(id)) {
+ out.push(problem([...at, "id"], "is not a reference id (letters, digits, `_ . : -`; at most 64)"));
+ return;
+ }
+ const first = anchors.get(id);
+ if (first) out.push(problem([...at, "id"], `${JSON.stringify(id)} is already the id of ${first}`));
+ else anchors.set(id, jsonPath(at));
+ };
+ report.sections.forEach((section, si) => {
+ const sp: PathSegment[] = ["sections", si];
+ anchor(section.id, sp);
+ if (blank(section.title)) out.push(problem([...sp, "title"], "must not be blank"));
+ (section.claims ?? []).forEach((claim, ci) => {
+ const cp: PathSegment[] = [...sp, "claims", ci];
+ anchor(claim.id, cp);
+ if (blank(claim.text)) out.push(problem([...cp, "text"], "must not be blank"));
+ if (report.kind === "sweep" && claim.verdict !== undefined) {
+ out.push(problem([...cp, "verdict"], "a sweep's claims carry no verdict (make the report a factcheck)"));
+ }
+ const listed = new Set<string>();
+ (claim.citations ?? []).forEach((id, i) => {
+ if (listed.has(id)) out.push(problem([...cp, "citations", i], `lists ${JSON.stringify(id)} twice`));
+ listed.add(id);
+ });
+ });
+ });
+
+ for (const use of reportCitationUses(report)) {
+ const id = use.citationId;
+ if (use.field === "citations" || use.field === "sourceQuote") {
+ if (!Object.hasOwn(citations, id)) {
+ out.push(problem(use.path, `names no citation (${JSON.stringify(id)} is not in citations)`));
+ } else if (use.field === "sourceQuote" && citations[id].kind !== "source") {
+ out.push(problem(use.path, `names a ${citations[id].kind} citation; a claim's source sentence is a \`source\` citation`));
+ }
+ continue;
+ }
+ if (id === "") out.push(problem(use.path, `the link [${use.label}](cite:) names no citation`));
+ else if (!Object.hasOwn(citations, id)) {
+ out.push(problem(use.path, `the link [${use.label}](cite:${id}) names no citation (${JSON.stringify(id)} is not in citations)`));
+ }
+ }
+ return out;
+}
+
+// A report: parsed, and every problem. `ok` means it parsed (its shape is a
+// report); `problems` may still be non-empty, and a report with problems must
+// not be published.
+export function parseReport(raw: unknown, opts: ReportValidateOptions = {}): Parsed<Report> {
+ const r = reportSchema.safeParse(raw);
+ if (!r.success) return { ok: false, problems: zodProblems(r.error) };
+ return { ok: true, value: r.data, problems: reportProblems(r.data, opts) };
+}
+
+// Every problem with a report; empty when it is sound.
+export function validateReport(raw: unknown, opts: ReportValidateOptions = {}): Problem[] {
+ return parseReport(raw, opts).problems;
+}