commit 9f1b0c07ef07d94eb3bc2784aa9decb2fed9ba65
parent 3f5a9f50f512dbcfc15b9b96a22ed5bc21586db7
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 02:18:43 -0400
common: the citation model (lib/citations/): schema, validation, moment keys, inline cites
A citation is a discriminated union on kind (video, audio, post, source,
page) with a verbatim quote, optional speaker/date/label/note and a computed
verification block. zod checks shape; validate.ts reports every value
problem with its JSON path. Moment keys name a cited span or post and its
page under /m/; inline.ts extracts [label](cite:id) links and numbers
citations by first appearance. A standalone citation set
(archilyzer-citations, version 1) carries citations outside a report.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
5 files changed, 1072 insertions(+), 0 deletions(-)
diff --git a/common/lib/citations/citations.test.ts b/common/lib/citations/citations.test.ts
@@ -0,0 +1,266 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ CITATION_KINDS,
+ citationSchema,
+ MAX_CITATION_SPAN_SECONDS,
+ type Citation,
+} from "./schema";
+import {
+ citationProblems,
+ isPartialDate,
+ jsonPath,
+ parseCitationSet,
+ relativePathProblem,
+ validateCitationSet,
+ type Problem,
+} from "./validate";
+import {
+ momentKey,
+ momentKeyOf,
+ momentOf,
+ momentPath,
+ parseMomentKey,
+ parseMomentPath,
+ isSafeIdSegment,
+ isSafeChannelSegment,
+} from "./moments";
+import { citationAnchor, citeHref, extractCiteRefs, numberCitations } from "./inline";
+
+const video = (over: Partial<Extract<Citation, { kind: "video" }>> = {}): Citation => ({
+ kind: "video",
+ channel: "demo-channel",
+ id: "abc123",
+ start: 10,
+ end: 20,
+ quote: "the words as said",
+ ...over,
+});
+
+const paths = (ps: Problem[]) => ps.map((p) => p.path);
+
+const SET = {
+ format: "archilyzer-citations",
+ version: 1,
+ sources: {
+ s0: {
+ kind: "article",
+ title: "A demo article",
+ url: "https://example.test/article",
+ archives: [{ label: "archived", url: "https://archive.example.test/a", context: "as linked here" }],
+ saved: "sources/s0/page.html",
+ },
+ },
+ citations: {
+ c01: { kind: "video", channel: "demo-channel", id: "-abc123", start: 1.5, end: 9.25, quote: "a span", pad: { before: 5, after: 5 } },
+ c02: { kind: "audio", channel: "demo-podcast", id: "ep42", start: 60, end: 75, quote: "an episode", speaker: "A guest" },
+ p01: { kind: "post", channel: "demo-channel", id: "1234567890", quote: "a post", thread: true },
+ a01: { kind: "source", source: "s0", quote: "the sentence", image: "stills/a01.png" },
+ w01: { kind: "page", url: "https://example.test/page", title: "A page", archiveUrl: "https://archive.example.test/p", quote: "on the page" },
+ },
+};
+
+// ─── schema ───
+
+test("a citation set with every kind parses with no problems", () => {
+ const r = parseCitationSet(SET);
+ assert.ok(r.ok);
+ assert.deepEqual(r.problems, []);
+});
+
+test("every kind is a member of the union, and the union is closed", () => {
+ assert.deepEqual(
+ CITATION_KINDS.map((s) => s.shape.kind.value),
+ ["video", "audio", "post", "source", "page"],
+ );
+ assert.equal(citationSchema.safeParse({ kind: "tweet", quote: "x" }).success, false);
+});
+
+test("shape problems: unknown keys, wrong types, a missing quote, the wrong format or version", () => {
+ const bad = {
+ ...SET,
+ version: 2,
+ citations: {
+ c01: { ...SET.citations.c01, colour: "red" },
+ c02: { ...SET.citations.c02, start: "60" },
+ p01: { kind: "post", channel: "demo-channel", id: "1" },
+ },
+ };
+ const r = parseCitationSet(bad);
+ assert.equal(r.ok, false);
+ const ps = paths(r.problems);
+ assert.ok(ps.includes("version"), ps.join());
+ assert.ok(ps.includes("citations.c01"), ps.join()); // the unknown key
+ assert.ok(ps.includes("citations.c02.start"), ps.join());
+ assert.ok(ps.includes("citations.p01.quote"), ps.join());
+ assert.equal(parseCitationSet({ ...SET, format: "other" }).ok, false);
+});
+
+// ─── value rules ───
+
+test("span rules: end after start, pad at least 0, at most the cap with the pad", () => {
+ assert.deepEqual(citationProblems(video(), ["c"]), []);
+ assert.deepEqual(paths(citationProblems(video({ end: 10 }), ["c"])), ["c.end"]);
+ assert.deepEqual(paths(citationProblems(video({ end: 5 }), ["c"])), ["c.end"]);
+ assert.deepEqual(paths(citationProblems(video({ start: -1 }), ["c"])), ["c.start"]);
+ assert.deepEqual(paths(citationProblems(video({ pad: { before: -1 } }), ["c"])), ["c.pad.before"]);
+ // At the cap exactly is fine; past it, counting the pad, is not.
+ assert.deepEqual(citationProblems(video({ start: 0, end: MAX_CITATION_SPAN_SECONDS }), ["c"]), []);
+ assert.deepEqual(citationProblems(video({ start: 0, end: 110, pad: { before: 5, after: 5 } }), ["c"]), []);
+ const over = citationProblems(video({ start: 0, end: 110, pad: { before: 5, after: 6 } }), ["c"]);
+ assert.deepEqual(paths(over), ["c"]);
+ assert.match(over[0].message, /121 s \(at most 120 s\)/);
+ // A span that vanishes at the key's precision.
+ assert.deepEqual(paths(citationProblems(video({ start: 1.001, end: 1.004 }), ["c"])), ["c.end"]);
+});
+
+test("channel and id must be safe path segments; an id may start with -", () => {
+ assert.deepEqual(citationProblems(video({ id: "-dQw4w9WgXcQ" }), ["c"]), []);
+ assert.deepEqual(citationProblems(video({ id: "_x" }), ["c"]), []);
+ for (const id of ["..", ".", "a/b", "a\\b", "", "a b", "a?b"]) {
+ assert.deepEqual(paths(citationProblems(video({ id }), ["c"])), ["c.id"], id);
+ }
+ for (const channel of ["-lead", "..", "a/b", ""]) {
+ assert.deepEqual(paths(citationProblems(video({ channel }), ["c"])), ["c.channel"], channel);
+ }
+});
+
+test("common fields: a blank quote, a two-line label, a bad date", () => {
+ const ps = citationProblems(video({ quote: " ", label: "a\nb", date: "2024-02-30", speaker: "" }), ["c"]);
+ assert.deepEqual(paths(ps).sort(), ["c.date", "c.label", "c.quote", "c.speaker"]);
+});
+
+test("verification is computed: values that no check writes are flagged", () => {
+ const ok = video({ verification: { quoteScore: 0.93, quoteCheckedAt: "2026-10-01T12:00:00Z", voiceChecked: true, method: "cue-window" } });
+ assert.deepEqual(citationProblems(ok, ["c"]), []);
+ const ps = citationProblems(video({ verification: { quoteScore: 1.4, quoteCheckedAt: "yesterday" } }), ["c"]);
+ assert.deepEqual(paths(ps).sort(), ["c.verification.quoteCheckedAt", "c.verification.quoteScore"]);
+ assert.deepEqual(paths(citationProblems(video({ verification: { quoteScore: 0.5 } }), ["c"])), ["c.verification"]);
+ const post: Citation = { kind: "post", channel: "demo-channel", id: "1", quote: "q", verification: { voiceChecked: true } };
+ assert.deepEqual(paths(citationProblems(post, ["p"])), ["p.verification.voiceChecked"]);
+});
+
+test("source and page citations: the source must exist, paths may not escape, URLs are http(s)", () => {
+ const set = structuredClone(SET) as typeof SET & { citations: Record<string, unknown> };
+ set.citations.a02 = { kind: "source", source: "s9", quote: "q", image: "../outside.png" };
+ set.citations.w02 = { kind: "page", url: "javascript:alert(1)", quote: "q", archiveUrl: "ftp://x" };
+ (set.sources.s0 as { saved: string }).saved = "/abs/page.html";
+ const ps = paths(validateCitationSet(set));
+ assert.deepEqual(ps.sort(), [
+ "citations.a02.image",
+ "citations.a02.source",
+ "citations.w02.archiveUrl",
+ "citations.w02.url",
+ "sources.s0.saved",
+ ]);
+});
+
+test("ids: reference ids only, and no id names both a source and a citation", () => {
+ const set = structuredClone(SET) as unknown as { citations: Record<string, unknown>; sources: Record<string, unknown> };
+ set.citations["bad id"] = { ...SET.citations.p01 };
+ set.citations.s0 = { ...SET.citations.p01 };
+ const ps = validateCitationSet(set);
+ assert.deepEqual(paths(ps).sort(), ['citations["bad id"]', "sources.s0"]);
+});
+
+test("relative paths: relative, forward slashes, inside the directory", () => {
+ assert.equal(relativePathProblem("stills/a01.png"), null);
+ for (const p of ["/etc/x", "C:/x", "a\\b", "../x", "a/../../x", "a//b", "./a", "https://x/y", "", "a/"]) {
+ assert.notEqual(relativePathProblem(p), null, p);
+ }
+});
+
+test("dates: partial dates and zoned date-times, real calendar days only", () => {
+ for (const d of ["2024", "2024-02", "2024-02-29", "2026-10-01T12:00Z", "2026-10-01T12:00:00.5+02:00"]) {
+ assert.ok(isPartialDate(d), d);
+ }
+ for (const d of ["2023-02-29", "2024-13", "24-01-01", "2026-10-01T12:00", "October 2026"]) {
+ assert.ok(!isPartialDate(d), d);
+ }
+});
+
+test("jsonPath: plain keys dotted, other keys quoted, indexes bracketed", () => {
+ assert.equal(jsonPath(["sections", 0, "claims", 2, "findings"]), "sections[0].claims[2].findings");
+ assert.equal(jsonPath(["citations", "c.1", "end"]), 'citations["c.1"].end');
+ assert.equal(jsonPath([]), "");
+});
+
+// ─── moments ───
+
+test("moment keys: video and audio by span, post by id, none for source and page", () => {
+ const set = parseCitationSet(SET);
+ assert.ok(set.ok);
+ const c = set.value.citations;
+ assert.equal(momentKeyOf(c.c01), "demo-channel/-abc123/1.50-9.25");
+ assert.equal(momentKeyOf(c.c02), "demo-podcast/ep42/60.00-75.00");
+ assert.equal(momentKeyOf(c.p01), "demo-channel/1234567890");
+ assert.equal(momentKeyOf(c.a01), null);
+ assert.equal(momentKeyOf(c.w01), null);
+ assert.equal(momentPath(momentOf(c.c01)!), "/m/demo-channel/-abc123/1.50-9.25/");
+ assert.equal(momentPath("demo-channel/1234567890"), "/m/demo-channel/1234567890/");
+});
+
+test("moment keys round-trip, an id starting with - included; the pad is not part of the key", () => {
+ for (const key of ["demo-channel/-abc123/0.00-12.34", "demo-channel/abc_123/3600.10-3610.00", "demo-channel/-1"]) {
+ const m = parseMomentKey(key);
+ assert.ok(m, key);
+ assert.equal(momentKey(m!), key);
+ assert.deepEqual(parseMomentPath(momentPath(m!)), m);
+ assert.deepEqual(parseMomentPath(`/m/${key}`), m);
+ }
+ assert.equal(momentKeyOf(video({ pad: { before: 3 } })), momentKeyOf(video()));
+ // Rounded to the key's precision, and the moment carries the rounded numbers.
+ const m = momentOf(video({ start: 1.234, end: 5.678 }));
+ assert.deepEqual(m, { kind: "span", channel: "demo-channel", id: "abc123", start: 1.23, end: 5.68 });
+});
+
+test("moment keys: only the canonical spelling parses; unsafe segments never key", () => {
+ for (const key of [
+ "demo-channel/abc/1.5-2.00",
+ "demo-channel/abc/01.00-2.00",
+ "demo-channel/abc/2.00-1.00",
+ "demo-channel/abc/1.00-1.00",
+ "demo-channel/abc/-1.00-2.00",
+ "demo-channel/../1.00-2.00",
+ "demo-channel",
+ "demo-channel/abc/1.00-2.00/x",
+ "-x/abc",
+ ]) {
+ assert.equal(parseMomentKey(key), null, key);
+ }
+ assert.equal(parseMomentPath("/reports/x/"), null);
+ assert.throws(() => momentKey({ kind: "post", channel: "demo-channel", id: "a/b" }), /safe path segment/);
+ assert.equal(momentKeyOf(video({ id: ".." })), null);
+ assert.ok(isSafeIdSegment("-abc") && !isSafeChannelSegment("-abc"));
+});
+
+// ─── inline citations ───
+
+test("extractCiteRefs: every cite link in order, with labels and offsets; other links ignored", () => {
+ const md = "He said so [here](cite:c01) and [again](cite:c02), see [the site](https://example.test) and [x]( cite:c01 ).";
+ const refs = extractCiteRefs(md);
+ assert.deepEqual(refs.map((r) => [r.id, r.label]), [["c01", "here"], ["c02", "again"], ["c01", "x"]]);
+ assert.equal(md.slice(refs[0].offset, refs[0].offset + 6), "[here]");
+ assert.deepEqual(extractCiteRefs(undefined), []);
+ assert.deepEqual(extractCiteRefs("[empty](cite:)").map((r) => r.id), [""]);
+});
+
+test("extractCiteRefs: a cite link in code is text about the syntax", () => {
+ const md = [
+ "Write `[label](cite:id)` to cite, like [this](cite:c01).",
+ "```md",
+ "[in a fence](cite:c09)",
+ "```",
+ "~~~~",
+ "[tilde fence](cite:c08)",
+ "~~~~",
+ "And ``[double](cite:c07)`` too, then [after](cite:c02).",
+ ].join("\n");
+ assert.deepEqual(extractCiteRefs(md).map((r) => r.id), ["c01", "c02"]);
+});
+
+test("numberCitations: by first appearance, a repeat keeps its number", () => {
+ assert.deepEqual([...numberCitations(["c03", "c01", "c03", "c02", "c01"])], [["c03", 1], ["c01", 2], ["c02", 3]]);
+ assert.equal(citeHref("c01"), "cite:c01");
+ assert.equal(citationAnchor("c01"), "c-c01");
+});
diff --git a/common/lib/citations/inline.ts b/common/lib/citations/inline.ts
@@ -0,0 +1,92 @@
+// INLINE CITATIONS — the one syntax for citing in markdown, and the numbers a
+// reader sees.
+//
+// [label](cite:<id>)
+//
+// anywhere a document's markdown is (a report's summary, a section's body, a
+// claim's findings). `<id>` names a citation in the same document. A renderer
+// replaces the link with a numbered marker that previews the citation and
+// jumps to it; the citation's anchor is `#c-<id>` (`citationAnchor`), stable
+// across edits because it is the id, not the number.
+//
+// NUMBERS ARE PER DOCUMENT, BY FIRST APPEARANCE: the first citation cited is
+// [1], the next new one [2], and a citation cited again keeps its number. The
+// document decides the order it is read in (lib/report/uses.ts walks a
+// report); `numberCitations` only counts.
+//
+// Code is not citing: a `cite:` link inside an inline code span or a fenced
+// block is text about the syntax, and is skipped.
+//
+// Pure, no imports: the export site's pages and the browser can use it.
+
+export const CITE_SCHEME = "cite:";
+
+export type CiteRef = {
+ id: string;
+ label: string;
+ // Where the link starts in the markdown (a UTF-16 offset).
+ offset: number;
+};
+
+// `[label](cite:id)`. The label may not contain `]`; the id runs to the first
+// `)` or space (an empty id is still a ref, so validation can name it).
+const CITE_LINK_RE = /\[([^\]]*)\]\(\s*cite:([^)\s]*)\s*\)/g;
+
+// Fenced blocks: a line opening with ``` or ~~~ (up to three spaces in) to
+// the line closing it with at least as many of the same, or the end.
+const FENCE_OPEN_RE = /^ {0,3}(`{3,}|~{3,})/;
+
+// An inline code span: a run of backticks, to the next run of the same length.
+const CODE_SPAN_RE = /(?<!`)(`+)(?!`)[\s\S]*?(?<!`)\1(?!`)/g;
+
+const blank = (s: string) => s.replace(/[^\n]/g, " ");
+
+// The markdown with every fenced block and code span blanked to spaces, so
+// offsets still point into the original.
+function blankCode(md: string): string {
+ const lines = md.split("\n");
+ let fence: string | null = null;
+ for (let i = 0; i < lines.length; i++) {
+ if (fence) {
+ const close = new RegExp(`^ {0,3}${fence[0] === "`" ? "`" : "~"}{${fence.length},}\\s*$`);
+ if (close.test(lines[i])) fence = null;
+ lines[i] = blank(lines[i]);
+ continue;
+ }
+ const open = FENCE_OPEN_RE.exec(lines[i]);
+ if (open) {
+ fence = open[1];
+ lines[i] = blank(lines[i]);
+ }
+ }
+ return lines.join("\n").replace(CODE_SPAN_RE, blank);
+}
+
+// Every `cite:` link in the markdown, in order.
+export function extractCiteRefs(md: string | null | undefined): CiteRef[] {
+ if (!md) return [];
+ const scan = blankCode(md);
+ const out: CiteRef[] = [];
+ for (const m of scan.matchAll(CITE_LINK_RE)) {
+ out.push({ id: m[2], label: md.slice(m.index + 1, m.index + 1 + m[1].length), offset: m.index });
+ }
+ return out;
+}
+
+// Numbers by first appearance: each id's number, from 1, in the order the ids
+// are given; a repeated id keeps its first number.
+export function numberCitations(ids: Iterable<string>): Map<string, number> {
+ const out = new Map<string, number>();
+ for (const id of ids) if (!out.has(id)) out.set(id, out.size + 1);
+ return out;
+}
+
+// The href an inline citation is written with.
+export function citeHref(id: string): string {
+ return `${CITE_SCHEME}${id}`;
+}
+
+// The anchor (fragment id, without `#`) of a citation's entry in a document.
+export function citationAnchor(id: string): string {
+ return `c-${id}`;
+}
diff --git a/common/lib/citations/moments.ts b/common/lib/citations/moments.ts
@@ -0,0 +1,151 @@
+// MOMENT KEYS — the one name of a cited moment, and the URL of its page.
+//
+// span (video, audio) <channel>/<id>/<start>-<end> page /m/<channel>/<id>/<start>-<end>/
+// post <channel>/<id> page /m/<channel>/<id>/
+//
+// `source` and `page` citations have no moment: they are shown where they are
+// cited, never on a page of their own.
+//
+// THE KEY IS THE PAGE, so it must be a function of the cited numbers alone:
+// `start` and `end` are written with TWO DECIMALS, always (`Number#toFixed(2)`,
+// the same precision as a clip window's name, lib/clipWindow.ts), and a parse
+// accepts only that spelling — `12.5` is not a key, `12.50` is — so one moment
+// has exactly one URL. Two citations of the same span share a page; the pad is
+// not part of the key (it widens the clip, not the moment). Consumers that cut
+// or show the span use the ROUNDED numbers (`momentOf`), so the page, the clip
+// file and the key agree.
+//
+// SLUG SAFETY: the channel and the id are path segments of a page that a static
+// build writes to disk, so each must be one safe segment — no `/`, no `\`, not
+// `.` or `..`, nothing outside `[A-Za-z0-9._-]`. A video id MAY START WITH `-`
+// (YouTube ids do); that is a safe path segment, but such an id must never be
+// handed to a command line as a bare argument. `momentKey` throws on an unsafe
+// segment rather than build a path from it; `momentKeyOf` answers null.
+//
+// Pure, no imports but types: the export site's pages and the browser can use it.
+
+import type { Citation } from "./schema";
+
+export const MOMENT_SECONDS_DECIMALS = 2;
+
+// The route prefix of every moment page.
+export const MOMENT_ROUTE_PREFIX = "/m/";
+
+// A channel slug: the corpus's own rule (controller/channels.ts
+// CHANNEL_SLUG_RE, which lib may not import), capped in length.
+const CHANNEL_SEGMENT_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$/;
+
+// A record id: like a slug, but it may also start with `-` or `_`.
+const ID_SEGMENT_RE = /^[A-Za-z0-9_-][A-Za-z0-9._-]{0,127}$/;
+
+// A span segment as a key spells it: two decimals each, no sign, no leading
+// zeros but the one before the point.
+const SPAN_SEGMENT_RE = /^(0|[1-9]\d*)\.(\d{2})-(0|[1-9]\d*)\.(\d{2})$/;
+
+export function isSafeChannelSegment(v: unknown): v is string {
+ return typeof v === "string" && v !== ".." && CHANNEL_SEGMENT_RE.test(v);
+}
+
+export function isSafeIdSegment(v: unknown): v is string {
+ return typeof v === "string" && v !== "." && v !== ".." && ID_SEGMENT_RE.test(v);
+}
+
+export type SpanMoment = { kind: "span"; channel: string; id: string; start: number; end: number };
+export type PostMoment = { kind: "post"; channel: string; id: string };
+export type Moment = SpanMoment | PostMoment;
+
+// Seconds as a key spells them.
+export function formatMomentSeconds(s: number): string {
+ return s.toFixed(MOMENT_SECONDS_DECIMALS);
+}
+
+// Seconds rounded to the key's precision, as a number.
+export function roundMomentSeconds(s: number): number {
+ return Number(formatMomentSeconds(s));
+}
+
+// The moment a citation opens, its span rounded to the key's precision; null
+// for a kind without a page (source, page).
+export function momentOf(c: Citation): Moment | null {
+ switch (c.kind) {
+ case "video":
+ case "audio":
+ return {
+ kind: "span",
+ channel: c.channel,
+ id: c.id,
+ start: roundMomentSeconds(c.start),
+ end: roundMomentSeconds(c.end),
+ };
+ case "post":
+ return { kind: "post", channel: c.channel, id: c.id };
+ default:
+ return null;
+ }
+}
+
+// Why a moment cannot be keyed, as a sentence, or null when it can.
+export function momentProblem(m: Moment): string | null {
+ if (!isSafeChannelSegment(m.channel)) return `channel ${JSON.stringify(m.channel)} is not a safe path segment`;
+ if (!isSafeIdSegment(m.id)) return `id ${JSON.stringify(m.id)} is not a safe path segment`;
+ if (m.kind === "span") {
+ if (!Number.isFinite(m.start) || !Number.isFinite(m.end) || m.start < 0) {
+ return "a span's start and end must be numbers, the start at least 0";
+ }
+ if (!(roundMomentSeconds(m.end) > roundMomentSeconds(m.start))) {
+ return `the span ends (${formatMomentSeconds(m.end)}) at or before it starts (${formatMomentSeconds(m.start)}) at the key's precision`;
+ }
+ }
+ return null;
+}
+
+// The moment's key. Throws on a moment `momentProblem` refuses.
+export function momentKey(m: Moment): string {
+ const problem = momentProblem(m);
+ if (problem) throw new Error(`moment key: ${problem}`);
+ const base = `${m.channel}/${m.id}`;
+ return m.kind === "span" ? `${base}/${formatMomentSeconds(m.start)}-${formatMomentSeconds(m.end)}` : base;
+}
+
+// The citation's moment key, or null: a kind without a page, or a citation
+// whose moment cannot be keyed (validation names why).
+export function momentKeyOf(c: Citation): string | null {
+ const m = momentOf(c);
+ if (!m || momentProblem(m)) return null;
+ return momentKey(m);
+}
+
+// The page of a moment (or of a key): `/m/<key>/`.
+export function momentPath(m: Moment | string): string {
+ return `${MOMENT_ROUTE_PREFIX}${typeof m === "string" ? m : momentKey(m)}/`;
+}
+
+// A key back into its moment, or null for anything that is not a key exactly
+// as `momentKey` spells it.
+export function parseMomentKey(key: string): Moment | null {
+ const parts = key.split("/");
+ if (parts.length === 2) {
+ const [channel, id] = parts;
+ if (!isSafeChannelSegment(channel) || !isSafeIdSegment(id)) return null;
+ return { kind: "post", channel, id };
+ }
+ if (parts.length === 3) {
+ const [channel, id, spanPart] = parts;
+ if (!isSafeChannelSegment(channel) || !isSafeIdSegment(id)) return null;
+ const m = SPAN_SEGMENT_RE.exec(spanPart);
+ if (!m) return null;
+ const start = Number(`${m[1]}.${m[2]}`);
+ const end = Number(`${m[3]}.${m[4]}`);
+ if (!(end > start)) return null;
+ return { kind: "span", channel, id, start, end };
+ }
+ return null;
+}
+
+// A moment page's path (`/m/<key>/`, the trailing slash optional) back into
+// its moment, or null.
+export function parseMomentPath(p: string): Moment | null {
+ if (!p.startsWith(MOMENT_ROUTE_PREFIX)) return null;
+ const key = p.slice(MOMENT_ROUTE_PREFIX.length).replace(/\/$/, "");
+ return parseMomentKey(key);
+}
diff --git a/common/lib/citations/schema.ts b/common/lib/citations/schema.ts
@@ -0,0 +1,268 @@
+// THE CITATION MODEL — one definition of what a citation is, for every document
+// that cites the corpus: a report (`report.json`, lib/report/), a standalone
+// citation set (`archilyzer-citations`, below — what a converter of a sweep, an
+// /ask answer or a video manifest emits), and whatever comes next.
+//
+// A citation is a discriminated union on `kind`:
+//
+// video a span of a video record: channel, id, start, end, pad
+// audio a span of an audio record (a podcast): the same fields as video;
+// rendered with a poster instead of a picture
+// post a post record: channel, id, thread
+// source a sentence of a document under review: the source's id in
+// `sources`, a still of the sentence (`image`); rendered with that
+// source's archive links
+// page a web page that is not in the corpus: url, title, archiveUrl
+//
+// and every kind carries the common fields: a VERBATIM `quote`, and optional
+// `speaker`, `date`, `label`, `note`, and `verification` — the one block that is
+// COMPUTED (filled by compose when it checks the quote against the cues, the
+// voice against the speaker), never typed by hand; the validator flags a block
+// whose values could not have come from a check.
+//
+// EXTENDING THE UNION: a new kind is one more member schema in CITATION_KINDS,
+// its *_FIELD_DOCS record (CITATIONS.md is generated from them), its moment
+// shape in ./moments.ts when it has a page of its own, and its rules in
+// ./validate.ts. A reader of an older version refuses a kind it does not know
+// (the union is closed), which is why the containers carry a `version`.
+//
+// THE SPLIT, and why: zod here checks SHAPE only — types, required keys,
+// literal kinds, no unknown keys. Every rule about VALUES (ids, spans, paths,
+// dates, URLs, references between citations and sources) is ./validate.ts's,
+// which runs after a successful parse and reports EVERY problem with its JSON
+// path, instead of the first shape error hiding the rest.
+//
+// SERVER-ONLY: zod. A client importer takes the types with `import type`; the
+// pure helpers (./moments.ts, ./inline.ts) import nothing from here but types.
+
+import { z } from "zod";
+import type { FieldDocs } from "../fieldDocs";
+
+// The version of the citation model a container declares. Bumped when a reader
+// of the old version would misread a document of the new one.
+export const CITATIONS_VERSION = 1;
+
+// The longest span a video or audio citation may cut, pad included, in seconds.
+// A moment page carries one self-hosted evidence clip of it; this keeps every
+// clip a citation (not a re-upload) and well under a static host's file limit.
+export const MAX_CITATION_SPAN_SECONDS = 120;
+
+const text = z.string();
+
+export const verificationSchema = z.strictObject({
+ quoteScore: z.number().optional(),
+ quoteCheckedAt: text.optional(),
+ voiceChecked: z.boolean().optional(),
+ method: text.optional(),
+});
+
+export type CitationVerification = z.infer<typeof verificationSchema>;
+
+export const CITATION_VERIFICATION_FIELD_DOCS: FieldDocs<CitationVerification> = {
+ quoteScore:
+ "How closely `quote` matches what the record says at the cited place, from 0 (nothing alike) to 1 (verbatim). Written by the check, with `quoteCheckedAt`.",
+ quoteCheckedAt: "When the quote was checked: an ISO 8601 date-time. Written with `quoteScore`.",
+ voiceChecked:
+ "True when the speaker's voice in the cited span was checked against `speaker`. Only a span (video, audio) has a voice to check.",
+ method: "What did the checking, e.g. the cue-window comparison and its version. Free text, one line.",
+};
+
+const common = {
+ quote: text,
+ speaker: text.optional(),
+ date: text.optional(),
+ label: text.optional(),
+ note: text.optional(),
+ verification: verificationSchema.optional(),
+};
+
+const pad = z.strictObject({
+ before: z.number().optional(),
+ after: z.number().optional(),
+});
+
+const span = {
+ channel: text,
+ id: text,
+ start: z.number(),
+ end: z.number(),
+ pad: pad.optional(),
+};
+
+export const videoCitationSchema = z.strictObject({ kind: z.literal("video"), ...span, ...common });
+export const audioCitationSchema = z.strictObject({ kind: z.literal("audio"), ...span, ...common });
+export const postCitationSchema = z.strictObject({
+ kind: z.literal("post"),
+ channel: text,
+ id: text,
+ thread: z.boolean().optional(),
+ ...common,
+});
+export const sourceCitationSchema = z.strictObject({
+ kind: z.literal("source"),
+ source: text,
+ image: text.optional(),
+ ...common,
+});
+export const pageCitationSchema = z.strictObject({
+ kind: z.literal("page"),
+ url: text,
+ title: text.optional(),
+ archiveUrl: text.optional(),
+ ...common,
+});
+
+// Every member, in the order CITATIONS.md documents them.
+export const CITATION_KINDS = [
+ videoCitationSchema,
+ audioCitationSchema,
+ postCitationSchema,
+ sourceCitationSchema,
+ pageCitationSchema,
+] as const;
+
+export const citationSchema = z.discriminatedUnion("kind", [...CITATION_KINDS]);
+
+export type VideoCitation = z.infer<typeof videoCitationSchema>;
+export type AudioCitation = z.infer<typeof audioCitationSchema>;
+export type PostCitation = z.infer<typeof postCitationSchema>;
+export type SourceCitation = z.infer<typeof sourceCitationSchema>;
+export type PageCitation = z.infer<typeof pageCitationSchema>;
+export type Citation = z.infer<typeof citationSchema>;
+export type CitationKind = Citation["kind"];
+// The kinds that cite a span of a record's media: a clip, a start and an end.
+export type SpanCitation = VideoCitation | AudioCitation;
+export type CitationPad = z.infer<typeof pad>;
+
+export const SPAN_KINDS: readonly CitationKind[] = ["video", "audio"];
+
+export function isSpanCitation(c: Citation): c is SpanCitation {
+ return c.kind === "video" || c.kind === "audio";
+}
+
+export type CitationCommon = Pick<VideoCitation, keyof typeof common>;
+
+export const CITATION_COMMON_FIELD_DOCS: FieldDocs<CitationCommon> = {
+ quote:
+ "The cited words, VERBATIM — as the record says them (a span's cues, a post's text, the source's sentence, the page's text). Never a paraphrase: compose checks a span's quote against its cues and fails on drift.",
+ speaker: "Who says the quote, when that is not the record's own channel (a guest, a co-host, a caller).",
+ date: "When the quote was said or written: `YYYY`, `YYYY-MM`, `YYYY-MM-DD` or an ISO 8601 date-time. Absent = the record's own date.",
+ label: "A short display name for the citation (one line), used where its number alone would be too little.",
+ note: "An editorial note shown with the citation (plain text): context the quote needs.",
+ verification:
+ "COMPUTED, not authored: what checking this citation found, written by compose. A hand-typed block that could not have come from a check (a score outside 0–1, a score without its time) is a validation problem.",
+};
+
+type Own<T> = Omit<T, keyof CitationCommon>;
+
+export const VIDEO_CITATION_FIELD_DOCS: FieldDocs<Own<VideoCitation>> = {
+ kind: '`"video"`.',
+ channel: "The record's channel slug (its directory under `transcripts/channels/`). A path segment of the moment page.",
+ id: "The record's video id (its directory under `data/`; may start with `-`). A path segment of the moment page.",
+ start: "Where the cited span starts, in seconds from the start of the record.",
+ end: "Where the cited span ends, in seconds; after `start`.",
+ pad: "Context around the span in the evidence clip, in seconds. Absent = none. The span plus its pad is at most 120 s.",
+};
+
+export const AUDIO_CITATION_FIELD_DOCS: FieldDocs<Own<AudioCitation>> = {
+ ...VIDEO_CITATION_FIELD_DOCS,
+ kind: '`"audio"`: a span of a record with no picture worth showing (a podcast). Rendered with a poster.',
+};
+
+export const CITATION_PAD_FIELD_DOCS: FieldDocs<CitationPad> = {
+ before: "Seconds of context before `start`; ≥ 0. Absent = 0.",
+ after: "Seconds of context after `end`; ≥ 0. Absent = 0.",
+};
+
+export const POST_CITATION_FIELD_DOCS: FieldDocs<Own<PostCitation>> = {
+ kind: '`"post"`.',
+ channel: "The post's channel slug. A path segment of the moment page.",
+ id: "The post's id. A path segment of the moment page.",
+ thread: "True to show the post with the thread it belongs to. Absent = the post alone.",
+};
+
+export const SOURCE_CITATION_FIELD_DOCS: FieldDocs<Own<SourceCitation>> = {
+ kind: '`"source"`: a sentence of a document under review.',
+ source: "The id of the document in the container's `sources`. Its archive links are shown with the citation.",
+ image:
+ "A still of the sentence as the document shows it: a path relative to the container's directory (e.g. `stills/a01.png`), never absolute, never leaving it.",
+};
+
+export const PAGE_CITATION_FIELD_DOCS: FieldDocs<Own<PageCitation>> = {
+ kind: '`"page"`: a web page outside the corpus.',
+ url: "The page's address: an http(s) URL.",
+ title: "The page's title.",
+ archiveUrl: "An archived copy of the page (an http(s) URL), shown beside the live link.",
+};
+
+// ─── Sources: the documents a `source` citation quotes ───
+
+export const SOURCE_KINDS = ["article", "page", "video", "post", "document", "other"] as const;
+
+export const sourceArchiveSchema = z.strictObject({
+ label: text,
+ url: text,
+ context: text.optional(),
+});
+
+export const sourceSchema = z.strictObject({
+ kind: z.enum(SOURCE_KINDS),
+ title: text,
+ url: text.optional(),
+ publisher: text.optional(),
+ author: text.optional(),
+ date: text.optional(),
+ archives: z.array(sourceArchiveSchema).optional(),
+ saved: text.optional(),
+});
+
+export type Source = z.infer<typeof sourceSchema>;
+export type SourceArchive = z.infer<typeof sourceArchiveSchema>;
+
+export const SOURCE_FIELD_DOCS: FieldDocs<Source> = {
+ kind: `What the document is: ${SOURCE_KINDS.map((k) => `\`${k}\``).join(", ")}.`,
+ title: "The document's title.",
+ url: "Where the document lives: an http(s) URL.",
+ publisher: "Who published it (the outlet, the site).",
+ author: "Who wrote it.",
+ date: "When it was published: `YYYY`, `YYYY-MM`, `YYYY-MM-DD` or an ISO 8601 date-time.",
+ archives: "The document's archive links, in context — as the document had them. Shown with every citation of it.",
+ saved:
+ "A saved copy of the document, relative to the container's directory (e.g. `sources/s0/page.html`): the input the stills are shot from. NEVER published.",
+};
+
+export const SOURCE_ARCHIVE_FIELD_DOCS: FieldDocs<SourceArchive> = {
+ label: "The link's text.",
+ url: "The archived copy: an http(s) URL.",
+ context: "The words around the link in the document, so a reader sees what it was offered as.",
+};
+
+// ─── A standalone citation set ───
+
+export const CITATION_SET_FORMAT = "archilyzer-citations";
+
+export const citationSetSchema = z.strictObject({
+ format: z.literal(CITATION_SET_FORMAT),
+ version: z.literal(CITATIONS_VERSION),
+ sources: z.record(z.string(), sourceSchema).optional(),
+ citations: z.record(z.string(), citationSchema),
+});
+
+export type CitationSet = z.infer<typeof citationSetSchema>;
+
+export const CITATION_SET_FIELD_DOCS: FieldDocs<CitationSet> = {
+ format: `\`"${CITATION_SET_FORMAT}"\`.`,
+ version: `\`${CITATIONS_VERSION}\`.`,
+ sources: "The documents the `source` citations quote, by id. Absent = none.",
+ citations: "The citations, by id. An id is what a `[label](cite:<id>)` link names.",
+};
+
+// A reference id: a citation's, a source's, a section's or a claim's. Letters,
+// digits and `_ . : -`, starting with a letter or digit, at most 64 — the same
+// rule umtool's fact-check claim ids follow. An id is used in a URL fragment
+// (`#c-<id>`), never as a path segment.
+export const REF_ID_RE = /^[A-Za-z0-9][A-Za-z0-9_.:-]{0,63}$/;
+
+export function isRefId(v: unknown): v is string {
+ return typeof v === "string" && REF_ID_RE.test(v);
+}
diff --git a/common/lib/citations/validate.ts b/common/lib/citations/validate.ts
@@ -0,0 +1,295 @@
+// CITATION VALIDATION — every rule about a citation's VALUES, as a list of
+// problems with JSON paths. Never throws.
+//
+// ./schema.ts checks shape (zod); this checks what shape cannot: ids, spans,
+// path safety, dates and URLs, references from a citation to its source, and a
+// computed `verification` block that could not have come from a check. A
+// document validator (lib/report/validate.ts, `validateCitationSet` below)
+// runs zod first, maps its issues to the same problem list, and runs these only
+// on a document that parsed — so a reader gets every value problem at once.
+
+import type { z } from "zod";
+import {
+ MAX_CITATION_SPAN_SECONDS,
+ citationSetSchema,
+ isRefId,
+ isSpanCitation,
+ type Citation,
+ type CitationSet,
+ type CitationVerification,
+ type Source,
+} from "./schema";
+import { formatMomentSeconds, isSafeChannelSegment, isSafeIdSegment, roundMomentSeconds } from "./moments";
+
+export type Problem = {
+ // Where, as a JSON path from the document's root: `citations.c01.end`,
+ // `sections[0].claims[2].findings`; `""` for the root.
+ path: string;
+ message: string;
+};
+
+export type PathSegment = string | number;
+
+const IDENT_RE = /^[A-Za-z_$][A-Za-z0-9_$]*$/;
+
+// A JSON path: `.key` for a plain key, `["k.y"]` for any other, `[i]` for an
+// index.
+export function jsonPath(segs: readonly PathSegment[]): string {
+ let out = "";
+ for (const s of segs) {
+ if (typeof s === "number") out += `[${s}]`;
+ else if (IDENT_RE.test(s)) out += out ? `.${s}` : s;
+ else out += `[${JSON.stringify(s)}]`;
+ }
+ return out;
+}
+
+export function problem(at: readonly PathSegment[], message: string): Problem {
+ return { path: jsonPath(at), message };
+}
+
+// zod's issues as problems, under `at`.
+export function zodProblems(error: z.ZodError, at: readonly PathSegment[] = []): Problem[] {
+ return error.issues.map((i) =>
+ problem([...at, ...i.path.map((p) => (typeof p === "number" ? p : String(p)))], i.message),
+ );
+}
+
+const blank = (s: string) => !/\S/.test(s);
+
+// `YYYY`, `YYYY-MM`, `YYYY-MM-DD`, or a date-time with a zone.
+const PARTIAL_DATE_RE = /^(\d{4})(?:-(\d{2})(?:-(\d{2}))?)?$/;
+const DATE_TIME_RE = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2}(?:\.\d+)?)?(?:Z|[+-]\d{2}:\d{2})$/;
+
+export function isDateTime(v: string): boolean {
+ return DATE_TIME_RE.test(v) && Number.isFinite(Date.parse(v)) && isPartialDate(v.slice(0, 10));
+}
+
+export function isPartialDate(v: string): boolean {
+ if (DATE_TIME_RE.test(v)) return isDateTime(v);
+ const m = PARTIAL_DATE_RE.exec(v);
+ if (!m) return false;
+ if (m[2] === undefined) return true;
+ const month = Number(m[2]);
+ if (month < 1 || month > 12) return false;
+ if (m[3] === undefined) return true;
+ const day = Number(m[3]);
+ const days = new Date(Date.UTC(Number(m[1]), month, 0)).getUTCDate();
+ return day >= 1 && day <= days;
+}
+
+export function isHttpUrl(v: string): boolean {
+ try {
+ const u = new URL(v);
+ return (u.protocol === "http:" || u.protocol === "https:") && !!u.hostname;
+ } catch {
+ return false;
+ }
+}
+
+// Why `p` is not a safe path relative to a document's directory, or null: it
+// must be relative, `/`-separated, with no empty, `.` or `..` segment — so it
+// can neither start elsewhere nor climb out.
+export function relativePathProblem(p: string): string | null {
+ if (blank(p)) return "is empty";
+ if (p.includes("\\")) return "must use `/`, not `\\`";
+ if (p.startsWith("/") || /^[A-Za-z]:/.test(p)) return "must be relative, not absolute";
+ if (/^[A-Za-z][A-Za-z0-9+.-]*:/.test(p)) return "must be a path, not a URL";
+ if (p.split("/").some((s) => s === "" || s === "." || s === "..")) {
+ return "must not have empty, `.` or `..` segments (it may not leave the document's directory)";
+ }
+ if (/[\x00-\x1f]/.test(p)) return "must not contain control characters";
+ return null;
+}
+
+function textProblems(
+ out: Problem[],
+ at: readonly PathSegment[],
+ value: string | undefined,
+ { required = false, oneLine = false }: { required?: boolean; oneLine?: boolean } = {},
+): void {
+ if (value === undefined) return;
+ if (blank(value)) {
+ out.push(problem(at, required ? "must not be blank" : "must not be blank (leave it out instead)"));
+ return;
+ }
+ if (oneLine && /[\r\n]/.test(value)) out.push(problem(at, "must be one line"));
+}
+
+function dateProblems(out: Problem[], at: readonly PathSegment[], value: string | undefined): void {
+ if (value === undefined) return;
+ if (!isPartialDate(value)) {
+ out.push(problem(at, "must be a date: YYYY, YYYY-MM, YYYY-MM-DD or an ISO 8601 date-time with a zone"));
+ }
+}
+
+function urlProblems(out: Problem[], at: readonly PathSegment[], value: string | undefined): void {
+ if (value === undefined) return;
+ if (!isHttpUrl(value)) out.push(problem(at, "must be an http(s) URL"));
+}
+
+function pathProblems(out: Problem[], at: readonly PathSegment[], value: string | undefined): void {
+ if (value === undefined) return;
+ const why = relativePathProblem(value);
+ if (why) out.push(problem(at, why));
+}
+
+function verificationProblems(
+ out: Problem[],
+ at: readonly PathSegment[],
+ v: CitationVerification,
+ c: Citation,
+): void {
+ if (v.quoteScore !== undefined && !(v.quoteScore >= 0 && v.quoteScore <= 1)) {
+ out.push(problem([...at, "quoteScore"], "must be from 0 to 1 — a check writes it; it is not typed by hand"));
+ }
+ if (v.quoteCheckedAt !== undefined && !isDateTime(v.quoteCheckedAt)) {
+ out.push(problem([...at, "quoteCheckedAt"], "must be an ISO 8601 date-time with a zone"));
+ }
+ if ((v.quoteScore === undefined) !== (v.quoteCheckedAt === undefined)) {
+ out.push(
+ problem(
+ at,
+ "quoteScore and quoteCheckedAt are written together by the check — one without the other was not",
+ ),
+ );
+ }
+ if (v.voiceChecked !== undefined && !isSpanCitation(c)) {
+ out.push(problem([...at, "voiceChecked"], `a ${c.kind} citation has no voice to check`));
+ }
+ textProblems(out, [...at, "method"], v.method, { oneLine: true });
+}
+
+export type CitationContext = {
+ // The container's sources, for a `source` citation's reference.
+ sources?: Readonly<Record<string, Source>>;
+};
+
+// Every problem with one citation, under `at` (its path in the document).
+export function citationProblems(
+ c: Citation,
+ at: readonly PathSegment[],
+ ctx: CitationContext = {},
+): Problem[] {
+ const out: Problem[] = [];
+ textProblems(out, [...at, "quote"], c.quote, { required: true });
+ textProblems(out, [...at, "speaker"], c.speaker, { oneLine: true });
+ textProblems(out, [...at, "label"], c.label, { oneLine: true });
+ textProblems(out, [...at, "note"], c.note);
+ dateProblems(out, [...at, "date"], c.date);
+ if (c.verification) verificationProblems(out, [...at, "verification"], c.verification, c);
+
+ switch (c.kind) {
+ case "video":
+ case "audio": {
+ if (!isSafeChannelSegment(c.channel)) {
+ out.push(problem([...at, "channel"], "must be a channel slug (one safe path segment)"));
+ }
+ if (!isSafeIdSegment(c.id)) {
+ out.push(problem([...at, "id"], "must be a record id (one safe path segment: letters, digits, `.`, `_`, `-`)"));
+ }
+ const before = c.pad?.before ?? 0;
+ const after = c.pad?.after ?? 0;
+ if (c.pad?.before !== undefined && c.pad.before < 0) out.push(problem([...at, "pad", "before"], "must be at least 0"));
+ if (c.pad?.after !== undefined && c.pad.after < 0) out.push(problem([...at, "pad", "after"], "must be at least 0"));
+ if (c.start < 0) out.push(problem([...at, "start"], "must be at least 0"));
+ if (!(c.end > c.start)) {
+ out.push(problem([...at, "end"], `must be after start (${c.start})`));
+ } else if (!(roundMomentSeconds(c.end) > roundMomentSeconds(c.start))) {
+ out.push(
+ problem(
+ [...at, "end"],
+ `rounds to the start (${formatMomentSeconds(c.start)}): a span is at least 0.01 s at the moment key's precision`,
+ ),
+ );
+ } else {
+ const total = c.end - c.start + Math.max(0, before) + Math.max(0, after);
+ if (total > MAX_CITATION_SPAN_SECONDS) {
+ out.push(
+ problem(
+ at,
+ `the span with its pad is ${Math.round(total * 100) / 100} s (at most ${MAX_CITATION_SPAN_SECONDS} s)`,
+ ),
+ );
+ }
+ }
+ break;
+ }
+ case "post":
+ if (!isSafeChannelSegment(c.channel)) {
+ out.push(problem([...at, "channel"], "must be a channel slug (one safe path segment)"));
+ }
+ if (!isSafeIdSegment(c.id)) {
+ out.push(problem([...at, "id"], "must be a record id (one safe path segment: letters, digits, `.`, `_`, `-`)"));
+ }
+ break;
+ case "source":
+ if (!ctx.sources || !Object.hasOwn(ctx.sources, c.source)) {
+ out.push(problem([...at, "source"], `names no source (${JSON.stringify(c.source)} is not in sources)`));
+ }
+ pathProblems(out, [...at, "image"], c.image);
+ break;
+ case "page":
+ urlProblems(out, [...at, "url"], c.url);
+ urlProblems(out, [...at, "archiveUrl"], c.archiveUrl);
+ textProblems(out, [...at, "title"], c.title, { oneLine: true });
+ break;
+ }
+ return out;
+}
+
+// Every problem with one source, under `at`.
+export function sourceProblems(s: Source, at: readonly PathSegment[]): Problem[] {
+ const out: Problem[] = [];
+ textProblems(out, [...at, "title"], s.title, { required: true, oneLine: true });
+ urlProblems(out, [...at, "url"], s.url);
+ textProblems(out, [...at, "publisher"], s.publisher, { oneLine: true });
+ textProblems(out, [...at, "author"], s.author, { oneLine: true });
+ dateProblems(out, [...at, "date"], s.date);
+ (s.archives ?? []).forEach((a, i) => {
+ textProblems(out, [...at, "archives", i, "label"], a.label, { required: true, oneLine: true });
+ urlProblems(out, [...at, "archives", i, "url"], a.url);
+ textProblems(out, [...at, "archives", i, "context"], a.context);
+ });
+ pathProblems(out, [...at, "saved"], s.saved);
+ return out;
+}
+
+// Every problem with a container's `sources` and `citations` maps: each id is a
+// reference id, no id names both a source and a citation (a `cite:` link names
+// citations only — one id meaning two things is a trap), and every entry's own
+// problems.
+export function citationMapProblems(
+ citations: Readonly<Record<string, Citation>>,
+ sources: Readonly<Record<string, Source>> | undefined,
+ at: readonly PathSegment[] = [],
+): Problem[] {
+ const out: Problem[] = [];
+ for (const [id, s] of Object.entries(sources ?? {})) {
+ const where = [...at, "sources", id];
+ if (!isRefId(id)) out.push(problem(where, "is not a reference id (letters, digits, `_ . : -`; at most 64)"));
+ if (Object.hasOwn(citations, id)) out.push(problem(where, "is also a citation's id — ids name one thing"));
+ out.push(...sourceProblems(s, where));
+ }
+ for (const [id, c] of Object.entries(citations)) {
+ const where = [...at, "citations", id];
+ if (!isRefId(id)) out.push(problem(where, "is not a reference id (letters, digits, `_ . : -`; at most 64)"));
+ out.push(...citationProblems(c, where, { sources }));
+ }
+ return out;
+}
+
+export type Parsed<T> = { ok: true; value: T; problems: Problem[] } | { ok: false; problems: Problem[] };
+
+// A standalone citation set (`archilyzer-citations`): parsed, and every problem.
+// `ok` means it parsed; `problems` may still be non-empty.
+export function parseCitationSet(raw: unknown): Parsed<CitationSet> {
+ const r = citationSetSchema.safeParse(raw);
+ if (!r.success) return { ok: false, problems: zodProblems(r.error) };
+ return { ok: true, value: r.data, problems: citationMapProblems(r.data.citations, r.data.sources) };
+}
+
+// Every problem with a citation set; empty when it is sound.
+export function validateCitationSet(raw: unknown): Problem[] {
+ return parseCitationSet(raw).problems;
+}