// THE CITATION MODEL — one definition of what a citation is, for every document // that cites the corpus: a report (`report.json`, lib/report/), a standalone // citation set (`archilyzer-citations`, below — what a converter of a sweep, an // /ask answer or a video manifest emits), and whatever comes next. // // A citation is a discriminated union on `kind`: // // video a span of a video record: channel, id, start, end, pad // audio a span of an audio record (a podcast): the same fields as video; // rendered with a poster instead of a picture // post a post record: channel, id, thread // source a sentence of a document under review: the source's id in // `sources`, a still of the sentence (`image`); rendered with that // source's archive links // page a web page that is not in the corpus: url, title, archiveUrl // // and every kind carries the common fields: a VERBATIM `quote`, and optional // `speaker`, `date`, `label`, `note`, and `verification` — the one block that is // COMPUTED (filled by compose when it checks the quote against the cues, the // voice against the speaker), never typed by hand; the validator flags a block // whose values could not have come from a check. // // EXTENDING THE UNION: a new kind is one more member schema in CITATION_KINDS, // its *_FIELD_DOCS record (CITATIONS.md is generated from them), its moment // shape in ./moments.ts when it has a page of its own, and its rules in // ./validate.ts. A reader of an older version refuses a kind it does not know // (the union is closed), which is why the containers carry a `version`. // // THE SPLIT, and why: zod here checks SHAPE only — types, required keys, // literal kinds, no unknown keys. Every rule about VALUES (ids, spans, paths, // dates, URLs, references between citations and sources) is ./validate.ts's, // which runs after a successful parse and reports EVERY problem with its JSON // path, instead of the first shape error hiding the rest. // // SERVER-ONLY: zod. A client importer takes the types with `import type`; the // pure helpers (./moments.ts, ./inline.ts) import nothing from here but types. import { z } from "zod"; import type { FieldDocs } from "../fieldDocs"; // The version of the citation model a container declares. Bumped when a reader // of the old version would misread a document of the new one. export const CITATIONS_VERSION = 1; // The longest span a video or audio citation may cut, pad included, in seconds. // A moment page carries one self-hosted evidence clip of it; this keeps every // clip a citation (not a re-upload) and well under a static host's file limit. export const MAX_CITATION_SPAN_SECONDS = 120; const text = z.string(); export const verificationSchema = z.strictObject({ quoteScore: z.number().optional(), quoteCheckedAt: text.optional(), voiceChecked: z.boolean().optional(), method: text.optional(), }); export type CitationVerification = z.infer; export const CITATION_VERIFICATION_FIELD_DOCS: FieldDocs = { quoteScore: "How closely `quote` matches what the record says at the cited place, from 0 (nothing alike) to 1 (verbatim). Written by the check, with `quoteCheckedAt`.", quoteCheckedAt: "When the quote was checked: an ISO 8601 date-time. Written with `quoteScore`.", voiceChecked: "True when the speaker's voice in the cited span was checked against `speaker`. Only a span (video, audio) has a voice to check.", method: "What did the checking, e.g. the cue-window comparison and its version. Free text, one line.", }; // Where a report's evidence came from, relative to the document it reviews: // the document itself gave it ("subject"), or the report's author found it // ("added"). Absent = not known. export const CITATION_ORIGINS = ["subject", "added"] as const; export type CitationOrigin = (typeof CITATION_ORIGINS)[number]; const common = { quote: text, speaker: text.optional(), date: text.optional(), label: text.optional(), note: text.optional(), verification: verificationSchema.optional(), origin: z.enum(CITATION_ORIGINS).optional(), }; const pad = z.strictObject({ before: z.number().optional(), after: z.number().optional(), }); const span = { channel: text, id: text, start: z.number(), end: z.number(), pad: pad.optional(), }; export const videoCitationSchema = z.strictObject({ kind: z.literal("video"), ...span, ...common }); export const audioCitationSchema = z.strictObject({ kind: z.literal("audio"), ...span, ...common }); export const postCitationSchema = z.strictObject({ kind: z.literal("post"), channel: text, id: text, thread: z.boolean().optional(), ...common, }); export const sourceCitationSchema = z.strictObject({ kind: z.literal("source"), source: text, image: text.optional(), ...common, }); export const pageCitationSchema = z.strictObject({ kind: z.literal("page"), url: text, title: text.optional(), archiveUrl: text.optional(), ...common, }); // Every member, in the order CITATIONS.md documents them. export const CITATION_KINDS = [ videoCitationSchema, audioCitationSchema, postCitationSchema, sourceCitationSchema, pageCitationSchema, ] as const; export const citationSchema = z.discriminatedUnion("kind", [...CITATION_KINDS]); export type VideoCitation = z.infer; export type AudioCitation = z.infer; export type PostCitation = z.infer; export type SourceCitation = z.infer; export type PageCitation = z.infer; export type Citation = z.infer; export type CitationKind = Citation["kind"]; // The kinds that cite a span of a record's media: a clip, a start and an end. export type SpanCitation = VideoCitation | AudioCitation; export type CitationPad = z.infer; export const SPAN_KINDS: readonly CitationKind[] = ["video", "audio"]; export function isSpanCitation(c: Citation): c is SpanCitation { return c.kind === "video" || c.kind === "audio"; } export type CitationCommon = Pick; export const CITATION_COMMON_FIELD_DOCS: FieldDocs = { quote: "The cited words, VERBATIM — as the record says them (a span's cues, a post's text, the source's sentence, the page's text). Never a paraphrase: compose checks a span's quote against its cues and fails on drift.", speaker: "Who says the quote, when that is not the record's own channel (a guest, a co-host, a caller).", date: "When the quote was said or written: `YYYY`, `YYYY-MM`, `YYYY-MM-DD` or an ISO 8601 date-time. Absent = the record's own date.", label: "A short display name for the citation (one line), used where its number alone would be too little.", note: "An editorial note shown with the citation (plain text): context the quote needs.", origin: 'In a report that reviews a document (`subject`): where this evidence came from. `"subject"` — the document itself gave it (its link, its embedded clip, its picture); `"added"` — the report\'s author found it, the document did not give it. Absent = not known. A report with no `subject` carries none.', verification: "COMPUTED, not authored: what checking this citation found, written by compose. A hand-typed block that could not have come from a check (a score outside 0–1, a score without its time) is a validation problem.", }; type Own = Omit; export const VIDEO_CITATION_FIELD_DOCS: FieldDocs> = { kind: '`"video"`.', channel: "The record's channel slug (its directory under `transcripts/channels/`). A path segment of the moment page.", id: "The record's video id (its directory under `data/`; may start with `-`). A path segment of the moment page.", start: "Where the cited span starts, in seconds from the start of the record.", end: "Where the cited span ends, in seconds; after `start`.", pad: "Context around the span in the evidence clip, in seconds. Absent = none. The span plus its pad is at most 120 s.", }; export const AUDIO_CITATION_FIELD_DOCS: FieldDocs> = { ...VIDEO_CITATION_FIELD_DOCS, kind: '`"audio"`: a span of a record with no picture worth showing (a podcast). Rendered with a poster.', }; export const CITATION_PAD_FIELD_DOCS: FieldDocs = { before: "Seconds of context before `start`; ≥ 0. Absent = 0.", after: "Seconds of context after `end`; ≥ 0. Absent = 0.", }; export const POST_CITATION_FIELD_DOCS: FieldDocs> = { kind: '`"post"`.', channel: "The post's channel slug. A path segment of the moment page.", id: "The post's id. A path segment of the moment page.", thread: "True to show the post with the thread it belongs to. Absent = the post alone.", }; export const SOURCE_CITATION_FIELD_DOCS: FieldDocs> = { kind: '`"source"`: a sentence of a document under review.', source: "The id of the document in the container's `sources`. Its archive links are shown with the citation.", image: "A still of the sentence as the document shows it: a path relative to the container's directory (e.g. `stills/a01.png`), never absolute, never leaving it.", }; export const PAGE_CITATION_FIELD_DOCS: FieldDocs> = { kind: '`"page"`: a web page outside the corpus.', url: "The page's address: an http(s) URL.", title: "The page's title.", archiveUrl: "An archived copy of the page (an http(s) URL), shown beside the live link.", }; // ─── Sources: the documents a `source` citation quotes ─── export const SOURCE_KINDS = ["article", "page", "video", "post", "document", "other"] as const; export const sourceArchiveSchema = z.strictObject({ label: text, url: text, context: text.optional(), }); // A source's colour: `#rrggbb` only — it is written into a page's style. export const SOURCE_ACCENT_RE = /^#[0-9a-fA-F]{6}$/; export const sourceSchema = z.strictObject({ kind: z.enum(SOURCE_KINDS), title: text, url: text.optional(), publisher: text.optional(), author: text.optional(), date: text.optional(), archives: z.array(sourceArchiveSchema).optional(), note: text.optional(), accent: text.regex(SOURCE_ACCENT_RE).optional(), saved: text.optional(), }); export type Source = z.infer; export type SourceArchive = z.infer; export const SOURCE_FIELD_DOCS: FieldDocs = { kind: `What the document is: ${SOURCE_KINDS.map((k) => `\`${k}\``).join(", ")}.`, title: "The document's title.", url: "Where the document lives: an http(s) URL.", publisher: "Who published it (the outlet, the site).", author: "Who wrote it.", date: "When it was published: `YYYY`, `YYYY-MM`, `YYYY-MM-DD` or an ISO 8601 date-time.", archives: "The document's archive links, in context — as the document had them. Shown with every citation of it.", note: "A note shown with the document's byline (plain text), e.g. which copy of it was read.", accent: "The document's colour, `\"#rrggbb\"`: the rail beside each of its quoted sentences and the edge of its box, so they read as one source. Absent = the theme's border colour.", saved: "A saved copy of the document, relative to the container's directory (e.g. `sources/s0/page.html`): the input the stills are shot from. NEVER published.", }; export const SOURCE_ARCHIVE_FIELD_DOCS: FieldDocs = { label: "The link's text.", url: "The archived copy: an http(s) URL.", context: "The words around the link in the document, so a reader sees what it was offered as.", }; // ─── A standalone citation set ─── export const CITATION_SET_FORMAT = "archilyzer-citations"; export const citationSetSchema = z.strictObject({ format: z.literal(CITATION_SET_FORMAT), version: z.literal(CITATIONS_VERSION), sources: z.record(z.string(), sourceSchema).optional(), citations: z.record(z.string(), citationSchema), }); export type CitationSet = z.infer; export const CITATION_SET_FIELD_DOCS: FieldDocs = { format: `\`"${CITATION_SET_FORMAT}"\`.`, version: `\`${CITATIONS_VERSION}\`.`, sources: "The documents the `source` citations quote, by id. Absent = none.", citations: "The citations, by id. An id is what a `[label](cite:)` link names.", }; // A reference id: a citation's, a source's, a section's or a claim's. Letters, // digits and `_ . : -`, starting with a letter or digit, at most 64 — the same // rule umtool's fact-check claim ids follow. An id is used in a URL fragment // (`#c-`), never as a path segment. export const REF_ID_RE = /^[A-Za-z0-9][A-Za-z0-9_.:-]{0,63}$/; export function isRefId(v: unknown): v is string { return typeof v === "string" && REF_ID_RE.test(v); }