// CITATION VALIDATION — every rule about a citation's VALUES, as a list of // problems with JSON paths. Never throws. // // ./schema.ts checks shape (zod); this checks what shape cannot: ids, spans, // path safety, dates and URLs, references from a citation to its source, and a // computed `verification` block that could not have come from a check. A // document validator (lib/report/validate.ts, `validateCitationSet` below) // runs zod first, maps its issues to the same problem list, and runs these only // on a document that parsed — so a reader gets every value problem at once. import type { z } from "zod"; import { MAX_CITATION_SPAN_SECONDS, citationSetSchema, isRefId, isSpanCitation, type Citation, type CitationSet, type CitationVerification, type Source, } from "./schema"; import { formatMomentSeconds, isSafeChannelSegment, isSafeIdSegment, roundMomentSeconds } from "./moments"; export type Problem = { // Where, as a JSON path from the document's root: `citations.c01.end`, // `sections[0].claims[2].findings`; `""` for the root. path: string; message: string; }; export type PathSegment = string | number; const IDENT_RE = /^[A-Za-z_$][A-Za-z0-9_$]*$/; // A JSON path: `.key` for a plain key, `["k.y"]` for any other, `[i]` for an // index. export function jsonPath(segs: readonly PathSegment[]): string { let out = ""; for (const s of segs) { if (typeof s === "number") out += `[${s}]`; else if (IDENT_RE.test(s)) out += out ? `.${s}` : s; else out += `[${JSON.stringify(s)}]`; } return out; } export function problem(at: readonly PathSegment[], message: string): Problem { return { path: jsonPath(at), message }; } // zod's issues as problems, under `at`. export function zodProblems(error: z.ZodError, at: readonly PathSegment[] = []): Problem[] { return error.issues.map((i) => problem([...at, ...i.path.map((p) => (typeof p === "number" ? p : String(p)))], i.message), ); } const blank = (s: string) => !/\S/.test(s); // `YYYY`, `YYYY-MM`, `YYYY-MM-DD`, or a date-time with a zone. const PARTIAL_DATE_RE = /^(\d{4})(?:-(\d{2})(?:-(\d{2}))?)?$/; const DATE_TIME_RE = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2}(?:\.\d+)?)?(?:Z|[+-]\d{2}:\d{2})$/; export function isDateTime(v: string): boolean { return DATE_TIME_RE.test(v) && Number.isFinite(Date.parse(v)) && isPartialDate(v.slice(0, 10)); } export function isPartialDate(v: string): boolean { if (DATE_TIME_RE.test(v)) return isDateTime(v); const m = PARTIAL_DATE_RE.exec(v); if (!m) return false; if (m[2] === undefined) return true; const month = Number(m[2]); if (month < 1 || month > 12) return false; if (m[3] === undefined) return true; const day = Number(m[3]); const days = new Date(Date.UTC(Number(m[1]), month, 0)).getUTCDate(); return day >= 1 && day <= days; } export function isHttpUrl(v: string): boolean { try { const u = new URL(v); return (u.protocol === "http:" || u.protocol === "https:") && !!u.hostname; } catch { return false; } } // Why `p` is not a safe path relative to a document's directory, or null: it // must be relative, `/`-separated, with no empty, `.` or `..` segment — so it // can neither start elsewhere nor climb out. export function relativePathProblem(p: string): string | null { if (blank(p)) return "is empty"; if (p.includes("\\")) return "must use `/`, not `\\`"; if (p.startsWith("/") || /^[A-Za-z]:/.test(p)) return "must be relative, not absolute"; if (/^[A-Za-z][A-Za-z0-9+.-]*:/.test(p)) return "must be a path, not a URL"; if (p.split("/").some((s) => s === "" || s === "." || s === "..")) { return "must not have empty, `.` or `..` segments (it may not leave the document's directory)"; } if (/[\x00-\x1f]/.test(p)) return "must not contain control characters"; return null; } function textProblems( out: Problem[], at: readonly PathSegment[], value: string | undefined, { required = false, oneLine = false }: { required?: boolean; oneLine?: boolean } = {}, ): void { if (value === undefined) return; if (blank(value)) { out.push(problem(at, required ? "must not be blank" : "must not be blank (leave it out instead)")); return; } if (oneLine && /[\r\n]/.test(value)) out.push(problem(at, "must be one line")); } function dateProblems(out: Problem[], at: readonly PathSegment[], value: string | undefined): void { if (value === undefined) return; if (!isPartialDate(value)) { out.push(problem(at, "must be a date: YYYY, YYYY-MM, YYYY-MM-DD or an ISO 8601 date-time with a zone")); } } function urlProblems(out: Problem[], at: readonly PathSegment[], value: string | undefined): void { if (value === undefined) return; if (!isHttpUrl(value)) out.push(problem(at, "must be an http(s) URL")); } function pathProblems(out: Problem[], at: readonly PathSegment[], value: string | undefined): void { if (value === undefined) return; const why = relativePathProblem(value); if (why) out.push(problem(at, why)); } function verificationProblems( out: Problem[], at: readonly PathSegment[], v: CitationVerification, c: Citation, ): void { if (v.quoteScore !== undefined && !(v.quoteScore >= 0 && v.quoteScore <= 1)) { out.push(problem([...at, "quoteScore"], "must be from 0 to 1 — a check writes it; it is not typed by hand")); } if (v.quoteCheckedAt !== undefined && !isDateTime(v.quoteCheckedAt)) { out.push(problem([...at, "quoteCheckedAt"], "must be an ISO 8601 date-time with a zone")); } if ((v.quoteScore === undefined) !== (v.quoteCheckedAt === undefined)) { out.push( problem( at, "quoteScore and quoteCheckedAt are written together by the check — one without the other was not", ), ); } if (v.voiceChecked !== undefined && !isSpanCitation(c)) { out.push(problem([...at, "voiceChecked"], `a ${c.kind} citation has no voice to check`)); } textProblems(out, [...at, "method"], v.method, { oneLine: true }); } export type CitationContext = { // The container's sources, for a `source` citation's reference. sources?: Readonly>; }; // Every problem with one citation, under `at` (its path in the document). export function citationProblems( c: Citation, at: readonly PathSegment[], ctx: CitationContext = {}, ): Problem[] { const out: Problem[] = []; textProblems(out, [...at, "quote"], c.quote, { required: true }); textProblems(out, [...at, "speaker"], c.speaker, { oneLine: true }); textProblems(out, [...at, "label"], c.label, { oneLine: true }); textProblems(out, [...at, "note"], c.note); dateProblems(out, [...at, "date"], c.date); if (c.verification) verificationProblems(out, [...at, "verification"], c.verification, c); switch (c.kind) { case "video": case "audio": { if (!isSafeChannelSegment(c.channel)) { out.push(problem([...at, "channel"], "must be a channel slug (one safe path segment)")); } if (!isSafeIdSegment(c.id)) { out.push(problem([...at, "id"], "must be a record id (one safe path segment: letters, digits, `.`, `_`, `-`)")); } const before = c.pad?.before ?? 0; const after = c.pad?.after ?? 0; if (c.pad?.before !== undefined && c.pad.before < 0) out.push(problem([...at, "pad", "before"], "must be at least 0")); if (c.pad?.after !== undefined && c.pad.after < 0) out.push(problem([...at, "pad", "after"], "must be at least 0")); if (c.start < 0) out.push(problem([...at, "start"], "must be at least 0")); if (!(c.end > c.start)) { out.push(problem([...at, "end"], `must be after start (${c.start})`)); } else if (!(roundMomentSeconds(c.end) > roundMomentSeconds(c.start))) { out.push( problem( [...at, "end"], `rounds to the start (${formatMomentSeconds(c.start)}): a span is at least 0.01 s at the moment key's precision`, ), ); } else { const total = c.end - c.start + Math.max(0, before) + Math.max(0, after); if (total > MAX_CITATION_SPAN_SECONDS) { out.push( problem( at, `the span with its pad is ${Math.round(total * 100) / 100} s (at most ${MAX_CITATION_SPAN_SECONDS} s)`, ), ); } } break; } case "post": if (!isSafeChannelSegment(c.channel)) { out.push(problem([...at, "channel"], "must be a channel slug (one safe path segment)")); } if (!isSafeIdSegment(c.id)) { out.push(problem([...at, "id"], "must be a record id (one safe path segment: letters, digits, `.`, `_`, `-`)")); } break; case "source": if (!ctx.sources || !Object.hasOwn(ctx.sources, c.source)) { out.push(problem([...at, "source"], `names no source (${JSON.stringify(c.source)} is not in sources)`)); } pathProblems(out, [...at, "image"], c.image); break; case "page": urlProblems(out, [...at, "url"], c.url); urlProblems(out, [...at, "archiveUrl"], c.archiveUrl); textProblems(out, [...at, "title"], c.title, { oneLine: true }); break; } return out; } // Every problem with one source, under `at`. export function sourceProblems(s: Source, at: readonly PathSegment[]): Problem[] { const out: Problem[] = []; textProblems(out, [...at, "title"], s.title, { required: true, oneLine: true }); urlProblems(out, [...at, "url"], s.url); textProblems(out, [...at, "publisher"], s.publisher, { oneLine: true }); textProblems(out, [...at, "author"], s.author, { oneLine: true }); dateProblems(out, [...at, "date"], s.date); (s.archives ?? []).forEach((a, i) => { textProblems(out, [...at, "archives", i, "label"], a.label, { required: true, oneLine: true }); urlProblems(out, [...at, "archives", i, "url"], a.url); textProblems(out, [...at, "archives", i, "context"], a.context); }); pathProblems(out, [...at, "saved"], s.saved); return out; } // Every problem with a container's `sources` and `citations` maps: each id is a // reference id, no id names both a source and a citation (a `cite:` link names // citations only — one id meaning two things is a trap), and every entry's own // problems. export function citationMapProblems( citations: Readonly>, sources: Readonly> | undefined, at: readonly PathSegment[] = [], ): Problem[] { const out: Problem[] = []; for (const [id, s] of Object.entries(sources ?? {})) { const where = [...at, "sources", id]; if (!isRefId(id)) out.push(problem(where, "is not a reference id (letters, digits, `_ . : -`; at most 64)")); if (Object.hasOwn(citations, id)) out.push(problem(where, "is also a citation's id — ids name one thing")); out.push(...sourceProblems(s, where)); } for (const [id, c] of Object.entries(citations)) { const where = [...at, "citations", id]; if (!isRefId(id)) out.push(problem(where, "is not a reference id (letters, digits, `_ . : -`; at most 64)")); out.push(...citationProblems(c, where, { sources })); } return out; } export type Parsed = { ok: true; value: T; problems: Problem[] } | { ok: false; problems: Problem[] }; // A standalone citation set (`archilyzer-citations`): parsed, and every problem. // `ok` means it parsed; `problems` may still be non-empty. export function parseCitationSet(raw: unknown): Parsed { const r = citationSetSchema.safeParse(raw); if (!r.success) return { ok: false, problems: zodProblems(r.error) }; return { ok: true, value: r.data, problems: citationMapProblems(r.data.citations, r.data.sources) }; } // Every problem with a citation set; empty when it is sound. export function validateCitationSet(raw: unknown): Problem[] { return parseCitationSet(raw).problems; }