// REPORT VALIDATION — every problem with a `report.json`, as a list with JSON // paths. Never throws. // // Shape first (./schema.ts, zod): a report that does not parse gets zod's // problems and nothing else, since the rest cannot be walked. A report that // parses gets every value problem at once: // // - its id is a report id (and, when asked, its directory's name); // - the dates are dates, `updated` not before `published`; // - the verdict overrides follow the shared rule (./verdicts.mjs); // - section, claim and entry ids are reference ids, unique together (they // are the report page's anchors) and none of the page's own // (./slideRules.ts); // - a timeline entry has a title (one line), a body and dates, its // `updated` not before its `date`; // - the slide fields are in bounds and cite what the article cites // (slideProblems, below); // - a sweep's claims carry no verdict; // - every citation and source is sound (lib/citations/validate.ts: ids, // spans, safe paths, URLs, verification); // - every reference resolves: `subject.source`, each claim's `sourceQuote` // (to a `source` citation), each listed citation (none twice in one // claim), and every `[label](cite:)` link in the summary, the // entries, the bodies and the findings. // // What this cannot check is the disk and the corpus — that a still or the video exists, that // a quote matches its cues; compose does those, where both are at hand. import { citationMapProblems, isDateTime, jsonPath, problem, relativePathProblem, zodProblems, type Parsed, type PathSegment, type Problem, } from "../citations/validate"; import { extractCiteRefs } from "../citations/inline"; import { isRefId } from "../citations/schema"; import { reportDateMs } from "./entries"; import { reportSchema, isReportId, type Claim, type Report, type SlideSpec } from "./schema"; import { RESERVED_ANCHOR_IDS, SLIDE_LINE_MAX, SLIDE_POINT_MAX, SLIDE_POINTS_MAX } from "./slideRules"; import { slidePointText } from "./slides"; import { reportCitationUses } from "./uses"; import { verdictOverrideProblems } from "./verdicts"; const DATE_RE = /^\d{4}-\d{2}-\d{2}$/; function isReportDate(v: string): boolean { if (DATE_RE.test(v)) return isDateTime(`${v}T00:00Z`); return isDateTime(v); } const blank = (s: string) => !/\S/.test(s); export type ReportValidateOptions = { // The report's directory name: its `id` must be this. id?: string; }; function reportProblems(report: Report, opts: ReportValidateOptions): Problem[] { const out: Problem[] = []; const citations = report.citations ?? {}; const sources = report.sources ?? {}; if (!isReportId(report.id)) { out.push(problem(["id"], "must be a lowercase slug (`[a-z0-9][a-z0-9-]*`, at most 64)")); } else if (opts.id !== undefined && opts.id !== report.id) { out.push(problem(["id"], `is ${JSON.stringify(report.id)} but the report's directory is ${JSON.stringify(opts.id)}`)); } if (blank(report.title)) out.push(problem(["title"], "must not be blank")); else if (/[\r\n]/.test(report.title)) out.push(problem(["title"], "must be one line")); if (report.subtitle !== undefined && /[\r\n]/.test(report.subtitle)) { out.push(problem(["subtitle"], "must be one line")); } if (report.video) { const { src, poster, caption } = report.video; const srcProblem = relativePathProblem(src); if (srcProblem) out.push(problem(["video", "src"], srcProblem)); else if (!/\.mp4$/i.test(src)) out.push(problem(["video", "src"], "must be an mp4")); if (poster !== undefined) { const posterProblem = relativePathProblem(poster); if (posterProblem) out.push(problem(["video", "poster"], posterProblem)); else if (!/\.(png|jpe?g|webp)$/i.test(poster)) out.push(problem(["video", "poster"], "must be a png, jpg or webp")); } if (caption !== undefined) { if (blank(caption)) out.push(problem(["video", "caption"], "must not be blank")); else if (/[\r\n]/.test(caption)) out.push(problem(["video", "caption"], "must be one line")); } } for (const key of ["published", "updated"] as const) { const v = report[key]; if (v !== undefined && !isReportDate(v)) { out.push(problem([key], "must be YYYY-MM-DD or an ISO 8601 date-time with a zone")); } } if ( report.published !== undefined && report.updated !== undefined && isReportDate(report.published) && isReportDate(report.updated) ) { const p = Date.parse(DATE_RE.test(report.published) ? `${report.published}T00:00Z` : report.published); const u = Date.parse(DATE_RE.test(report.updated) ? `${report.updated}T00:00Z` : report.updated); if (u < p) out.push(problem(["updated"], "is before published")); } for (const [v, override] of Object.entries(report.verdicts ?? {})) { for (const message of verdictOverrideProblems(override, "")) { // The shared rule names its own sub-path (`.label`); split it back off. const m = /^\.(\w+) (.*)$/s.exec(message); out.push(m ? problem(["verdicts", v, m[1]], m[2]) : problem(["verdicts", v], message.trim())); } } if (report.subject && !Object.hasOwn(sources, report.subject.source)) { out.push(problem(["subject", "source"], `names no source (${JSON.stringify(report.subject.source)} is not in sources)`)); } out.push(...citationMapProblems(citations, report.sources)); if (report.method !== undefined && extractCiteRefs(report.method).length > 0) { out.push(problem(["method"], "cites nothing: a citation belongs in the summary, an entry, a section or a claim")); } if (!report.subject) { for (const [id, c] of Object.entries(citations)) { if (c.origin !== undefined) { out.push(problem(["citations", id, "origin"], "says where evidence came from relative to the document under review; the report has no `subject`")); } } } const anchors = new Map(); const anchor = (id: string, at: PathSegment[]) => { if (!isRefId(id)) { out.push(problem([...at, "id"], "is not a reference id (letters, digits, `_ . : -`; at most 64)")); return; } const first = anchors.get(id); if (first) out.push(problem([...at, "id"], `${JSON.stringify(id)} is already the id of ${first}`)); else if (RESERVED_ANCHOR_IDS.includes(id)) { out.push(problem([...at, "id"], `${JSON.stringify(id)} is the report page's own anchor (${RESERVED_ANCHOR_IDS.join(", ")}); pick another`)); } else anchors.set(id, jsonPath(at)); }; report.sections.forEach((section, si) => { const sp: PathSegment[] = ["sections", si]; anchor(section.id, sp); if (blank(section.title)) out.push(problem([...sp, "title"], "must not be blank")); (section.claims ?? []).forEach((claim, ci) => { const cp: PathSegment[] = [...sp, "claims", ci]; anchor(claim.id, cp); if (blank(claim.text)) out.push(problem([...cp, "text"], "must not be blank")); if (claim.gist !== undefined) { if (blank(claim.gist)) out.push(problem([...cp, "gist"], "must not be blank")); else if (/[\r\n]/.test(claim.gist)) out.push(problem([...cp, "gist"], "must be one line")); } if (claim.flag !== undefined) { if (blank(claim.flag)) out.push(problem([...cp, "flag"], "must not be blank")); else if (/[\r\n]/.test(claim.flag)) out.push(problem([...cp, "flag"], "must be one line")); } if (report.kind === "sweep" && claim.verdict !== undefined) { out.push(problem([...cp, "verdict"], "a sweep's claims carry no verdict (make the report a factcheck)")); } const listed = new Set(); (claim.citations ?? []).forEach((id, i) => { if (listed.has(id)) out.push(problem([...cp, "citations", i], `lists ${JSON.stringify(id)} twice`)); listed.add(id); }); }); }); (report.entries ?? []).forEach((entry, ei) => { const ep: PathSegment[] = ["entries", ei]; anchor(entry.id, ep); if (blank(entry.title)) out.push(problem([...ep, "title"], "must not be blank")); else if (/[\r\n]/.test(entry.title)) out.push(problem([...ep, "title"], "must be one line")); if (blank(entry.body)) out.push(problem([...ep, "body"], "must not be blank")); const dated = isReportDate(entry.date); if (!dated) out.push(problem([...ep, "date"], "must be YYYY-MM-DD or an ISO 8601 date-time with a zone")); if (entry.updated !== undefined) { if (!isReportDate(entry.updated)) { out.push(problem([...ep, "updated"], "must be YYYY-MM-DD or an ISO 8601 date-time with a zone")); } else if (dated && reportDateMs(entry.updated) < reportDateMs(entry.date)) { out.push(problem([...ep, "updated"], "is before the entry's date")); } } }); out.push(...slideProblems(report)); for (const use of reportCitationUses(report)) { const id = use.citationId; if (use.field === "citations" || use.field === "sourceQuote") { if (!Object.hasOwn(citations, id)) { out.push(problem(use.path, `names no citation (${JSON.stringify(id)} is not in citations)`)); } else if (use.field === "sourceQuote" && citations[id].kind !== "source") { out.push(problem(use.path, `names a ${citations[id].kind} citation; a claim's source sentence is a \`source\` citation`)); } continue; } if (id === "") out.push(problem(use.path, `the link [${use.label}](cite:) names no citation`)); else if (!Object.hasOwn(citations, id)) { out.push(problem(use.path, `the link [${use.label}](cite:${id}) names no citation (${JSON.stringify(id)} is not in citations)`)); } } return out; } // ─── Slides ─── // // The slide fields (./schema.ts `slides`, a section's or a claim's `slide`): // lines are one line and not too long, points are few and short, a `cite` // names a citation of its own section or claim, a point's inline citation // names one the report cites elsewhere (a slide shows the article's evidence; // it never brings its own, which would have no number), and a layout that // shows evidence has some to show. function slideLineProblems(v: string | undefined, at: PathSegment[]): Problem[] { if (v === undefined) return []; if (blank(v)) return [problem(at, "must not be blank")]; if (/[\r\n]/.test(v)) return [problem(at, "must be one line")]; if (v.length > SLIDE_LINE_MAX) return [problem(at, `is ${v.length} characters; a slide's line is at most ${SLIDE_LINE_MAX}`)]; return []; } function slidePointsProblems(points: readonly string[] | undefined, at: PathSegment[], cited: ReadonlySet): Problem[] { if (points === undefined) return []; const out: Problem[] = []; if (points.length === 0) out.push(problem(at, "must hold at least one point (leave it out to derive the slide)")); if (points.length > SLIDE_POINTS_MAX) out.push(problem(at, `holds ${points.length} points; a slide shows at most ${SLIDE_POINTS_MAX}`)); points.forEach((p, i) => { const pp = [...at, i]; if (blank(p)) return void out.push(problem(pp, "must not be blank")); if (/[\r\n]/.test(p)) out.push(problem(pp, "must be one line")); const visible = slidePointText(p); if (visible.length > SLIDE_POINT_MAX) { out.push(problem(pp, `is ${visible.length} characters; a point is at most ${SLIDE_POINT_MAX}`)); } for (const ref of extractCiteRefs(p)) { if (ref.id === "") out.push(problem(pp, `the link [${ref.label}](cite:) names no citation`)); else if (!cited.has(ref.id)) { out.push(problem(pp, `the link [${ref.label}](cite:${ref.id}) names a citation the report does not cite elsewhere (a slide cites what the article cites)`)); } } }); return out; } function slideSpecProblems( spec: SlideSpec | undefined, at: PathSegment[], own: ReadonlySet, cited: ReadonlySet, what: "section" | "claim", hasDefaultEvidence: boolean, ): Problem[] { if (!spec) return []; const out: Problem[] = []; out.push(...slideLineProblems(spec.title, [...at, "title"])); out.push(...slidePointsProblems(spec.points, [...at, "points"], cited)); if (spec.cite !== undefined && !own.has(spec.cite)) { out.push(problem([...at, "cite"], `names ${JSON.stringify(spec.cite)}, which this ${what} does not cite`)); } if ((spec.layout === "evidence" || spec.layout === "quote") && spec.cite === undefined && !hasDefaultEvidence) { out.push(problem([...at, "layout"], `"${spec.layout}" shows a citation, and this ${what}'s slide has none: name one in \`cite\``)); } return out; } // The citations a claim of the document cites: its sentence, its findings' // inline citations and its list. function claimCiteIds(claim: Claim): Set { const ids = new Set(); if (claim.sourceQuote) ids.add(claim.sourceQuote.citation); for (const r of extractCiteRefs(claim.findings)) ids.add(r.id); for (const id of claim.citations ?? []) ids.add(id); return ids; } function slideProblems(report: Report): Problem[] { const out: Problem[] = []; const cited = new Set(reportCitationUses(report).map((u) => u.citationId)); if (report.slides) { out.push(...slideLineProblems(report.slides.title, ["slides", "title"])); out.push(...slideLineProblems(report.slides.closing, ["slides", "closing"])); out.push(...slidePointsProblems(report.slides.points, ["slides", "points"], cited)); } report.sections.forEach((section, si) => { const sp: PathSegment[] = ["sections", si]; const sectionIds = new Set(extractCiteRefs(section.body).map((r) => r.id)); (section.claims ?? []).forEach((claim, ci) => { const own = claimCiteIds(claim); for (const id of own) sectionIds.add(id); const evidence = (claim.citations ?? []).some((id) => id !== claim.sourceQuote?.citation); out.push(...slideSpecProblems(claim.slide, [...sp, "claims", ci, "slide"], own, cited, "claim", evidence)); }); out.push(...slideSpecProblems(section.slide, [...sp, "slide"], sectionIds, cited, "section", false)); }); return out; } // A report: parsed, and every problem. `ok` means it parsed (its shape is a // report); `problems` may still be non-empty, and a report with problems must // not be published. export function parseReport(raw: unknown, opts: ReportValidateOptions = {}): Parsed { const r = reportSchema.safeParse(raw); if (!r.success) return { ok: false, problems: zodProblems(r.error) }; return { ok: true, value: r.data, problems: reportProblems(r.data, opts) }; } // Every problem with a report; empty when it is sound. export function validateReport(raw: unknown, opts: ReportValidateOptions = {}): Problem[] { return parseReport(raw, opts).problems; }