Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 9f1b0c07ef07d94eb3bc2784aa9decb2fed9ba65
parent 3f5a9f50f512dbcfc15b9b96a22ed5bc21586db7
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 02:18:43 -0400

common: the citation model (lib/citations/): schema, validation, moment keys, inline cites

A citation is a discriminated union on kind (video, audio, post, source,
page) with a verbatim quote, optional speaker/date/label/note and a computed
verification block. zod checks shape; validate.ts reports every value
problem with its JSON path. Moment keys name a cited span or post and its
page under /m/; inline.ts extracts [label](cite:id) links and numbers
citations by first appearance. A standalone citation set
(archilyzer-citations, version 1) carries citations outside a report.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/lib/citations/citations.test.ts | 266+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/citations/inline.ts | 92+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/citations/moments.ts | 151++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/citations/schema.ts | 268+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/citations/validate.ts | 295+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
5 files changed, 1072 insertions(+), 0 deletions(-)

diff --git a/common/lib/citations/citations.test.ts b/common/lib/citations/citations.test.ts @@ -0,0 +1,266 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + CITATION_KINDS, + citationSchema, + MAX_CITATION_SPAN_SECONDS, + type Citation, +} from "./schema"; +import { + citationProblems, + isPartialDate, + jsonPath, + parseCitationSet, + relativePathProblem, + validateCitationSet, + type Problem, +} from "./validate"; +import { + momentKey, + momentKeyOf, + momentOf, + momentPath, + parseMomentKey, + parseMomentPath, + isSafeIdSegment, + isSafeChannelSegment, +} from "./moments"; +import { citationAnchor, citeHref, extractCiteRefs, numberCitations } from "./inline"; + +const video = (over: Partial<Extract<Citation, { kind: "video" }>> = {}): Citation => ({ + kind: "video", + channel: "demo-channel", + id: "abc123", + start: 10, + end: 20, + quote: "the words as said", + ...over, +}); + +const paths = (ps: Problem[]) => ps.map((p) => p.path); + +const SET = { + format: "archilyzer-citations", + version: 1, + sources: { + s0: { + kind: "article", + title: "A demo article", + url: "https://example.test/article", + archives: [{ label: "archived", url: "https://archive.example.test/a", context: "as linked here" }], + saved: "sources/s0/page.html", + }, + }, + citations: { + c01: { kind: "video", channel: "demo-channel", id: "-abc123", start: 1.5, end: 9.25, quote: "a span", pad: { before: 5, after: 5 } }, + c02: { kind: "audio", channel: "demo-podcast", id: "ep42", start: 60, end: 75, quote: "an episode", speaker: "A guest" }, + p01: { kind: "post", channel: "demo-channel", id: "1234567890", quote: "a post", thread: true }, + a01: { kind: "source", source: "s0", quote: "the sentence", image: "stills/a01.png" }, + w01: { kind: "page", url: "https://example.test/page", title: "A page", archiveUrl: "https://archive.example.test/p", quote: "on the page" }, + }, +}; + +// ─── schema ─── + +test("a citation set with every kind parses with no problems", () => { + const r = parseCitationSet(SET); + assert.ok(r.ok); + assert.deepEqual(r.problems, []); +}); + +test("every kind is a member of the union, and the union is closed", () => { + assert.deepEqual( + CITATION_KINDS.map((s) => s.shape.kind.value), + ["video", "audio", "post", "source", "page"], + ); + assert.equal(citationSchema.safeParse({ kind: "tweet", quote: "x" }).success, false); +}); + +test("shape problems: unknown keys, wrong types, a missing quote, the wrong format or version", () => { + const bad = { + ...SET, + version: 2, + citations: { + c01: { ...SET.citations.c01, colour: "red" }, + c02: { ...SET.citations.c02, start: "60" }, + p01: { kind: "post", channel: "demo-channel", id: "1" }, + }, + }; + const r = parseCitationSet(bad); + assert.equal(r.ok, false); + const ps = paths(r.problems); + assert.ok(ps.includes("version"), ps.join()); + assert.ok(ps.includes("citations.c01"), ps.join()); // the unknown key + assert.ok(ps.includes("citations.c02.start"), ps.join()); + assert.ok(ps.includes("citations.p01.quote"), ps.join()); + assert.equal(parseCitationSet({ ...SET, format: "other" }).ok, false); +}); + +// ─── value rules ─── + +test("span rules: end after start, pad at least 0, at most the cap with the pad", () => { + assert.deepEqual(citationProblems(video(), ["c"]), []); + assert.deepEqual(paths(citationProblems(video({ end: 10 }), ["c"])), ["c.end"]); + assert.deepEqual(paths(citationProblems(video({ end: 5 }), ["c"])), ["c.end"]); + assert.deepEqual(paths(citationProblems(video({ start: -1 }), ["c"])), ["c.start"]); + assert.deepEqual(paths(citationProblems(video({ pad: { before: -1 } }), ["c"])), ["c.pad.before"]); + // At the cap exactly is fine; past it, counting the pad, is not. + assert.deepEqual(citationProblems(video({ start: 0, end: MAX_CITATION_SPAN_SECONDS }), ["c"]), []); + assert.deepEqual(citationProblems(video({ start: 0, end: 110, pad: { before: 5, after: 5 } }), ["c"]), []); + const over = citationProblems(video({ start: 0, end: 110, pad: { before: 5, after: 6 } }), ["c"]); + assert.deepEqual(paths(over), ["c"]); + assert.match(over[0].message, /121 s \(at most 120 s\)/); + // A span that vanishes at the key's precision. + assert.deepEqual(paths(citationProblems(video({ start: 1.001, end: 1.004 }), ["c"])), ["c.end"]); +}); + +test("channel and id must be safe path segments; an id may start with -", () => { + assert.deepEqual(citationProblems(video({ id: "-dQw4w9WgXcQ" }), ["c"]), []); + assert.deepEqual(citationProblems(video({ id: "_x" }), ["c"]), []); + for (const id of ["..", ".", "a/b", "a\\b", "", "a b", "a?b"]) { + assert.deepEqual(paths(citationProblems(video({ id }), ["c"])), ["c.id"], id); + } + for (const channel of ["-lead", "..", "a/b", ""]) { + assert.deepEqual(paths(citationProblems(video({ channel }), ["c"])), ["c.channel"], channel); + } +}); + +test("common fields: a blank quote, a two-line label, a bad date", () => { + const ps = citationProblems(video({ quote: " ", label: "a\nb", date: "2024-02-30", speaker: "" }), ["c"]); + assert.deepEqual(paths(ps).sort(), ["c.date", "c.label", "c.quote", "c.speaker"]); +}); + +test("verification is computed: values that no check writes are flagged", () => { + const ok = video({ verification: { quoteScore: 0.93, quoteCheckedAt: "2026-10-01T12:00:00Z", voiceChecked: true, method: "cue-window" } }); + assert.deepEqual(citationProblems(ok, ["c"]), []); + const ps = citationProblems(video({ verification: { quoteScore: 1.4, quoteCheckedAt: "yesterday" } }), ["c"]); + assert.deepEqual(paths(ps).sort(), ["c.verification.quoteCheckedAt", "c.verification.quoteScore"]); + assert.deepEqual(paths(citationProblems(video({ verification: { quoteScore: 0.5 } }), ["c"])), ["c.verification"]); + const post: Citation = { kind: "post", channel: "demo-channel", id: "1", quote: "q", verification: { voiceChecked: true } }; + assert.deepEqual(paths(citationProblems(post, ["p"])), ["p.verification.voiceChecked"]); +}); + +test("source and page citations: the source must exist, paths may not escape, URLs are http(s)", () => { + const set = structuredClone(SET) as typeof SET & { citations: Record<string, unknown> }; + set.citations.a02 = { kind: "source", source: "s9", quote: "q", image: "../outside.png" }; + set.citations.w02 = { kind: "page", url: "javascript:alert(1)", quote: "q", archiveUrl: "ftp://x" }; + (set.sources.s0 as { saved: string }).saved = "/abs/page.html"; + const ps = paths(validateCitationSet(set)); + assert.deepEqual(ps.sort(), [ + "citations.a02.image", + "citations.a02.source", + "citations.w02.archiveUrl", + "citations.w02.url", + "sources.s0.saved", + ]); +}); + +test("ids: reference ids only, and no id names both a source and a citation", () => { + const set = structuredClone(SET) as unknown as { citations: Record<string, unknown>; sources: Record<string, unknown> }; + set.citations["bad id"] = { ...SET.citations.p01 }; + set.citations.s0 = { ...SET.citations.p01 }; + const ps = validateCitationSet(set); + assert.deepEqual(paths(ps).sort(), ['citations["bad id"]', "sources.s0"]); +}); + +test("relative paths: relative, forward slashes, inside the directory", () => { + assert.equal(relativePathProblem("stills/a01.png"), null); + for (const p of ["/etc/x", "C:/x", "a\\b", "../x", "a/../../x", "a//b", "./a", "https://x/y", "", "a/"]) { + assert.notEqual(relativePathProblem(p), null, p); + } +}); + +test("dates: partial dates and zoned date-times, real calendar days only", () => { + for (const d of ["2024", "2024-02", "2024-02-29", "2026-10-01T12:00Z", "2026-10-01T12:00:00.5+02:00"]) { + assert.ok(isPartialDate(d), d); + } + for (const d of ["2023-02-29", "2024-13", "24-01-01", "2026-10-01T12:00", "October 2026"]) { + assert.ok(!isPartialDate(d), d); + } +}); + +test("jsonPath: plain keys dotted, other keys quoted, indexes bracketed", () => { + assert.equal(jsonPath(["sections", 0, "claims", 2, "findings"]), "sections[0].claims[2].findings"); + assert.equal(jsonPath(["citations", "c.1", "end"]), 'citations["c.1"].end'); + assert.equal(jsonPath([]), ""); +}); + +// ─── moments ─── + +test("moment keys: video and audio by span, post by id, none for source and page", () => { + const set = parseCitationSet(SET); + assert.ok(set.ok); + const c = set.value.citations; + assert.equal(momentKeyOf(c.c01), "demo-channel/-abc123/1.50-9.25"); + assert.equal(momentKeyOf(c.c02), "demo-podcast/ep42/60.00-75.00"); + assert.equal(momentKeyOf(c.p01), "demo-channel/1234567890"); + assert.equal(momentKeyOf(c.a01), null); + assert.equal(momentKeyOf(c.w01), null); + assert.equal(momentPath(momentOf(c.c01)!), "/m/demo-channel/-abc123/1.50-9.25/"); + assert.equal(momentPath("demo-channel/1234567890"), "/m/demo-channel/1234567890/"); +}); + +test("moment keys round-trip, an id starting with - included; the pad is not part of the key", () => { + for (const key of ["demo-channel/-abc123/0.00-12.34", "demo-channel/abc_123/3600.10-3610.00", "demo-channel/-1"]) { + const m = parseMomentKey(key); + assert.ok(m, key); + assert.equal(momentKey(m!), key); + assert.deepEqual(parseMomentPath(momentPath(m!)), m); + assert.deepEqual(parseMomentPath(`/m/${key}`), m); + } + assert.equal(momentKeyOf(video({ pad: { before: 3 } })), momentKeyOf(video())); + // Rounded to the key's precision, and the moment carries the rounded numbers. + const m = momentOf(video({ start: 1.234, end: 5.678 })); + assert.deepEqual(m, { kind: "span", channel: "demo-channel", id: "abc123", start: 1.23, end: 5.68 }); +}); + +test("moment keys: only the canonical spelling parses; unsafe segments never key", () => { + for (const key of [ + "demo-channel/abc/1.5-2.00", + "demo-channel/abc/01.00-2.00", + "demo-channel/abc/2.00-1.00", + "demo-channel/abc/1.00-1.00", + "demo-channel/abc/-1.00-2.00", + "demo-channel/../1.00-2.00", + "demo-channel", + "demo-channel/abc/1.00-2.00/x", + "-x/abc", + ]) { + assert.equal(parseMomentKey(key), null, key); + } + assert.equal(parseMomentPath("/reports/x/"), null); + assert.throws(() => momentKey({ kind: "post", channel: "demo-channel", id: "a/b" }), /safe path segment/); + assert.equal(momentKeyOf(video({ id: ".." })), null); + assert.ok(isSafeIdSegment("-abc") && !isSafeChannelSegment("-abc")); +}); + +// ─── inline citations ─── + +test("extractCiteRefs: every cite link in order, with labels and offsets; other links ignored", () => { + const md = "He said so [here](cite:c01) and [again](cite:c02), see [the site](https://example.test) and [x]( cite:c01 )."; + const refs = extractCiteRefs(md); + assert.deepEqual(refs.map((r) => [r.id, r.label]), [["c01", "here"], ["c02", "again"], ["c01", "x"]]); + assert.equal(md.slice(refs[0].offset, refs[0].offset + 6), "[here]"); + assert.deepEqual(extractCiteRefs(undefined), []); + assert.deepEqual(extractCiteRefs("[empty](cite:)").map((r) => r.id), [""]); +}); + +test("extractCiteRefs: a cite link in code is text about the syntax", () => { + const md = [ + "Write `[label](cite:id)` to cite, like [this](cite:c01).", + "```md", + "[in a fence](cite:c09)", + "```", + "~~~~", + "[tilde fence](cite:c08)", + "~~~~", + "And ``[double](cite:c07)`` too, then [after](cite:c02).", + ].join("\n"); + assert.deepEqual(extractCiteRefs(md).map((r) => r.id), ["c01", "c02"]); +}); + +test("numberCitations: by first appearance, a repeat keeps its number", () => { + assert.deepEqual([...numberCitations(["c03", "c01", "c03", "c02", "c01"])], [["c03", 1], ["c01", 2], ["c02", 3]]); + assert.equal(citeHref("c01"), "cite:c01"); + assert.equal(citationAnchor("c01"), "c-c01"); +}); diff --git a/common/lib/citations/inline.ts b/common/lib/citations/inline.ts @@ -0,0 +1,92 @@ +// INLINE CITATIONS — the one syntax for citing in markdown, and the numbers a +// reader sees. +// +// [label](cite:<id>) +// +// anywhere a document's markdown is (a report's summary, a section's body, a +// claim's findings). `<id>` names a citation in the same document. A renderer +// replaces the link with a numbered marker that previews the citation and +// jumps to it; the citation's anchor is `#c-<id>` (`citationAnchor`), stable +// across edits because it is the id, not the number. +// +// NUMBERS ARE PER DOCUMENT, BY FIRST APPEARANCE: the first citation cited is +// [1], the next new one [2], and a citation cited again keeps its number. The +// document decides the order it is read in (lib/report/uses.ts walks a +// report); `numberCitations` only counts. +// +// Code is not citing: a `cite:` link inside an inline code span or a fenced +// block is text about the syntax, and is skipped. +// +// Pure, no imports: the export site's pages and the browser can use it. + +export const CITE_SCHEME = "cite:"; + +export type CiteRef = { + id: string; + label: string; + // Where the link starts in the markdown (a UTF-16 offset). + offset: number; +}; + +// `[label](cite:id)`. The label may not contain `]`; the id runs to the first +// `)` or space (an empty id is still a ref, so validation can name it). +const CITE_LINK_RE = /\[([^\]]*)\]\(\s*cite:([^)\s]*)\s*\)/g; + +// Fenced blocks: a line opening with ``` or ~~~ (up to three spaces in) to +// the line closing it with at least as many of the same, or the end. +const FENCE_OPEN_RE = /^ {0,3}(`{3,}|~{3,})/; + +// An inline code span: a run of backticks, to the next run of the same length. +const CODE_SPAN_RE = /(?<!`)(`+)(?!`)[\s\S]*?(?<!`)\1(?!`)/g; + +const blank = (s: string) => s.replace(/[^\n]/g, " "); + +// The markdown with every fenced block and code span blanked to spaces, so +// offsets still point into the original. +function blankCode(md: string): string { + const lines = md.split("\n"); + let fence: string | null = null; + for (let i = 0; i < lines.length; i++) { + if (fence) { + const close = new RegExp(`^ {0,3}${fence[0] === "`" ? "`" : "~"}{${fence.length},}\\s*$`); + if (close.test(lines[i])) fence = null; + lines[i] = blank(lines[i]); + continue; + } + const open = FENCE_OPEN_RE.exec(lines[i]); + if (open) { + fence = open[1]; + lines[i] = blank(lines[i]); + } + } + return lines.join("\n").replace(CODE_SPAN_RE, blank); +} + +// Every `cite:` link in the markdown, in order. +export function extractCiteRefs(md: string | null | undefined): CiteRef[] { + if (!md) return []; + const scan = blankCode(md); + const out: CiteRef[] = []; + for (const m of scan.matchAll(CITE_LINK_RE)) { + out.push({ id: m[2], label: md.slice(m.index + 1, m.index + 1 + m[1].length), offset: m.index }); + } + return out; +} + +// Numbers by first appearance: each id's number, from 1, in the order the ids +// are given; a repeated id keeps its first number. +export function numberCitations(ids: Iterable<string>): Map<string, number> { + const out = new Map<string, number>(); + for (const id of ids) if (!out.has(id)) out.set(id, out.size + 1); + return out; +} + +// The href an inline citation is written with. +export function citeHref(id: string): string { + return `${CITE_SCHEME}${id}`; +} + +// The anchor (fragment id, without `#`) of a citation's entry in a document. +export function citationAnchor(id: string): string { + return `c-${id}`; +} diff --git a/common/lib/citations/moments.ts b/common/lib/citations/moments.ts @@ -0,0 +1,151 @@ +// MOMENT KEYS — the one name of a cited moment, and the URL of its page. +// +// span (video, audio) <channel>/<id>/<start>-<end> page /m/<channel>/<id>/<start>-<end>/ +// post <channel>/<id> page /m/<channel>/<id>/ +// +// `source` and `page` citations have no moment: they are shown where they are +// cited, never on a page of their own. +// +// THE KEY IS THE PAGE, so it must be a function of the cited numbers alone: +// `start` and `end` are written with TWO DECIMALS, always (`Number#toFixed(2)`, +// the same precision as a clip window's name, lib/clipWindow.ts), and a parse +// accepts only that spelling — `12.5` is not a key, `12.50` is — so one moment +// has exactly one URL. Two citations of the same span share a page; the pad is +// not part of the key (it widens the clip, not the moment). Consumers that cut +// or show the span use the ROUNDED numbers (`momentOf`), so the page, the clip +// file and the key agree. +// +// SLUG SAFETY: the channel and the id are path segments of a page that a static +// build writes to disk, so each must be one safe segment — no `/`, no `\`, not +// `.` or `..`, nothing outside `[A-Za-z0-9._-]`. A video id MAY START WITH `-` +// (YouTube ids do); that is a safe path segment, but such an id must never be +// handed to a command line as a bare argument. `momentKey` throws on an unsafe +// segment rather than build a path from it; `momentKeyOf` answers null. +// +// Pure, no imports but types: the export site's pages and the browser can use it. + +import type { Citation } from "./schema"; + +export const MOMENT_SECONDS_DECIMALS = 2; + +// The route prefix of every moment page. +export const MOMENT_ROUTE_PREFIX = "/m/"; + +// A channel slug: the corpus's own rule (controller/channels.ts +// CHANNEL_SLUG_RE, which lib may not import), capped in length. +const CHANNEL_SEGMENT_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$/; + +// A record id: like a slug, but it may also start with `-` or `_`. +const ID_SEGMENT_RE = /^[A-Za-z0-9_-][A-Za-z0-9._-]{0,127}$/; + +// A span segment as a key spells it: two decimals each, no sign, no leading +// zeros but the one before the point. +const SPAN_SEGMENT_RE = /^(0|[1-9]\d*)\.(\d{2})-(0|[1-9]\d*)\.(\d{2})$/; + +export function isSafeChannelSegment(v: unknown): v is string { + return typeof v === "string" && v !== ".." && CHANNEL_SEGMENT_RE.test(v); +} + +export function isSafeIdSegment(v: unknown): v is string { + return typeof v === "string" && v !== "." && v !== ".." && ID_SEGMENT_RE.test(v); +} + +export type SpanMoment = { kind: "span"; channel: string; id: string; start: number; end: number }; +export type PostMoment = { kind: "post"; channel: string; id: string }; +export type Moment = SpanMoment | PostMoment; + +// Seconds as a key spells them. +export function formatMomentSeconds(s: number): string { + return s.toFixed(MOMENT_SECONDS_DECIMALS); +} + +// Seconds rounded to the key's precision, as a number. +export function roundMomentSeconds(s: number): number { + return Number(formatMomentSeconds(s)); +} + +// The moment a citation opens, its span rounded to the key's precision; null +// for a kind without a page (source, page). +export function momentOf(c: Citation): Moment | null { + switch (c.kind) { + case "video": + case "audio": + return { + kind: "span", + channel: c.channel, + id: c.id, + start: roundMomentSeconds(c.start), + end: roundMomentSeconds(c.end), + }; + case "post": + return { kind: "post", channel: c.channel, id: c.id }; + default: + return null; + } +} + +// Why a moment cannot be keyed, as a sentence, or null when it can. +export function momentProblem(m: Moment): string | null { + if (!isSafeChannelSegment(m.channel)) return `channel ${JSON.stringify(m.channel)} is not a safe path segment`; + if (!isSafeIdSegment(m.id)) return `id ${JSON.stringify(m.id)} is not a safe path segment`; + if (m.kind === "span") { + if (!Number.isFinite(m.start) || !Number.isFinite(m.end) || m.start < 0) { + return "a span's start and end must be numbers, the start at least 0"; + } + if (!(roundMomentSeconds(m.end) > roundMomentSeconds(m.start))) { + return `the span ends (${formatMomentSeconds(m.end)}) at or before it starts (${formatMomentSeconds(m.start)}) at the key's precision`; + } + } + return null; +} + +// The moment's key. Throws on a moment `momentProblem` refuses. +export function momentKey(m: Moment): string { + const problem = momentProblem(m); + if (problem) throw new Error(`moment key: ${problem}`); + const base = `${m.channel}/${m.id}`; + return m.kind === "span" ? `${base}/${formatMomentSeconds(m.start)}-${formatMomentSeconds(m.end)}` : base; +} + +// The citation's moment key, or null: a kind without a page, or a citation +// whose moment cannot be keyed (validation names why). +export function momentKeyOf(c: Citation): string | null { + const m = momentOf(c); + if (!m || momentProblem(m)) return null; + return momentKey(m); +} + +// The page of a moment (or of a key): `/m/<key>/`. +export function momentPath(m: Moment | string): string { + return `${MOMENT_ROUTE_PREFIX}${typeof m === "string" ? m : momentKey(m)}/`; +} + +// A key back into its moment, or null for anything that is not a key exactly +// as `momentKey` spells it. +export function parseMomentKey(key: string): Moment | null { + const parts = key.split("/"); + if (parts.length === 2) { + const [channel, id] = parts; + if (!isSafeChannelSegment(channel) || !isSafeIdSegment(id)) return null; + return { kind: "post", channel, id }; + } + if (parts.length === 3) { + const [channel, id, spanPart] = parts; + if (!isSafeChannelSegment(channel) || !isSafeIdSegment(id)) return null; + const m = SPAN_SEGMENT_RE.exec(spanPart); + if (!m) return null; + const start = Number(`${m[1]}.${m[2]}`); + const end = Number(`${m[3]}.${m[4]}`); + if (!(end > start)) return null; + return { kind: "span", channel, id, start, end }; + } + return null; +} + +// A moment page's path (`/m/<key>/`, the trailing slash optional) back into +// its moment, or null. +export function parseMomentPath(p: string): Moment | null { + if (!p.startsWith(MOMENT_ROUTE_PREFIX)) return null; + const key = p.slice(MOMENT_ROUTE_PREFIX.length).replace(/\/$/, ""); + return parseMomentKey(key); +} diff --git a/common/lib/citations/schema.ts b/common/lib/citations/schema.ts @@ -0,0 +1,268 @@ +// THE CITATION MODEL — one definition of what a citation is, for every document +// that cites the corpus: a report (`report.json`, lib/report/), a standalone +// citation set (`archilyzer-citations`, below — what a converter of a sweep, an +// /ask answer or a video manifest emits), and whatever comes next. +// +// A citation is a discriminated union on `kind`: +// +// video a span of a video record: channel, id, start, end, pad +// audio a span of an audio record (a podcast): the same fields as video; +// rendered with a poster instead of a picture +// post a post record: channel, id, thread +// source a sentence of a document under review: the source's id in +// `sources`, a still of the sentence (`image`); rendered with that +// source's archive links +// page a web page that is not in the corpus: url, title, archiveUrl +// +// and every kind carries the common fields: a VERBATIM `quote`, and optional +// `speaker`, `date`, `label`, `note`, and `verification` — the one block that is +// COMPUTED (filled by compose when it checks the quote against the cues, the +// voice against the speaker), never typed by hand; the validator flags a block +// whose values could not have come from a check. +// +// EXTENDING THE UNION: a new kind is one more member schema in CITATION_KINDS, +// its *_FIELD_DOCS record (CITATIONS.md is generated from them), its moment +// shape in ./moments.ts when it has a page of its own, and its rules in +// ./validate.ts. A reader of an older version refuses a kind it does not know +// (the union is closed), which is why the containers carry a `version`. +// +// THE SPLIT, and why: zod here checks SHAPE only — types, required keys, +// literal kinds, no unknown keys. Every rule about VALUES (ids, spans, paths, +// dates, URLs, references between citations and sources) is ./validate.ts's, +// which runs after a successful parse and reports EVERY problem with its JSON +// path, instead of the first shape error hiding the rest. +// +// SERVER-ONLY: zod. A client importer takes the types with `import type`; the +// pure helpers (./moments.ts, ./inline.ts) import nothing from here but types. + +import { z } from "zod"; +import type { FieldDocs } from "../fieldDocs"; + +// The version of the citation model a container declares. Bumped when a reader +// of the old version would misread a document of the new one. +export const CITATIONS_VERSION = 1; + +// The longest span a video or audio citation may cut, pad included, in seconds. +// A moment page carries one self-hosted evidence clip of it; this keeps every +// clip a citation (not a re-upload) and well under a static host's file limit. +export const MAX_CITATION_SPAN_SECONDS = 120; + +const text = z.string(); + +export const verificationSchema = z.strictObject({ + quoteScore: z.number().optional(), + quoteCheckedAt: text.optional(), + voiceChecked: z.boolean().optional(), + method: text.optional(), +}); + +export type CitationVerification = z.infer<typeof verificationSchema>; + +export const CITATION_VERIFICATION_FIELD_DOCS: FieldDocs<CitationVerification> = { + quoteScore: + "How closely `quote` matches what the record says at the cited place, from 0 (nothing alike) to 1 (verbatim). Written by the check, with `quoteCheckedAt`.", + quoteCheckedAt: "When the quote was checked: an ISO 8601 date-time. Written with `quoteScore`.", + voiceChecked: + "True when the speaker's voice in the cited span was checked against `speaker`. Only a span (video, audio) has a voice to check.", + method: "What did the checking, e.g. the cue-window comparison and its version. Free text, one line.", +}; + +const common = { + quote: text, + speaker: text.optional(), + date: text.optional(), + label: text.optional(), + note: text.optional(), + verification: verificationSchema.optional(), +}; + +const pad = z.strictObject({ + before: z.number().optional(), + after: z.number().optional(), +}); + +const span = { + channel: text, + id: text, + start: z.number(), + end: z.number(), + pad: pad.optional(), +}; + +export const videoCitationSchema = z.strictObject({ kind: z.literal("video"), ...span, ...common }); +export const audioCitationSchema = z.strictObject({ kind: z.literal("audio"), ...span, ...common }); +export const postCitationSchema = z.strictObject({ + kind: z.literal("post"), + channel: text, + id: text, + thread: z.boolean().optional(), + ...common, +}); +export const sourceCitationSchema = z.strictObject({ + kind: z.literal("source"), + source: text, + image: text.optional(), + ...common, +}); +export const pageCitationSchema = z.strictObject({ + kind: z.literal("page"), + url: text, + title: text.optional(), + archiveUrl: text.optional(), + ...common, +}); + +// Every member, in the order CITATIONS.md documents them. +export const CITATION_KINDS = [ + videoCitationSchema, + audioCitationSchema, + postCitationSchema, + sourceCitationSchema, + pageCitationSchema, +] as const; + +export const citationSchema = z.discriminatedUnion("kind", [...CITATION_KINDS]); + +export type VideoCitation = z.infer<typeof videoCitationSchema>; +export type AudioCitation = z.infer<typeof audioCitationSchema>; +export type PostCitation = z.infer<typeof postCitationSchema>; +export type SourceCitation = z.infer<typeof sourceCitationSchema>; +export type PageCitation = z.infer<typeof pageCitationSchema>; +export type Citation = z.infer<typeof citationSchema>; +export type CitationKind = Citation["kind"]; +// The kinds that cite a span of a record's media: a clip, a start and an end. +export type SpanCitation = VideoCitation | AudioCitation; +export type CitationPad = z.infer<typeof pad>; + +export const SPAN_KINDS: readonly CitationKind[] = ["video", "audio"]; + +export function isSpanCitation(c: Citation): c is SpanCitation { + return c.kind === "video" || c.kind === "audio"; +} + +export type CitationCommon = Pick<VideoCitation, keyof typeof common>; + +export const CITATION_COMMON_FIELD_DOCS: FieldDocs<CitationCommon> = { + quote: + "The cited words, VERBATIM — as the record says them (a span's cues, a post's text, the source's sentence, the page's text). Never a paraphrase: compose checks a span's quote against its cues and fails on drift.", + speaker: "Who says the quote, when that is not the record's own channel (a guest, a co-host, a caller).", + date: "When the quote was said or written: `YYYY`, `YYYY-MM`, `YYYY-MM-DD` or an ISO 8601 date-time. Absent = the record's own date.", + label: "A short display name for the citation (one line), used where its number alone would be too little.", + note: "An editorial note shown with the citation (plain text): context the quote needs.", + verification: + "COMPUTED, not authored: what checking this citation found, written by compose. A hand-typed block that could not have come from a check (a score outside 0–1, a score without its time) is a validation problem.", +}; + +type Own<T> = Omit<T, keyof CitationCommon>; + +export const VIDEO_CITATION_FIELD_DOCS: FieldDocs<Own<VideoCitation>> = { + kind: '`"video"`.', + channel: "The record's channel slug (its directory under `transcripts/channels/`). A path segment of the moment page.", + id: "The record's video id (its directory under `data/`; may start with `-`). A path segment of the moment page.", + start: "Where the cited span starts, in seconds from the start of the record.", + end: "Where the cited span ends, in seconds; after `start`.", + pad: "Context around the span in the evidence clip, in seconds. Absent = none. The span plus its pad is at most 120 s.", +}; + +export const AUDIO_CITATION_FIELD_DOCS: FieldDocs<Own<AudioCitation>> = { + ...VIDEO_CITATION_FIELD_DOCS, + kind: '`"audio"`: a span of a record with no picture worth showing (a podcast). Rendered with a poster.', +}; + +export const CITATION_PAD_FIELD_DOCS: FieldDocs<CitationPad> = { + before: "Seconds of context before `start`; ≥ 0. Absent = 0.", + after: "Seconds of context after `end`; ≥ 0. Absent = 0.", +}; + +export const POST_CITATION_FIELD_DOCS: FieldDocs<Own<PostCitation>> = { + kind: '`"post"`.', + channel: "The post's channel slug. A path segment of the moment page.", + id: "The post's id. A path segment of the moment page.", + thread: "True to show the post with the thread it belongs to. Absent = the post alone.", +}; + +export const SOURCE_CITATION_FIELD_DOCS: FieldDocs<Own<SourceCitation>> = { + kind: '`"source"`: a sentence of a document under review.', + source: "The id of the document in the container's `sources`. Its archive links are shown with the citation.", + image: + "A still of the sentence as the document shows it: a path relative to the container's directory (e.g. `stills/a01.png`), never absolute, never leaving it.", +}; + +export const PAGE_CITATION_FIELD_DOCS: FieldDocs<Own<PageCitation>> = { + kind: '`"page"`: a web page outside the corpus.', + url: "The page's address: an http(s) URL.", + title: "The page's title.", + archiveUrl: "An archived copy of the page (an http(s) URL), shown beside the live link.", +}; + +// ─── Sources: the documents a `source` citation quotes ─── + +export const SOURCE_KINDS = ["article", "page", "video", "post", "document", "other"] as const; + +export const sourceArchiveSchema = z.strictObject({ + label: text, + url: text, + context: text.optional(), +}); + +export const sourceSchema = z.strictObject({ + kind: z.enum(SOURCE_KINDS), + title: text, + url: text.optional(), + publisher: text.optional(), + author: text.optional(), + date: text.optional(), + archives: z.array(sourceArchiveSchema).optional(), + saved: text.optional(), +}); + +export type Source = z.infer<typeof sourceSchema>; +export type SourceArchive = z.infer<typeof sourceArchiveSchema>; + +export const SOURCE_FIELD_DOCS: FieldDocs<Source> = { + kind: `What the document is: ${SOURCE_KINDS.map((k) => `\`${k}\``).join(", ")}.`, + title: "The document's title.", + url: "Where the document lives: an http(s) URL.", + publisher: "Who published it (the outlet, the site).", + author: "Who wrote it.", + date: "When it was published: `YYYY`, `YYYY-MM`, `YYYY-MM-DD` or an ISO 8601 date-time.", + archives: "The document's archive links, in context — as the document had them. Shown with every citation of it.", + saved: + "A saved copy of the document, relative to the container's directory (e.g. `sources/s0/page.html`): the input the stills are shot from. NEVER published.", +}; + +export const SOURCE_ARCHIVE_FIELD_DOCS: FieldDocs<SourceArchive> = { + label: "The link's text.", + url: "The archived copy: an http(s) URL.", + context: "The words around the link in the document, so a reader sees what it was offered as.", +}; + +// ─── A standalone citation set ─── + +export const CITATION_SET_FORMAT = "archilyzer-citations"; + +export const citationSetSchema = z.strictObject({ + format: z.literal(CITATION_SET_FORMAT), + version: z.literal(CITATIONS_VERSION), + sources: z.record(z.string(), sourceSchema).optional(), + citations: z.record(z.string(), citationSchema), +}); + +export type CitationSet = z.infer<typeof citationSetSchema>; + +export const CITATION_SET_FIELD_DOCS: FieldDocs<CitationSet> = { + format: `\`"${CITATION_SET_FORMAT}"\`.`, + version: `\`${CITATIONS_VERSION}\`.`, + sources: "The documents the `source` citations quote, by id. Absent = none.", + citations: "The citations, by id. An id is what a `[label](cite:<id>)` link names.", +}; + +// A reference id: a citation's, a source's, a section's or a claim's. Letters, +// digits and `_ . : -`, starting with a letter or digit, at most 64 — the same +// rule umtool's fact-check claim ids follow. An id is used in a URL fragment +// (`#c-<id>`), never as a path segment. +export const REF_ID_RE = /^[A-Za-z0-9][A-Za-z0-9_.:-]{0,63}$/; + +export function isRefId(v: unknown): v is string { + return typeof v === "string" && REF_ID_RE.test(v); +} diff --git a/common/lib/citations/validate.ts b/common/lib/citations/validate.ts @@ -0,0 +1,295 @@ +// CITATION VALIDATION — every rule about a citation's VALUES, as a list of +// problems with JSON paths. Never throws. +// +// ./schema.ts checks shape (zod); this checks what shape cannot: ids, spans, +// path safety, dates and URLs, references from a citation to its source, and a +// computed `verification` block that could not have come from a check. A +// document validator (lib/report/validate.ts, `validateCitationSet` below) +// runs zod first, maps its issues to the same problem list, and runs these only +// on a document that parsed — so a reader gets every value problem at once. + +import type { z } from "zod"; +import { + MAX_CITATION_SPAN_SECONDS, + citationSetSchema, + isRefId, + isSpanCitation, + type Citation, + type CitationSet, + type CitationVerification, + type Source, +} from "./schema"; +import { formatMomentSeconds, isSafeChannelSegment, isSafeIdSegment, roundMomentSeconds } from "./moments"; + +export type Problem = { + // Where, as a JSON path from the document's root: `citations.c01.end`, + // `sections[0].claims[2].findings`; `""` for the root. + path: string; + message: string; +}; + +export type PathSegment = string | number; + +const IDENT_RE = /^[A-Za-z_$][A-Za-z0-9_$]*$/; + +// A JSON path: `.key` for a plain key, `["k.y"]` for any other, `[i]` for an +// index. +export function jsonPath(segs: readonly PathSegment[]): string { + let out = ""; + for (const s of segs) { + if (typeof s === "number") out += `[${s}]`; + else if (IDENT_RE.test(s)) out += out ? `.${s}` : s; + else out += `[${JSON.stringify(s)}]`; + } + return out; +} + +export function problem(at: readonly PathSegment[], message: string): Problem { + return { path: jsonPath(at), message }; +} + +// zod's issues as problems, under `at`. +export function zodProblems(error: z.ZodError, at: readonly PathSegment[] = []): Problem[] { + return error.issues.map((i) => + problem([...at, ...i.path.map((p) => (typeof p === "number" ? p : String(p)))], i.message), + ); +} + +const blank = (s: string) => !/\S/.test(s); + +// `YYYY`, `YYYY-MM`, `YYYY-MM-DD`, or a date-time with a zone. +const PARTIAL_DATE_RE = /^(\d{4})(?:-(\d{2})(?:-(\d{2}))?)?$/; +const DATE_TIME_RE = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2}(?:\.\d+)?)?(?:Z|[+-]\d{2}:\d{2})$/; + +export function isDateTime(v: string): boolean { + return DATE_TIME_RE.test(v) && Number.isFinite(Date.parse(v)) && isPartialDate(v.slice(0, 10)); +} + +export function isPartialDate(v: string): boolean { + if (DATE_TIME_RE.test(v)) return isDateTime(v); + const m = PARTIAL_DATE_RE.exec(v); + if (!m) return false; + if (m[2] === undefined) return true; + const month = Number(m[2]); + if (month < 1 || month > 12) return false; + if (m[3] === undefined) return true; + const day = Number(m[3]); + const days = new Date(Date.UTC(Number(m[1]), month, 0)).getUTCDate(); + return day >= 1 && day <= days; +} + +export function isHttpUrl(v: string): boolean { + try { + const u = new URL(v); + return (u.protocol === "http:" || u.protocol === "https:") && !!u.hostname; + } catch { + return false; + } +} + +// Why `p` is not a safe path relative to a document's directory, or null: it +// must be relative, `/`-separated, with no empty, `.` or `..` segment — so it +// can neither start elsewhere nor climb out. +export function relativePathProblem(p: string): string | null { + if (blank(p)) return "is empty"; + if (p.includes("\\")) return "must use `/`, not `\\`"; + if (p.startsWith("/") || /^[A-Za-z]:/.test(p)) return "must be relative, not absolute"; + if (/^[A-Za-z][A-Za-z0-9+.-]*:/.test(p)) return "must be a path, not a URL"; + if (p.split("/").some((s) => s === "" || s === "." || s === "..")) { + return "must not have empty, `.` or `..` segments (it may not leave the document's directory)"; + } + if (/[\x00-\x1f]/.test(p)) return "must not contain control characters"; + return null; +} + +function textProblems( + out: Problem[], + at: readonly PathSegment[], + value: string | undefined, + { required = false, oneLine = false }: { required?: boolean; oneLine?: boolean } = {}, +): void { + if (value === undefined) return; + if (blank(value)) { + out.push(problem(at, required ? "must not be blank" : "must not be blank (leave it out instead)")); + return; + } + if (oneLine && /[\r\n]/.test(value)) out.push(problem(at, "must be one line")); +} + +function dateProblems(out: Problem[], at: readonly PathSegment[], value: string | undefined): void { + if (value === undefined) return; + if (!isPartialDate(value)) { + out.push(problem(at, "must be a date: YYYY, YYYY-MM, YYYY-MM-DD or an ISO 8601 date-time with a zone")); + } +} + +function urlProblems(out: Problem[], at: readonly PathSegment[], value: string | undefined): void { + if (value === undefined) return; + if (!isHttpUrl(value)) out.push(problem(at, "must be an http(s) URL")); +} + +function pathProblems(out: Problem[], at: readonly PathSegment[], value: string | undefined): void { + if (value === undefined) return; + const why = relativePathProblem(value); + if (why) out.push(problem(at, why)); +} + +function verificationProblems( + out: Problem[], + at: readonly PathSegment[], + v: CitationVerification, + c: Citation, +): void { + if (v.quoteScore !== undefined && !(v.quoteScore >= 0 && v.quoteScore <= 1)) { + out.push(problem([...at, "quoteScore"], "must be from 0 to 1 — a check writes it; it is not typed by hand")); + } + if (v.quoteCheckedAt !== undefined && !isDateTime(v.quoteCheckedAt)) { + out.push(problem([...at, "quoteCheckedAt"], "must be an ISO 8601 date-time with a zone")); + } + if ((v.quoteScore === undefined) !== (v.quoteCheckedAt === undefined)) { + out.push( + problem( + at, + "quoteScore and quoteCheckedAt are written together by the check — one without the other was not", + ), + ); + } + if (v.voiceChecked !== undefined && !isSpanCitation(c)) { + out.push(problem([...at, "voiceChecked"], `a ${c.kind} citation has no voice to check`)); + } + textProblems(out, [...at, "method"], v.method, { oneLine: true }); +} + +export type CitationContext = { + // The container's sources, for a `source` citation's reference. + sources?: Readonly<Record<string, Source>>; +}; + +// Every problem with one citation, under `at` (its path in the document). +export function citationProblems( + c: Citation, + at: readonly PathSegment[], + ctx: CitationContext = {}, +): Problem[] { + const out: Problem[] = []; + textProblems(out, [...at, "quote"], c.quote, { required: true }); + textProblems(out, [...at, "speaker"], c.speaker, { oneLine: true }); + textProblems(out, [...at, "label"], c.label, { oneLine: true }); + textProblems(out, [...at, "note"], c.note); + dateProblems(out, [...at, "date"], c.date); + if (c.verification) verificationProblems(out, [...at, "verification"], c.verification, c); + + switch (c.kind) { + case "video": + case "audio": { + if (!isSafeChannelSegment(c.channel)) { + out.push(problem([...at, "channel"], "must be a channel slug (one safe path segment)")); + } + if (!isSafeIdSegment(c.id)) { + out.push(problem([...at, "id"], "must be a record id (one safe path segment: letters, digits, `.`, `_`, `-`)")); + } + const before = c.pad?.before ?? 0; + const after = c.pad?.after ?? 0; + if (c.pad?.before !== undefined && c.pad.before < 0) out.push(problem([...at, "pad", "before"], "must be at least 0")); + if (c.pad?.after !== undefined && c.pad.after < 0) out.push(problem([...at, "pad", "after"], "must be at least 0")); + if (c.start < 0) out.push(problem([...at, "start"], "must be at least 0")); + if (!(c.end > c.start)) { + out.push(problem([...at, "end"], `must be after start (${c.start})`)); + } else if (!(roundMomentSeconds(c.end) > roundMomentSeconds(c.start))) { + out.push( + problem( + [...at, "end"], + `rounds to the start (${formatMomentSeconds(c.start)}): a span is at least 0.01 s at the moment key's precision`, + ), + ); + } else { + const total = c.end - c.start + Math.max(0, before) + Math.max(0, after); + if (total > MAX_CITATION_SPAN_SECONDS) { + out.push( + problem( + at, + `the span with its pad is ${Math.round(total * 100) / 100} s (at most ${MAX_CITATION_SPAN_SECONDS} s)`, + ), + ); + } + } + break; + } + case "post": + if (!isSafeChannelSegment(c.channel)) { + out.push(problem([...at, "channel"], "must be a channel slug (one safe path segment)")); + } + if (!isSafeIdSegment(c.id)) { + out.push(problem([...at, "id"], "must be a record id (one safe path segment: letters, digits, `.`, `_`, `-`)")); + } + break; + case "source": + if (!ctx.sources || !Object.hasOwn(ctx.sources, c.source)) { + out.push(problem([...at, "source"], `names no source (${JSON.stringify(c.source)} is not in sources)`)); + } + pathProblems(out, [...at, "image"], c.image); + break; + case "page": + urlProblems(out, [...at, "url"], c.url); + urlProblems(out, [...at, "archiveUrl"], c.archiveUrl); + textProblems(out, [...at, "title"], c.title, { oneLine: true }); + break; + } + return out; +} + +// Every problem with one source, under `at`. +export function sourceProblems(s: Source, at: readonly PathSegment[]): Problem[] { + const out: Problem[] = []; + textProblems(out, [...at, "title"], s.title, { required: true, oneLine: true }); + urlProblems(out, [...at, "url"], s.url); + textProblems(out, [...at, "publisher"], s.publisher, { oneLine: true }); + textProblems(out, [...at, "author"], s.author, { oneLine: true }); + dateProblems(out, [...at, "date"], s.date); + (s.archives ?? []).forEach((a, i) => { + textProblems(out, [...at, "archives", i, "label"], a.label, { required: true, oneLine: true }); + urlProblems(out, [...at, "archives", i, "url"], a.url); + textProblems(out, [...at, "archives", i, "context"], a.context); + }); + pathProblems(out, [...at, "saved"], s.saved); + return out; +} + +// Every problem with a container's `sources` and `citations` maps: each id is a +// reference id, no id names both a source and a citation (a `cite:` link names +// citations only — one id meaning two things is a trap), and every entry's own +// problems. +export function citationMapProblems( + citations: Readonly<Record<string, Citation>>, + sources: Readonly<Record<string, Source>> | undefined, + at: readonly PathSegment[] = [], +): Problem[] { + const out: Problem[] = []; + for (const [id, s] of Object.entries(sources ?? {})) { + const where = [...at, "sources", id]; + if (!isRefId(id)) out.push(problem(where, "is not a reference id (letters, digits, `_ . : -`; at most 64)")); + if (Object.hasOwn(citations, id)) out.push(problem(where, "is also a citation's id — ids name one thing")); + out.push(...sourceProblems(s, where)); + } + for (const [id, c] of Object.entries(citations)) { + const where = [...at, "citations", id]; + if (!isRefId(id)) out.push(problem(where, "is not a reference id (letters, digits, `_ . : -`; at most 64)")); + out.push(...citationProblems(c, where, { sources })); + } + return out; +} + +export type Parsed<T> = { ok: true; value: T; problems: Problem[] } | { ok: false; problems: Problem[] }; + +// A standalone citation set (`archilyzer-citations`): parsed, and every problem. +// `ok` means it parsed; `problems` may still be non-empty. +export function parseCitationSet(raw: unknown): Parsed<CitationSet> { + const r = citationSetSchema.safeParse(raw); + if (!r.success) return { ok: false, problems: zodProblems(r.error) }; + return { ok: true, value: r.data, problems: citationMapProblems(r.data.citations, r.data.sources) }; +} + +// Every problem with a citation set; empty when it is sound. +export function validateCitationSet(raw: unknown): Problem[] { + return parseCitationSet(raw).problems; +}