// `archilyzer reports check ` and `archilyzer reports verify-quotes // ` — what compose would say about a site's reports, without a // build; and one report's quotes against the transcripts, track by track. // // CHECK is the reports stage of compose with nothing written: // resolveSiteReports (publish/composeReports.ts) — every report parsed and // validated, every cited quote verified against its record, every cited // moment's prepared media present and current, the report's video on disk and // under the publish limit. Its problems are compose's, word for word. With // `--reports a,b` it checks those reports (drafts included: a report need not // be in site.json yet); `--allow-missing-media` checks the text before // `reports prepare` has cut anything. // // VERIFY-QUOTES runs compose's own quote check (checkSpanQuote, the post // check) over every video, audio and post citation of ONE report.json, // published or not, and prints each one's best score and track — and, where // the record has an `en-orig` track, that track's score. A served `en` track // can be a rewrite of what was said; a quote that matches it and not // `en-orig` is not what the speaker said, and is reported as such, even // though compose (which takes the best track) would pass it. // // Exit 0 when there is nothing to report, 1 with the list, 2 for usage (an // unknown site, an unreadable file). import { readFile } from "node:fs/promises"; import path from "node:path"; import { getPaths, type Paths } from "../lib/paths"; import { getSite, listSiteIds } from "../lib/site"; import { readChannelConfig } from "../controller/channels"; import { assertChannelTextReadable } from "../lib/channelMedia"; import { readAllPosts } from "../lib/posts-server"; import { parseReport } from "../lib/report/validate"; import type { Report } from "../lib/report/schema"; import { reportCitationNumbers } from "../lib/report/uses"; import { QUOTE_DRIFT_THRESHOLD, quoteDrifted, quoteVerification, roundScore } from "../lib/citations/verify"; import { ComposeReportsError, checkSpanQuote, formatComposeReportsProblems, quoteDriftMessage, readCitedRecord, resolveSiteReports, } from "../publish/composeReports"; type Out = { log: (s: string) => void; error: (s: string) => void }; // ─── reports check ─── export async function checkMain( opts: { siteId: string; reports?: string[]; allowMissingMedia?: boolean; paths?: Paths; settings?: { social?: { x?: { visibility?: unknown } } }; }, out: Out = console, ): Promise { const paths = opts.paths ?? getPaths(); if (!listSiteIds(paths).includes(opts.siteId)) { out.error(`reports check: no site "${opts.siteId}" (sites/${opts.siteId}/site.json)`); return 2; } const site = getSite(opts.siteId, paths); const ids = opts.reports ?? site.reports ?? []; if (ids.length === 0) { out.log(`reports check ${opts.siteId}: the site publishes no reports — nothing to check.`); return 0; } try { const resolved = await resolveSiteReports({ paths, site: { ...site, reports: ids }, allowMissingMedia: opts.allowMissingMedia === true, ...(opts.settings ? { settings: opts.settings } : {}), }); for (const line of formatComposeReportsProblems(resolved.allowed)) { out.log(` allowed (--allow-missing-media): ${line}`); } for (const r of resolved.reports) { const scores = Object.values(r.citations ?? {}) .map((c) => (c.kind === "video" || c.kind === "audio" || c.kind === "post" ? c.verification?.quoteScore : undefined)) .filter((s): s is number => typeof s === "number"); const low = scores.length ? Math.min(...scores) : null; out.log( ` ${r.id}: ${scores.length} quote(s) checked` + (low === null ? "" : `, lowest ${low.toFixed(2)}`), ); } out.log( `reports check ${opts.siteId}: ${resolved.reports.length} report(s), ${resolved.moments.length} moment(s) — compose would pass.`, ); return 0; } catch (err) { if (!(err instanceof ComposeReportsError)) { out.error(`reports check ${opts.siteId}: ${(err as Error).message}`); return 1; } out.error(`reports check ${opts.siteId}: ${err.problems.length} problem(s) — compose would fail:`); for (const line of formatComposeReportsProblems(err.problems)) out.error(` ${line}`); return 1; } } // ─── reports verify-quotes ─── export type QuoteStatus = | "ok" | "drift" | "en-orig-drift" | "missing-record" | "no-cues" | "missing-post" | "unreadable"; export type QuoteResult = { citation: string; kind: "video" | "audio" | "post"; channel: string; id: string; start?: number; end?: number; cited: boolean; status: QuoteStatus; // The best score and the track it came from (a post: its text). score?: number; track?: string; // Every track's score; and the en-orig track's, when the record has one. tracks?: { name: string; score: number }[]; enOrig?: number; message?: string; }; export const EN_ORIG_TRACK = "transcript.en-orig.vtt"; // Every video, audio and post citation of a report, checked against the // corpus at `paths.channelsDir` — cited or not (an uncited one is marked). export async function verifyReportQuotes( report: Report, opts: { paths: Paths; now?: string }, ): Promise { const { paths } = opts; const now = opts.now ?? new Date().toISOString(); const used = new Set(reportCitationNumbers(report).keys()); const results: QuoteResult[] = []; const unreadable = new Map(); const textProblem = async (slug: string) => { if (!unreadable.has(slug)) { try { await assertChannelTextReadable(paths, slug, await readChannelConfig(paths, slug).catch(() => null)); unreadable.set(slug, null); } catch (e) { unreadable.set(slug, (e as Error).message); } } return unreadable.get(slug) ?? null; }; const posts = new Map>(); for (const [cid, c] of Object.entries(report.citations ?? {})) { if (c.kind !== "video" && c.kind !== "audio" && c.kind !== "post") continue; const base = { citation: cid, kind: c.kind, channel: c.channel, id: c.id, cited: used.has(cid), ...(c.kind === "post" ? {} : { start: c.start, end: c.end }), }; const text = await textProblem(c.channel); if (text) { results.push({ ...base, status: "unreadable", message: text }); continue; } if (c.kind === "post") { if (!posts.has(c.channel)) { const all = await readAllPosts(path.join(paths.channelsDir, c.channel)).catch(() => []); posts.set(c.channel, new Map(all.map((p) => [p.id, p.text]))); } const postText = posts.get(c.channel)!.get(c.id); if (postText === undefined) { results.push({ ...base, status: "missing-post", message: `no post ${c.id} in the posts archive of "${c.channel}"` }); continue; } const v = quoteVerification(c.quote, postText, now); results.push({ ...base, score: v.quoteScore, track: "post", status: quoteDrifted(v) ? "drift" : "ok", ...(quoteDrifted(v) ? { message: quoteDriftMessage(v.quoteScore) } : {}), }); continue; } const record = await readCitedRecord(paths.channelsDir, c.channel, c.id); if (!record) { results.push({ ...base, status: "missing-record", message: `no record ${c.channel}/${c.id} (no metadata or transcript in its data dir)` }); continue; } if (record.cues.length === 0) { results.push({ ...base, status: "no-cues", message: `${c.channel}/${c.id} has no transcript cues to check the quote against` }); continue; } const checked = checkSpanQuote(record, c, now); const score = checked.verification.quoteScore ?? 0; const best = checked.tracks.reduce<{ name: string; score: number } | null>( (b, t) => (!b || t.score > b.score ? t : b), null, ); const enOrig = checked.tracks.find((t) => t.name === EN_ORIG_TRACK)?.score; let status: QuoteStatus = "ok"; let message: string | undefined; if (quoteDrifted(checked.verification)) { status = "drift"; message = quoteDriftMessage(score); } else if (enOrig !== undefined && enOrig < QUOTE_DRIFT_THRESHOLD) { status = "en-orig-drift"; message = `the quote matches ${best?.name ?? "a track"} (${score.toFixed(2)}) but only ${Math.round(enOrig * 100)}% of ` + `the en-orig track, the words as spoken — a served \`en\` track can rewrite them: quote en-orig, or check the audio`; } results.push({ ...base, score: roundScore(score), ...(best ? { track: best.name } : {}), tracks: checked.tracks, ...(enOrig !== undefined ? { enOrig } : {}), status, ...(message ? { message } : {}), }); } return results; } function describe(r: QuoteResult): string { const where = r.kind === "post" ? `${r.channel}/${r.id}` : `${r.channel}/${r.id} ${r.start}–${r.end} s`; const score = r.score === undefined ? "" : ` ${r.score.toFixed(2)}${r.track ? ` ${r.track}` : ""}`; const orig = r.enOrig === undefined || r.track === EN_ORIG_TRACK ? "" : ` · en-orig ${r.enOrig.toFixed(2)}`; const label = r.status === "ok" ? "ok" : r.status.toUpperCase(); return ` ${label.padEnd(14)} ${r.citation}${r.cited ? "" : " (not cited)"} ${r.kind} ${where}${score}${orig}` + (r.message ? `\n${" ".repeat(17)}${r.message}` : ""); } export async function verifyQuotesMain( opts: { file: string; json?: boolean; paths?: Paths; now?: string }, out: Out = console, ): Promise { const paths = opts.paths ?? getPaths(); let raw: unknown; try { raw = JSON.parse(await readFile(opts.file, "utf8")); } catch (err) { out.error(`reports verify-quotes: ${opts.file} is not readable JSON (${(err as Error).message})`); return 2; } const parsed = parseReport(raw); if (!parsed.ok) { out.error(`reports verify-quotes: ${opts.file} is not a report:`); for (const p of parsed.problems) out.error(` ${p.path}: ${p.message}`); return 1; } const results = await verifyReportQuotes(parsed.value, { paths, ...(opts.now ? { now: opts.now } : {}) }); const bad = results.filter((r) => r.status !== "ok"); if (opts.json) { out.log(JSON.stringify({ file: opts.file, problems: parsed.problems, results }, null, 2)); } else { for (const p of parsed.problems) out.error(` invalid: ${p.path}: ${p.message}`); for (const r of results) (r.status === "ok" ? out.log : out.error)(describe(r)); const counts = new Map(); for (const r of results) counts.set(r.status, (counts.get(r.status) ?? 0) + 1); out.log( `reports verify-quotes: ${results.length} quote(s): ` + [...counts].map(([s, n]) => `${n} ${s}`).join(", ") + (parsed.problems.length ? `; ${parsed.problems.length} validation problem(s)` : ""), ); } return bad.length > 0 || parsed.problems.length > 0 ? 1 : 0; }