Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 2da702073bcdf8dd3e2add17ca5d05c52c16cde3
parent fb8c3674cae5a4a0e4f14a74e683b6192bb5b20c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 03:32:37 -0400

common: report converters — /sweep markdown, /ask answers and report-to-video manifests into report.json, and a report into a starter manifest; archilyzer reports convert / to-manifest

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/bin/archilyzer.ts | 44++++++++++++++++++++++++++++++++++++++++++++
Acommon/bin/reports-convert.ts | 140+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/report/convert-server.ts | 123+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/report/convertAsk.ts | 172+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/report/convertManifest.ts | 552+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/report/convertShared.ts | 628+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/report/convertSweep.ts | 52++++++++++++++++++++++++++++++++++++++++++++++++++++
7 files changed, 1711 insertions(+), 0 deletions(-)

diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -162,6 +162,50 @@ export const COMMANDS: Command[] = [ }, }, { + path: ["reports", "convert"], + usage: + "<sweep|ask|manifest> <in> --out <report.json> [--channels-dir <dir>] [--id <id>] [--title <title>] a /sweep report (markdown), an /ask answer or a report-to-video manifest as a report.json, written only when it validates (--channels-dir: widen spans from the cues, find posts' channels)", + flags: { out: "string", "channels-dir": "string", id: "string", title: "string" }, + maxPositionals: 2, + run: async ({ positionals, flags }) => { + const [from, input] = positionals; + const { convertMain, isConvertFrom, CONVERT_FROM } = await import("./reports-convert"); + if (!isConvertFrom(from) || !input || typeof flags.out !== "string") { + console.error(`reports convert: give <${CONVERT_FROM.join("|")}> <in> --out <report.json>`); + return 2; + } + return convertMain({ + from, + input, + out: flags.out, + ...(typeof flags["channels-dir"] === "string" ? { channelsDir: flags["channels-dir"] } : {}), + ...(typeof flags.id === "string" ? { id: flags.id } : {}), + ...(typeof flags.title === "string" ? { title: flags.title } : {}), + }); + }, + }, + { + path: ["reports", "to-manifest"], + usage: + "<report.json> --out <manifest.json> [--channels-dir <dir>] [--site-origin <url>] a starter report-to-video manifest from a report (a chapter card per section, a claim's still and clips stamped with its verdict, its posts)", + flags: { out: "string", "channels-dir": "string", "site-origin": "string" }, + maxPositionals: 1, + run: async ({ positionals, flags }) => { + const [input] = positionals; + if (!input || typeof flags.out !== "string") { + console.error("reports to-manifest: give <report.json> --out <manifest.json>"); + return 2; + } + const { toManifestMain } = await import("./reports-convert"); + return toManifestMain({ + input, + out: flags.out, + ...(typeof flags["channels-dir"] === "string" ? { channelsDir: flags["channels-dir"] } : {}), + ...(typeof flags["site-origin"] === "string" ? { siteOrigin: flags["site-origin"] } : {}), + }); + }, + }, + { path: ["source", "publish"], usage: "[--force] [--check] [--keep-scratch] the scrubbed git mirror, raw tree, history pages (stagit, when installed) and tarball into homepage/public, behind the denied-literal gate (--check: audit and count, write nothing)", diff --git a/common/bin/reports-convert.ts b/common/bin/reports-convert.ts @@ -0,0 +1,140 @@ +// `archilyzer reports convert <sweep|ask|manifest> <in> --out <report.json>` +// and `archilyzer reports to-manifest <report.json> --out <manifest.json>` — +// the converters (lib/report/convert*.ts) over files. +// +// `--channels-dir <dir>` (a `transcripts/channels` tree) lets a conversion +// read what the documents do not carry: a cited record's cues, so a /sweep or +// /ask citation's one second widens to whole sentences; the channel that keeps +// a post cited by its platform link; a post's platform, link and date for a +// manifest's `posts`. Without it a span is the second plus 10 s and those +// posts are left out, each with a warning. Nothing is fetched. +// +// The output is checked before it is written: `convert` writes only a report +// the report validator passes, `to-manifest` reads only one. Exit 0 written +// (warnings on stderr), 1 refused with problems (nothing written), 2 an input +// that cannot be read. + +import { readFile } from "node:fs/promises"; +import path from "node:path"; +import { writeJsonAtomic } from "../lib/jsonFile-server"; +import type { Problem } from "../lib/citations/validate"; +import { askToReport, readAskAnswer } from "../lib/report/convertAsk"; +import { manifestToReport, reportToManifest } from "../lib/report/convertManifest"; +import { sweepToReport } from "../lib/report/convertSweep"; +import type { ConvertContext, Converted } from "../lib/report/convertShared"; +import { diskCuesOf, diskPostChannelOf, diskPostOf } from "../lib/report/convert-server"; + +type Out = { log: (s: string) => void; error: (s: string) => void }; + +export const CONVERT_FROM = ["sweep", "ask", "manifest"] as const; +export type ConvertFrom = (typeof CONVERT_FROM)[number]; + +export function isConvertFrom(v: unknown): v is ConvertFrom { + return typeof v === "string" && (CONVERT_FROM as readonly string[]).includes(v); +} + +function printProblems(out: Out, what: string, problems: Problem[]): void { + out.error(`${what}: ${problems.length} problem(s):`); + for (const p of problems) out.error(` ${p.path || "(root)"}: ${p.message}`); +} + +async function readInput(file: string, out: Out, what: string): Promise<string | null> { + try { + return await readFile(file, "utf8"); + } catch (err) { + out.error(`${what}: cannot read ${file}: ${(err as Error).message}`); + return null; + } +} + +function parseJson(text: string, file: string, out: Out, what: string): unknown { + try { + return JSON.parse(text); + } catch (err) { + out.error(`${what}: ${file} is not JSON: ${(err as Error).message}`); + return undefined; + } +} + +export async function convertMain( + opts: { from: ConvertFrom; input: string; out: string; channelsDir?: string; id?: string; title?: string }, + out: Out = console, +): Promise<number> { + const what = `reports convert ${opts.from}`; + const text = await readInput(opts.input, out, what); + if (text === null) return 2; + const warnings: string[] = []; + const ctx: ConvertContext = opts.channelsDir + ? { cuesOf: diskCuesOf(opts.channelsDir, warnings), postChannelOf: diskPostChannelOf(opts.channelsDir) } + : {}; + const named = { ...ctx, ...(opts.id ? { id: opts.id } : {}), ...(opts.title ? { title: opts.title } : {}) }; + + let converted: Converted; + if (opts.from === "sweep") { + converted = await sweepToReport(text, named); + } else { + const raw = parseJson(text, opts.input, out, what); + if (raw === undefined) return 2; + if (opts.from === "ask") { + const answer = readAskAnswer(raw); + if (!answer) { + out.error(`${what}: ${opts.input} is not an /ask answer ({ answer, sources }, a saved chat or an assistant message)`); + return 2; + } + converted = await askToReport(answer, named); + } else { + converted = await manifestToReport(raw, named); + const hasStills = Object.values(converted.report.citations ?? {}).some((c) => c.kind === "source" && c.image); + if (hasStills && path.resolve(path.dirname(opts.input)) !== path.resolve(path.dirname(opts.out))) { + warnings.push( + "the report's stills are paths relative to the manifest's directory — copy them beside the report (or write it there)", + ); + } + } + } + + for (const w of [...warnings, ...converted.warnings]) out.error(`warning: ${w}`); + if (converted.problems.length) { + printProblems(out, `${what}: not written`, converted.problems); + return 1; + } + await writeJsonAtomic(opts.out, converted.report, { mkdir: true }); + const r = converted.report; + const claims = r.sections.reduce((n, s) => n + (s.claims?.length ?? 0), 0); + out.log( + `${opts.out}: report ${r.id} (${r.kind}) — ${r.sections.length} section(s), ${claims} claim(s), ${Object.keys(r.citations ?? {}).length} citation(s)`, + ); + return 0; +} + +export async function toManifestMain( + opts: { input: string; out: string; channelsDir?: string; siteOrigin?: string }, + out: Out = console, +): Promise<number> { + const what = "reports to-manifest"; + const text = await readInput(opts.input, out, what); + if (text === null) return 2; + const raw = parseJson(text, opts.input, out, what); + if (raw === undefined) return 2; + const result = await reportToManifest(raw, { + ...(opts.siteOrigin ? { siteOrigin: opts.siteOrigin } : {}), + ...(opts.channelsDir ? { postOf: diskPostOf(opts.channelsDir) } : {}), + }); + if (!result.ok) { + printProblems(out, `${what}: ${opts.input} is not a sound report`, result.problems); + return 1; + } + const warnings = [...result.warnings]; + const timeline = result.manifest.timeline as { type: string }[]; + if ( + timeline.some((e) => e.type === "image") && + path.resolve(path.dirname(opts.input)) !== path.resolve(path.dirname(opts.out)) + ) { + warnings.push("the images' src are paths relative to the report's directory — copy the stills beside the manifest"); + } + for (const w of warnings) out.error(`warning: ${w}`); + await writeJsonAtomic(opts.out, result.manifest, { mkdir: true }); + const posts = (result.manifest.posts as unknown[] | undefined)?.length ?? 0; + out.log(`${opts.out}: ${timeline.length} timeline entr${timeline.length === 1 ? "y" : "ies"}, ${posts} post(s)`); + return 0; +} diff --git a/common/lib/report/convert-server.ts b/common/lib/report/convert-server.ts @@ -0,0 +1,123 @@ +// The converters' disk half: what a channels tree can tell a converter, as the +// callbacks the converters take — a record's cues, the archive channel that +// keeps a post, a post's record. The CLI is bin/reports-convert.ts. +// +// Reads only TEXT: `channels/<slug>/data/<id>/transcript.cues.json` (behind +// the text guard, lib/channelMedia.ts — a legacy channel or one mid-migration +// is "no cues", with a warning, never a read of the wrong tree), and each +// channel's `posts-archive` and `posts/*.jsonl`. Writes nothing. +// +// SERVER-ONLY (node:fs). + +import { readdir, readFile } from "node:fs/promises"; +import path from "node:path"; +import { assertChannelTextReadable } from "../channelMedia"; +import { isSafeChannelSegment, isSafeIdSegment } from "../citations/moments"; +import { readAllPosts, readSeenPostIds } from "../posts-server"; +import type { Post } from "../posts"; +import type { Cue, CuesOf, PostChannelOf } from "./convertShared"; +import type { PostOf, PostRecord } from "./convertManifest"; + +function isCue(v: unknown): v is Cue { + if (!v || typeof v !== "object") return false; + const c = v as Record<string, unknown>; + return typeof c.start === "number" && typeof c.end === "number" && typeof c.text === "string" && c.end >= c.start; +} + +// A record's cues off the channels tree, cached per record. A channel whose +// text is not readable is one warning and no cues. +export function diskCuesOf(channelsDir: string, warnings: string[] = []): CuesOf { + const guarded = new Map<string, Promise<boolean>>(); + const cache = new Map<string, Promise<Cue[] | null>>(); + const readable = (slug: string) => { + let p = guarded.get(slug); + if (!p) { + p = assertChannelTextReadable({ channelsDir }, slug).then( + () => true, + (err: Error) => { + warnings.push(`${slug}: ${err.message} — no cues read from it`); + return false; + }, + ); + guarded.set(slug, p); + } + return p; + }; + return (channel, id) => { + if (!isSafeChannelSegment(channel) || !isSafeIdSegment(id)) return Promise.resolve(null); + const key = `${channel}/${id}`; + let p = cache.get(key); + if (!p) { + p = (async () => { + if (!(await readable(channel))) return null; + let doc: unknown; + try { + doc = JSON.parse(await readFile(path.join(channelsDir, channel, "data", id, "transcript.cues.json"), "utf8")); + } catch { + return null; + } + const cues = (doc as { cues?: unknown })?.cues; + if (!Array.isArray(cues)) return null; + return cues.filter(isCue).sort((a, b) => a.start - b.start); + })(); + cache.set(key, p); + } + return p; + }; +} + +// The channel directories under the tree, by name. +async function channelSlugs(channelsDir: string): Promise<string[]> { + const entries = await readdir(channelsDir, { withFileTypes: true }).catch(() => []); + return entries.filter((e) => e.isDirectory() && isSafeChannelSegment(e.name)).map((e) => e.name).sort(); +} + +const squash = (v: string) => v.toLowerCase().replace(/^@/, "").replace(/[^a-z0-9]+/g, ""); + +// The archive channel that keeps a post: every channel's posts archive, read +// once; among several, the one whose slug carries the post's handle. +export function diskPostChannelOf(channelsDir: string): PostChannelOf { + let index: Promise<Map<string, string[]>> | null = null; + const build = async () => { + const byId = new Map<string, string[]>(); + for (const slug of await channelSlugs(channelsDir)) { + for (const id of await readSeenPostIds(path.join(channelsDir, slug))) { + byId.set(id, [...(byId.get(id) ?? []), slug]); + } + } + return byId; + }; + return async ({ id, handle }) => { + index ??= build(); + const slugs = (await index).get(id) ?? []; + if (slugs.length <= 1) return slugs[0] ?? null; + const h = handle ? squash(handle.split(".")[0]) : ""; + return (h && slugs.find((s) => squash(s).includes(h))) || slugs[0]; + }; +} + +// A post's record off its channel's archive, each channel read once. +export function diskPostOf(channelsDir: string): PostOf { + const channels = new Map<string, Promise<Map<string, Post>>>(); + return async (channel, id) => { + if (!isSafeChannelSegment(channel)) return null; + let p = channels.get(channel); + if (!p) { + p = readAllPosts(path.join(channelsDir, channel)).then( + (posts) => new Map(posts.map((post) => [post.id, post])), + () => new Map(), + ); + channels.set(channel, p); + } + const post = (await p).get(id); + if (!post) return null; + const rec: PostRecord = { + platform: post.platform === "bluesky" ? "bluesky" : "x", + url: post.url, + createdAt: post.createdAt, + author: post.author, + }; + if (post.authorName) rec.authorName = post.authorName; + return rec; + }; +} diff --git a/common/lib/report/convertAsk.ts b/common/lib/report/convertAsk.ts @@ -0,0 +1,172 @@ +// An /ask ANSWER → `report.json` (kind "sweep"). +// +// The export's /ask writes an answer (or a running report) in markdown that +// cites its sources with numbered markers — `[n]`, or `[n @ mm:ss]` for a +// moment (also `h:mm:ss`, stray spaces tolerated) — where `n` indexes the +// answer's source list, the retrieved records (export/app/ask/citations.tsx +// parses the same markers; export/app/lib/askRetrieval.ts `RetrievedVideo` is +// a source). Accepted as input, all JSON: +// +// { "answer": "<md>", "sources": [<source>…], "question"?: "…" } +// a saved chat ({ "messages": […], "report"?, "reportSources"? }): its +// report and report sources when it has them, else its last answer and +// that answer's sources, with the question before it +// one assistant message ({ "role": "assistant", "content", "sources" }) +// +// Each marker becomes `[label](cite:<id>)`: a post source a post citation +// (its words the quote), any other source a span at the marked second — +// snapped to the nearest second the source's excerpt lines carry, as the +// /ask page does, so a slightly-off model time lands on a real line — or at +// its first excerpt line when the marker has no time; its quote is that +// line's text. Then the answer is structured like a sweep (./convertShared.ts +// `structureMarkdown`): paragraphs and list items that cite are claims. + +import { + CitationRegistry, + emptyReport, + finish, + reportIdOf, + structureMarkdown, + type ConvertContext, + type Converted, +} from "./convertShared"; + +export type AskSource = { + // `<channel>/<id>`: the record (or post) the source is. + key: string; + title?: string; + uploadDate?: string; + snippets?: { clock?: string; seconds: number; text: string }[]; + isPost?: boolean; +}; + +export type AskAnswer = { answer: string; sources: AskSource[]; question?: string }; + +export type AskConvertOptions = ConvertContext & { + id?: string; + // Default: the question, else "Answer". + title?: string; + sectionTitle?: string; +}; + +const isObj = (v: unknown): v is Record<string, unknown> => !!v && typeof v === "object" && !Array.isArray(v); + +function sourcesOf(v: unknown): AskSource[] | null { + if (!Array.isArray(v)) return null; + return v.filter((s): s is AskSource => isObj(s) && typeof s.key === "string"); +} + +// The answer, its sources and its question out of any accepted shape; null for +// JSON that is none of them. +export function readAskAnswer(raw: unknown): AskAnswer | null { + if (!isObj(raw)) return null; + const question = typeof raw.question === "string" ? raw.question : undefined; + for (const [text, list] of [ + ["answer", "sources"], + ["report", "reportSources"], + ] as const) { + const sources = sourcesOf(raw[list]); + if (typeof raw[text] === "string" && sources) return { answer: raw[text] as string, sources, question }; + } + if (raw.role === "assistant" && typeof raw.content === "string") { + return { answer: raw.content, sources: sourcesOf(raw.sources) ?? [], question }; + } + if (Array.isArray(raw.messages)) { + const messages = raw.messages.filter(isObj); + for (let i = messages.length - 1; i >= 0; i--) { + const m = messages[i]; + if (m.role !== "assistant" || typeof m.content !== "string" || !m.content.trim()) continue; + const asked = messages.slice(0, i).reverse().find((u) => u.role === "user" && typeof u.content === "string"); + return { answer: m.content, sources: sourcesOf(m.sources) ?? [], question: question ?? (asked?.content as string | undefined) }; + } + } + return null; +} + +// `mm:ss` or `h:mm:ss` as whole seconds; null when malformed. +export function parseClock(s: string): number | null { + const parts = s.split(":"); + if (parts.length < 2 || parts.length > 3) return null; + let total = 0; + for (const p of parts) { + if (!/^\d+$/.test(p)) return null; + total = total * 60 + Number(p); + } + return total; +} + +// One citation marker: a source number, an optional `@ mm:ss`, stray spaces +// tolerated — not one already followed by a link target. +const MARKER_RE = /\[\s*(\d+)\s*(?:@\s*(\d{1,2}(?::\d{2}){1,2})\s*)?\](?!\()/g; + +// The excerpt line a marker cites: the one nearest the marked second, else +// the first. +function snippetAt(src: AskSource, seconds: number | null): { seconds: number; text: string } | null { + const lines = src.snippets ?? []; + if (!lines.length) return null; + if (seconds === null) return lines[0]; + return lines.reduce((best, s) => (Math.abs(s.seconds - seconds) < Math.abs(best.seconds - seconds) ? s : best)); +} + +const dateOfUpload = (d: string | undefined) => + d && /^\d{8}$/.test(d) ? `${d.slice(0, 4)}-${d.slice(4, 6)}-${d.slice(6, 8)}` : undefined; + +export async function askToReport(input: AskAnswer, opts: AskConvertOptions = {}): Promise<Converted> { + const warnings: string[] = []; + const registry = new CitationRegistry(opts, warnings); + + // The markers, outside code, as `[label](cite:<id>)`. + const segments = input.answer.split(/(```[\s\S]*?```|`[^`]*`)/g); + for (let i = 0; i < segments.length; i += 2) { + let out = ""; + let last = 0; + for (const m of segments[i].matchAll(MARKER_RE)) { + out += segments[i].slice(last, m.index); + last = m.index + m[0].length; + const n = Number(m[1]); + const src = input.sources[n - 1]; + if (!src) { + warnings.push(`[${m[1]}${m[2] ? ` @ ${m[2]}` : ""}] names no source (the answer has ${input.sources.length})`); + out += m[0]; + continue; + } + const slash = src.key.indexOf("/"); + if (slash <= 0) { + warnings.push(`source ${n} (${src.key}) is not a <channel>/<id> key — left as it is`); + out += m[0]; + continue; + } + const channel = src.key.slice(0, slash); + const id = src.key.slice(slash + 1); + const marked = m[2] ? parseClock(m[2]) : null; + let cid: string; + if (src.isPost) { + const words = (src.snippets ?? []).map((s) => s.text).join("\n").trim() || src.title || id; + cid = registry.post(channel, id, words, { date: dateOfUpload(src.uploadDate) }); + } else { + const line = snippetAt(src, marked); + if (!line) warnings.push(`source ${n} (${src.key}) has no excerpt lines — its quote is its title`); + cid = await registry.span(channel, id, line?.seconds ?? marked ?? 0, line?.text.trim() || src.title || id, { + label: src.title, + }); + } + out += `[${m[0].slice(1, -1).trim()}](cite:${cid})`; + } + segments[i] = out + segments[i].slice(last); + } + + const doc = await structureMarkdown( + segments.join(""), + async ({ href }) => (href.startsWith("cite:") ? href.slice("cite:".length) : null), + { defaultSection: opts.sectionTitle ?? "Answer" }, + ); + const question = input.question?.replace(/\s+/g, " ").trim(); + const title = opts.title ?? doc.title ?? question ?? "Answer"; + const report = emptyReport(reportIdOf(opts.id, title), "sweep", title); + if (question && question !== title) report.subtitle = question; + if (doc.summary) report.summary = doc.summary; + report.citations = registry.citations; + report.sections = doc.sections; + if (!Object.keys(registry.citations).length) warnings.push("the answer cites no source"); + return finish(report, warnings); +} diff --git a/common/lib/report/convertManifest.ts b/common/lib/report/convertManifest.ts @@ -0,0 +1,552 @@ +// A report-to-video MANIFEST ↔ `report.json`, both ways. The video and the +// report page can be cut from one source: a manifest becomes a report, and a +// report becomes a starter manifest to refine in umtool. +// +// The manifest is umtool/report-to-video's (its README, "Manifest shape"). +// What maps to what: +// +// manifest report +// ──────────────────────────────────────── ────────────────────────────────────── +// slug, title, subtitle id, title, subtitle +// card (style "chapter", or none) a section: heading → title, sub → body +// clip (channel ?? provenance.channelSlug, a video citation, id = the clip's id; +// video, cutStart ?? start, an `audioOnly` clip an audio one +// cutEnd ?? end, quote, date, note) +// image (src, or its first panel; quote, a source citation (its still = src) +// title, date, citeUrl, note) and a source of its own +// posts[] (siteChannel, its id, text, date) a post citation, id = the post's id +// `claim: { id, verdict }` on entries a claim with that verdict; its first +// image is its source sentence +// ledger[] (id, quote, label, date, a claim with no verdict (text = the +// entryId, channel, video, cite) quote, title = the label) +// +// An entry that carries no claim is cited in its section's body, one line +// each. A post goes with the clip it is attached to (`attachTo`, else the +// clip it follows by date, else the first) — under that clip's claim, or in +// its section's body. A claim's text is its ledger entry's quote, else the +// heading of a card that carries it, else its source sentence, else its first +// clip's quote. A report with any verdict is a fact-check, else a sweep. +// +// Left out, with a warning: a clip with its own media (`src`: no record to +// cite), an entry without a quote, a post no archive channel is known to keep. +// A manifest has more than a report (render settings, holds, the deck, cut +// edits, redactions, the ledger's adjudication) and a report more than a +// manifest (findings, a claim's own title, citation pads, labels, speakers); +// neither survives the trip. Image paths are written as the manifest has +// them: relative to ITS directory (the CLI says so when the report is written +// elsewhere). +// +// A report → a STARTER manifest: no cold open — a title card, then per section +// a chapter card, its body's citations in order, and per claim its source +// sentence's still (an `image`) and its video and audio citations (`clip`s), +// each carrying `claim: { id, verdict }` when the claim has a verdict, and its +// posts in `posts[]` attached to the claim's last clip. Spans are written as +// the report has them, so `resolve-windows.mjs` and the clip bench start from +// the cited sentences. A post needs its platform, its link and its date: from +// `postOf` (the CLI reads the channel's archive) or, for a numeric X id, its +// `x.com/i/status/` link and the citation's date; one that has none of them is +// left out with a warning. + +import type { Citation, Source, SourceCitation } from "../citations/schema"; +import { isHttpUrl, isPartialDate } from "../citations/validate"; +import { + IdAllocator, + emptyReport, + finish, + citedIds, + reportIdOf, + spanAt, + type CitationMap, + type ConvertContext, + type Converted, +} from "./convertShared"; +import type { Claim, Report, Section } from "./schema"; +import { isVerdict, type Verdict } from "./verdicts"; +import { parseReport } from "./validate"; +import type { Problem } from "../citations/validate"; + +type Obj = Record<string, unknown>; + +const isObj = (v: unknown): v is Obj => !!v && typeof v === "object" && !Array.isArray(v); +const str = (v: unknown): string | undefined => (typeof v === "string" && v.trim() ? v : undefined); +const num = (v: unknown): number | undefined => (typeof v === "number" && Number.isFinite(v) ? v : undefined); + +export type ManifestConvertOptions = ConvertContext & { id?: string; title?: string }; + +// A manifest entry's claim, when it carries a sound one. +function claimOfEntry(e: Obj): { id: string; verdict: Verdict } | null { + if (!isObj(e.claim) || e.type === "teaser") return null; + const id = str(e.claim.id)?.trim(); + return id && isVerdict(e.claim.verdict) ? { id, verdict: e.claim.verdict } : null; +} + +// The id a post has on its platform: `postId`, else the link's. +export function postNativeId(post: Obj): string | null { + const given = str(post.postId); + if (given && /^[A-Za-z0-9_-]{1,128}$/.test(given)) return given; + try { + const u = new URL(String(post.url ?? "")); + const m = /\/post\/([A-Za-z0-9]+)\/?$/.exec(u.pathname) ?? /\/status(?:es)?\/(\d+)(?:\/|$)/.exec(u.pathname); + return m ? m[1] : null; + } catch { + return null; + } +} + +const dayOf = (d: unknown) => (typeof d === "string" && /^\d{4}-\d{2}-\d{2}/.test(d) ? d.slice(0, 10) : null); + +// The clip a post goes with, as the build attaches it: `attachTo`, else the +// clip whose date most closely precedes the post's (ties to the later in the +// cut), else the first. +function clipForPost(post: Obj, clips: Obj[]): Obj | null { + const named = str(post.attachTo); + if (named) { + const c = clips.find((e) => e.id === named); + if (c) return c; + } + const day = dayOf(post.date); + let best: Obj | null = null; + if (day) { + for (const c of clips) { + const cd = dayOf(c.date); + if (cd && cd <= day && (!best || cd >= (dayOf(best.date) as string))) best = c; + } + } + return best ?? clips[0] ?? null; +} + +export async function manifestToReport(manifest: unknown, opts: ManifestConvertOptions = {}): Promise<Converted> { + const warnings: string[] = []; + const m: Obj = isObj(manifest) ? manifest : {}; + const timeline = (Array.isArray(m.timeline) ? m.timeline : []).filter(isObj); + const ledger = (Array.isArray(m.ledger) ? m.ledger : []).filter(isObj); + const posts = (Array.isArray(m.posts) ? m.posts : []).filter(isObj); + const prov = isObj(m.provenance) ? m.provenance : {}; + const title = opts.title ?? str(m.title) ?? str(m.slug) ?? "Report"; + const report = emptyReport(reportIdOf(opts.id ?? str(m.slug), title), "sweep", title); + if (str(m.subtitle)) report.subtitle = m.subtitle as string; + + const citations: CitationMap = {}; + const sources: Record<string, Source> = {}; + const citeIds = new IdAllocator(); + const anchors = new IdAllocator(); + const where = (e: Obj, i: number) => `timeline[${i}] ${str(e.id) ?? "?"}`; + + // ── The citations every entry is ── + const citeOf = new Map<Obj, string>(); + for (const [i, e] of timeline.entries()) { + if (e.type === "clip") { + if (str(e.src)) { + warnings.push(`${where(e, i)}: a clip of its own media (src) cites no record — left out`); + continue; + } + const channel = str(e.channel) ?? str(prov.channelSlug); + const video = str(e.video); + const start = num(e.cutStart) ?? num(e.start); + const end = num(e.cutEnd) ?? num(e.end); + const quote = str(e.quote); + if (!channel || !video || start === undefined || end === undefined) { + warnings.push(`${where(e, i)}: a clip needs a channel, a video, a start and an end — left out`); + continue; + } + if (!quote) { + warnings.push(`${where(e, i)}: a clip without a quote — left out (a citation's quote is verbatim)`); + continue; + } + const cid = citeIds.claim(str(e.id) ?? "c", "c"); + const c: Citation = { kind: e.audioOnly === true ? "audio" : "video", channel, id: video, start, end, quote }; + if (str(e.date) && isPartialDate(e.date as string)) c.date = e.date as string; + if (str(e.note)) c.note = e.note as string; + citations[cid] = c; + citeOf.set(e, cid); + } else if (e.type === "image") { + const panels = Array.isArray(e.panels) ? e.panels.filter(isObj) : []; + const src = str(e.src) ?? str(panels[0]?.src); + if (!src) { + warnings.push(`${where(e, i)}: an image with no src — left out`); + continue; + } + if (panels.length > 1) warnings.push(`${where(e, i)}: ${panels.length} panels — the citation's still is the first`); + let quote = str(e.quote); + if (!quote) { + quote = str(e.title) ?? str(e.id) ?? src; + warnings.push(`${where(e, i)}: an image without a quote — its citation quotes its title; give it the still's words`); + } + const cid = citeIds.claim(str(e.id) ?? "a", "a"); + const sid = citeIds.claim(`src-${cid}`); + const source: Source = { kind: "other", title: str(e.title) ?? cid }; + if (str(e.citeUrl) && isHttpUrl(e.citeUrl as string)) source.url = e.citeUrl as string; + if (str(e.date) && isPartialDate(e.date as string)) source.date = e.date as string; + sources[sid] = source; + const c: SourceCitation = { kind: "source", source: sid, quote, image: src }; + if (str(e.note)) c.note = e.note as string; + citations[cid] = c; + citeOf.set(e, cid); + } + } + + // ── Posts: each a citation, and the clip it goes with ── + const clips = timeline.filter((e) => e.type === "clip" && citeOf.has(e)); + const postsOfClip = new Map<Obj, string[]>(); + const loosePosts: string[] = []; + for (const [i, p] of posts.entries()) { + if (p.hide === true) continue; + const native = postNativeId(p); + const platform = p.platform === "bluesky" ? "bluesky" : "x"; + const channel = + str(p.siteChannel) ?? + (native && opts.postChannelOf ? await opts.postChannelOf({ platform, id: native, handle: str(p.handle) }) : null); + const text = str(p.text); + if (!native || !channel || !text) { + warnings.push( + `posts[${i}] ${str(p.id) ?? "?"}: ${!native ? "no id on its platform" : !text ? "no text" : "no archive channel known to keep it (siteChannel)"} — left out`, + ); + continue; + } + const cid = citeIds.claim(str(p.id) ?? "p", "p"); + citations[cid] = { + kind: "post", + channel, + id: native, + quote: text, + ...(str(p.date) && isPartialDate(p.date as string) ? { date: p.date as string } : {}), + }; + const clip = clipForPost(p, clips); + if (clip) postsOfClip.set(clip, [...(postsOfClip.get(clip) ?? []), cid]); + else loosePosts.push(cid); + } + + // ── Ledger claims: what the ledger says a claim is, and the clip it pins ── + const ledgerById = new Map<string, Obj>(); + for (const l of ledger) if (str(l.id)) ledgerById.set(l.id as string, l); + const timelineClaims = new Set(timeline.map(claimOfEntry).filter((c) => c).map((c) => c!.id)); + const ledgerClaimOfClip = new Map<string, string>(); + for (const l of ledger) { + const id = str(l.id); + const pinned = str(l.entryId); + if (id && pinned && !timelineClaims.has(id)) ledgerClaimOfClip.set(pinned, id); + } + + // ── The walk: sections, claims, bodies ── + const sections: { section: Section; body: string[] }[] = []; + const claims = new Map<string, { claim: Claim; texts: { from: string; text: string }[] }>(); + const current = () => { + if (!sections.length) { + sections.push({ section: { id: anchors.claim("opening"), title: "Opening", claims: [] }, body: [] }); + } + return sections[sections.length - 1]; + }; + const claimFor = (id: string, verdict: Verdict | null) => { + let k = claims.get(id); + if (!k) { + const claim: Claim = { id: anchors.claim(id, "k"), text: "", citations: [] }; + if (verdict) claim.verdict = verdict; + k = { claim, texts: [] }; + claims.set(id, k); + current().section.claims!.push(claim); + } + return k; + }; + const list = (claim: Claim, cid: string) => { + if (!claim.citations!.includes(cid)) claim.citations!.push(cid); + }; + const bodyLine = (cid: string) => { + const c = citations[cid]; + const quote = c.quote.replace(/\s+/g, " ").trim(); + const label = + c.kind === "post" ? "post" : c.kind === "source" ? (sources[c.source]?.title ?? cid) : c.kind === "page" ? cid : `${c.id}`; + return `- “${quote}” [${label.replace(/[[\]]/g, "")}](cite:${cid})`; + }; + + for (const e of timeline) { + const tagged = claimOfEntry(e); + if (e.type === "card") { + if (tagged) { + const k = claimFor(tagged.id, tagged.verdict); + if (str(e.heading)) k.texts.push({ from: "card", text: e.heading as string }); + continue; + } + const style = e.style; + if ((style === undefined || style === "chapter") && str(e.heading)) { + const section: Section = { id: anchors.claim(str(e.id) ?? "section", "section"), title: e.heading as string, claims: [] }; + sections.push({ section, body: str(e.sub) ? [e.sub as string] : [] }); + } + continue; + } + const cid = citeOf.get(e); + if (!cid) continue; + const ledgerClaim = !tagged && str(e.id) ? ledgerClaimOfClip.get(e.id as string) : undefined; + const claimId = tagged?.id ?? ledgerClaim; + if (claimId) { + const k = claimFor(claimId, tagged?.verdict ?? null); + const c = citations[cid]; + if (c.kind === "source" && !k.claim.sourceQuote) { + k.claim.sourceQuote = { citation: cid }; + k.texts.push({ from: "source", text: c.quote }); + } else { + list(k.claim, cid); + if (c.kind !== "source") k.texts.push({ from: "clip", text: c.quote }); + } + for (const pid of postsOfClip.get(e) ?? []) list(k.claim, pid); + } else { + const body = current().body; + body.push(bodyLine(cid)); + for (const pid of postsOfClip.get(e) ?? []) body.push(bodyLine(pid)); + } + } + + // Ledger claims no clip pins: their own moment, in a section of their own. + const unpinned = ledger.filter((l) => { + const id = str(l.id); + return id && !claims.has(id) && !timelineClaims.has(id); + }); + if (unpinned.length) { + sections.push({ section: { id: anchors.claim("ledger"), title: "Not clipped", claims: [] }, body: [] }); + for (const l of unpinned) { + const k = claimFor(l.id as string, null); + const channel = str(l.channel) ?? str(prov.channelSlug); + const video = str(l.video); + const second = num(l.cite) ?? num(l.start); + const quote = str(l.quote); + if (channel && video && second !== undefined && quote) { + const span = await spanAt(channel, video, second, quote, opts, num(l.end)); + const cid = citeIds.claim(`l-${l.id as string}`); + citations[cid] = { kind: "video", channel, id: video, start: span.start, end: span.end, quote }; + list(k.claim, cid); + } else { + warnings.push(`ledger ${l.id as string}: no channel, video, second and quote to cite — the claim has no citation`); + } + } + } + if (loosePosts.length) { + const body = current().body; + for (const pid of loosePosts) body.push(bodyLine(pid)); + } + + // ── Claim texts and titles ── + for (const [id, { claim, texts }] of claims) { + const l = ledgerById.get(id); + const pick = (from: string) => texts.find((t) => t.from === from)?.text; + const text = str(l?.quote) ?? pick("card") ?? pick("source") ?? pick("clip"); + if (str(l?.label) && str(l?.quote)) claim.title = l!.label as string; + if (text) claim.text = text.replace(/\s+/g, " ").trim(); + else { + claim.text = id; + warnings.push(`claim ${id}: nothing in the manifest states it — its text is its id`); + } + if (!claim.citations!.length) delete claim.citations; + } + + report.kind = [...claims.values()].some((k) => k.claim.verdict) ? "factcheck" : "sweep"; + report.sources = sources; + report.citations = citations; + report.sections = sections.map(({ section, body }) => { + const out: Section = { id: section.id, title: section.title }; + if (body.length) out.body = body.join("\n"); + if (section.claims!.length) out.claims = section.claims; + return out; + }); + return finish(report, warnings); +} + +// ─── report → manifest ─── + +// What a post record says that a post citation does not (the CLI reads it off +// the channel's archive). +export type PostRecord = { + platform: "x" | "bluesky"; + url: string; + createdAt?: string; + author?: string; + authorName?: string; +}; + +export type PostOf = (channel: string, id: string) => Promise<PostRecord | null>; + +export type ManifestOptions = { + // The archive the cut's QR codes link (provenance.siteOrigin). Default "" + // — which `umtool check` blocks on until it is set, as for `umtool new`. + siteOrigin?: string; + postOf?: PostOf; + // Replaces the starter `render` block. + render?: Record<string, unknown>; + // `generatedOn`; default today. + today?: string; +}; + +// The render block `umtool new` writes (umtool/lib/projects/scaffold.mjs +// `skeleton`), so a converted report renders as a new project does. Keep the +// two alike. +export const STARTER_RENDER: Readonly<Record<string, unknown>> = Object.freeze({ + width: 1920, + height: 1080, + fps: 30, + audioRate: 48000, + audioChannels: 2, + maxHeightSource: 1080, + fontRegular: "/usr/share/fonts/TTF/FiraSans-Regular.ttf", + fontBold: "/usr/share/fonts/TTF/FiraSans-Bold.ttf", + palette: { bg: "#12100c", fg: "#f6f1e6", muted: "#a2957f", accent: "#c8752a", amber: "#ffc860" }, + transition: 0.4, + fetchPad: 3, + snapWindow: 1.6, + silenceMinDur: 0.09, + silenceRelDb: 6, + headerHeight: 56, + footerHeight: 0, + crf: 21, + preset: "slow", + qr: { scale: 4, quiet: 3, ecc: "M", margin: 28 }, +}); + +// A manifest entry's `date` is a calendar day. +const DAY_RE = /^\d{4}-\d{2}-\d{2}$/; + +export const CARD_SECONDS = { title: 4, chapter: 3.5 } as const; +export const IMAGE_SECONDS = 6; + +export type ManifestResult = + | { ok: true; manifest: Record<string, unknown>; warnings: string[] } + | { ok: false; problems: Problem[] }; + +export async function reportToManifest(raw: unknown, opts: ManifestOptions = {}): Promise<ManifestResult> { + const parsed = parseReport(raw); + if (!parsed.ok || parsed.problems.length) return { ok: false, problems: parsed.problems }; + const report: Report = parsed.value; + const warnings: string[] = []; + const citations = report.citations ?? {}; + const sources = report.sources ?? {}; + // Timeline ids are file names in a build: letters, digits, `_` and `-`. + const entryIds = new IdAllocator(/[^A-Za-z0-9_-]+/g); + const postIds = new IdAllocator(/[^A-Za-z0-9_-]+/g); + const timeline: Record<string, unknown>[] = []; + const posts: Record<string, unknown>[] = []; + const postsDone = new Set<string>(); + const channels = new Map<string, number>(); + let lastClip: string | null = null; + + timeline.push({ + type: "card", + id: entryIds.claim("title"), + style: "title", + seconds: CARD_SECONDS.title, + heading: report.title, + sub: report.subtitle ?? "", + }); + + const claimTag = (claim: Claim | null) => (claim?.verdict ? { claim: { id: claim.id, verdict: claim.verdict } } : {}); + + const emit = async (cid: string, claim: Claim | null, sourceSentence = false) => { + const c = citations[cid]; + if (!c) return; + switch (c.kind) { + case "video": + case "audio": { + const id = entryIds.claim(cid, "c"); + timeline.push({ + type: "clip", + id, + channel: c.channel, + video: c.id, + start: c.start, + end: c.end, + cite: Math.floor(c.start), + quote: c.quote, + ...(c.kind === "audio" ? { audioOnly: true } : {}), + ...(c.date && DAY_RE.test(c.date) ? { date: c.date } : {}), + ...(c.note ? { note: c.note } : {}), + ...claimTag(claim), + }); + channels.set(c.channel, (channels.get(c.channel) ?? 0) + 1); + lastClip = id; + return; + } + case "source": { + if (!c.image) { + warnings.push( + `${cid}: ${sourceSentence ? "a claim's source sentence" : "a source citation"} with no still — no image entry (shoot it, then set its image)`, + ); + return; + } + const source = sources[c.source]; + const dayOfSource = [c.date, source?.date].find((d) => d && DAY_RE.test(d)); + timeline.push({ + type: "image", + id: entryIds.claim(cid, "a"), + src: c.image, + seconds: IMAGE_SECONDS, + title: source?.title ?? cid, + quote: c.quote, + ...(dayOfSource ? { date: dayOfSource } : {}), + ...(source?.url ? { citeUrl: source.url } : {}), + ...(c.note ? { note: c.note } : {}), + ...claimTag(claim), + }); + return; + } + case "post": { + if (postsDone.has(cid)) return; + postsDone.add(cid); + const rec = opts.postOf ? await opts.postOf(c.channel, c.id) : null; + const platform = rec?.platform ?? (/^\d+$/.test(c.id) ? "x" : null); + const url = rec?.url ?? (platform === "x" ? `https://x.com/i/status/${c.id}` : null); + const date = c.date ?? rec?.createdAt; + if (!platform || !url || !date || !/^\d{4}-\d{2}-\d{2}/.test(date)) { + warnings.push( + `${cid}: post ${c.channel}/${c.id} — ${!platform || !url ? "its platform and link are unknown (give a channels dir)" : "no date"} — left out of posts`, + ); + return; + } + posts.push({ + id: postIds.claim(cid, "p"), + platform, + ...(rec?.authorName ? { author: rec.authorName } : {}), + ...(rec?.author ? { handle: rec.author } : {}), + date, + text: c.quote.slice(0, 3000), + url, + attachTo: lastClip, + siteChannel: c.channel, + ...(/^[A-Za-z0-9_-]{1,128}$/.test(c.id) ? { postId: c.id } : {}), + }); + return; + } + case "page": + warnings.push(`${cid}: a page citation has no manifest entry — left out`); + return; + } + }; + + for (const section of report.sections) { + timeline.push({ + type: "card", + id: entryIds.claim(section.id, "section"), + style: "chapter", + seconds: CARD_SECONDS.chapter, + heading: section.title, + }); + for (const cid of citedIds(section.body)) await emit(cid, null); + for (const claim of section.claims ?? []) { + if (claim.sourceQuote) await emit(claim.sourceQuote.citation, claim, true); + const order = [...(claim.citations ?? []), ...citedIds(claim.findings)]; + for (const cid of [...new Set(order)]) { + if (cid !== claim.sourceQuote?.citation) await emit(cid, claim); + } + } + } + + const channelSlug = [...channels.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? ""; + const manifest: Record<string, unknown> = { + schemaVersion: 1, + slug: report.id, + title: report.title, + subtitle: report.subtitle ?? "", + generatedOn: opts.today ?? new Date().toISOString().slice(0, 10), + provenance: { siteOrigin: opts.siteOrigin ?? "", channelSlug, channel: "" }, + render: { ...(opts.render ?? STARTER_RENDER) }, + timelineNodes: [], + timeline, + ...(posts.length ? { posts } : {}), + }; + return { ok: true, manifest, warnings }; +} diff --git a/common/lib/report/convertShared.ts b/common/lib/report/convertShared.ts @@ -0,0 +1,628 @@ +// THE CONVERTERS' SHARED HALF — what bringing a cited document from elsewhere +// into the citation model takes, whatever the document was: +// +// - reading a link a report cites with (`parseCitationHref`): an archive +// viewer moment (`<origin>/?v=<channel>%2F<id>&t=<s>`, what /sweep and +// /ask write), an archive post (`…&vm=post`), a moment page +// (`/m/<key>/`), or a post on its own platform (x.com, bsky.app); +// - turning ONE cited second into a span (`spanAt`): a record's cues, when +// the caller can read them, widened to whole sentences by the one widening +// (lib/cueWiden.mjs); else the second plus a default span; +// - the markdown engine (`structureMarkdown`): a document's headings into +// sections, its list items and paragraphs that carry a citation into +// claims, everything else into the sections' bodies, and every citing link +// rewritten to `[label](cite:<id>)`; +// - ids, and the finished report checked by the report validator. +// +// The converters themselves are ./convertSweep.ts, ./convertAsk.ts and +// ./convertManifest.ts. Nothing here reads the disk: the cues and the channel +// that keeps a post arrive as callbacks (./convert-server.ts has the disk +// ones), so every converter runs in a test on literals. + +import { widen } from "../cueWiden.mjs"; +import { MAX_CITATION_SPAN_SECONDS, REF_ID_RE, type Citation } from "../citations/schema"; +import { citeHref, extractCiteRefs } from "../citations/inline"; +import { parseMomentPath, roundMomentSeconds } from "../citations/moments"; +import { isPartialDate, type Problem } from "../citations/validate"; +import { REPORT_FORMAT, REPORT_ID_RE, REPORT_VERSION, type Claim, type Report, type Section } from "./schema"; +import { parseReport } from "./validate"; + +export type Cue = { start: number; end: number; text: string }; + +// A record's cues, sorted by start, or null when they cannot be read. +export type CuesOf = (channel: string, id: string) => Promise<readonly Cue[] | null>; + +export type PostPlatformName = "x" | "bluesky"; + +// The archive channel that keeps a post, or null. +export type PostChannelOf = (post: { platform: PostPlatformName; id: string; handle?: string }) => Promise<string | null>; + +export type ConvertContext = { + cuesOf?: CuesOf; + postChannelOf?: PostChannelOf; + // A cited second with no cues to read becomes [second, second + this]. + spanSeconds?: number; +}; + +// What a converter hands back: the report, what it had to guess or leave out +// (warnings), and the report validator's problems — a report with problems is +// not to be written. +export type Converted = { report: Report; warnings: string[]; problems: Problem[] }; + +// A cited second with no cues is this long. A /sweep or /ask citation carries +// one second, never an end; ten seconds is about a spoken sentence. +export const DEFAULT_SPAN_SECONDS = 10; + +// ─── Links ─── + +export type ParsedCitationHref = + // An archive viewer moment, or a moment page: `seconds` is where it starts; + // a moment page also carries its end. + | { kind: "span"; channel: string; id: string; seconds: number; end?: number } + // An archived post on the archive. + | { kind: "post"; channel: string; id: string } + // A post on its own platform: the archive channel is not in the link. + | { kind: "post-original"; platform: PostPlatformName; id: string; handle?: string }; + +const X_HOSTS = new Set(["x.com", "twitter.com", "mobile.twitter.com", "www.x.com", "www.twitter.com", "mobile.x.com"]); +const BSKY_HOSTS = new Set(["bsky.app", "www.bsky.app"]); + +// What a link cites, or null for a link that cites nothing the model knows (a +// platform's video page, an article, a relative link). +export function parseCitationHref(href: string): ParsedCitationHref | null { + let u: URL; + try { + u = new URL(href.trim()); + } catch { + return null; + } + if (u.protocol !== "http:" && u.protocol !== "https:") return null; + const host = u.hostname.toLowerCase(); + + if (X_HOSTS.has(host)) { + const m = /^\/([^/]+)\/status(?:es)?\/(\d+)(?:\/|$)/.exec(u.pathname); + if (!m) return null; + return { kind: "post-original", platform: "x", id: m[2], ...(m[1] !== "i" ? { handle: m[1] } : {}) }; + } + if (BSKY_HOSTS.has(host)) { + const m = /^\/profile\/([^/]+)\/post\/([A-Za-z0-9]+)\/?$/.exec(u.pathname); + if (!m) return null; + return { kind: "post-original", platform: "bluesky", id: m[2], handle: m[1] }; + } + + const moment = parseMomentPath(u.pathname); + if (moment) { + return moment.kind === "span" + ? { kind: "span", channel: moment.channel, id: moment.id, seconds: moment.start, end: moment.end } + : { kind: "post", channel: moment.channel, id: moment.id }; + } + + // The viewer: `v` is `<channel>/<id>` (URL-encoded or not), `t` whole seconds. + const v = u.searchParams.get("v"); + if (!v) return null; + const slash = v.indexOf("/"); + if (slash <= 0 || slash === v.length - 1) return null; + const channel = v.slice(0, slash); + const id = v.slice(slash + 1); + if (u.searchParams.get("vm") === "post") return { kind: "post", channel, id }; + const t = u.searchParams.get("t"); + const seconds = t && /^\d+(\.\d+)?$/.test(t) ? Number(t) : 0; + return { kind: "span", channel, id, seconds }; +} + +// ─── Spans ─── + +const EPS = 0.02; + +const wordCount = (s: string) => s.split(/\s+/).filter((w) => /[\p{L}\p{N}]/u.test(w)).length; + +const round2 = (n: number) => roundMomentSeconds(n); + +// The cue a cited second points at: the viewer's `t` is a cue's start, +// floored, so the cue that starts in [t, t + 1) is the one cited; else the cue +// that holds t; else the nearest one after it. +function cueIndexAt(cues: readonly Cue[], t: number): number { + const starting = cues.findIndex((c) => c.start >= t - EPS && c.start < t + 1); + if (starting >= 0) return starting; + const holding = cues.findIndex((c) => c.start <= t + EPS && c.end > t); + if (holding >= 0) return holding; + const after = cues.findIndex((c) => c.start > t); + return after >= 0 ? after : cues.length - 1; +} + +export type Span = { start: number; end: number; from: "cues" | "default" | "link" }; + +// One cited second as a span. With the record's cues: the cue it cites, run +// on cue by cue until it holds as many words as the quote, then widened to +// whole sentences — and if the widened span is longer than a citation may be, +// the unwidened cue run, and failing that the default. Without cues (or past +// their end): [second, second + spanSeconds]. An `end` the link already gave +// (a moment page) is kept as it is. +export async function spanAt( + channel: string, + id: string, + seconds: number, + quote: string, + ctx: ConvertContext, + end?: number, +): Promise<Span> { + if (end !== undefined && end > seconds) return { start: round2(seconds), end: round2(end), from: "link" }; + const fallback: Span = { + start: round2(seconds), + end: round2(seconds + (ctx.spanSeconds ?? DEFAULT_SPAN_SECONDS)), + from: "default", + }; + const cues = ctx.cuesOf ? await ctx.cuesOf(channel, id) : null; + if (!cues || cues.length === 0 || seconds > cues[cues.length - 1].end) return fallback; + const i = cueIndexAt(cues, seconds); + const want = Math.max(1, wordCount(quote)); + let j = i; + let have = wordCount(cues[i].text); + while (have < want && j + 1 < cues.length && cues[j + 1].end - cues[i].start <= MAX_CITATION_SPAN_SECONDS) { + j += 1; + have += wordCount(cues[j].text); + } + const fits = (s: number, e: number) => e > s && e - s <= MAX_CITATION_SPAN_SECONDS && round2(e) > round2(s); + const w = widen(cues, cues[i].start, cues[j].end); + if (fits(w.start, w.end)) return { start: round2(w.start), end: round2(w.end), from: "cues" }; + if (fits(cues[i].start, cues[j].end)) return { start: round2(cues[i].start), end: round2(cues[j].end), from: "cues" }; + return fallback; +} + +// ─── Ids ─── + +// Unique ids in one namespace, each a reference id (lib/citations/schema.ts +// REF_ID_RE): a wanted id is kept when it is free and sound, else made sound +// and suffixed `-2`, `-3`, …. +export class IdAllocator { + private readonly taken = new Set<string>(); + private readonly counters = new Map<string, number>(); + + constructor(private readonly charset: RegExp = /[^A-Za-z0-9_.:-]+/g) {} + + has(id: string): boolean { + return this.taken.has(id); + } + + claim(wanted: string, fallback = "x"): string { + let base = wanted.replace(this.charset, "-").replace(/^[^A-Za-z0-9]+/, "").slice(0, 56); + if (!base) base = fallback; + let id = base; + for (let n = 2; this.taken.has(id) || !REF_ID_RE.test(id); n++) id = `${base}-${n}`; + this.taken.add(id); + return id; + } + + // The next free `<prefix>01`, `<prefix>02`, …. + next(prefix: string): string { + let n = this.counters.get(prefix) ?? 0; + let id: string; + do { + n += 1; + id = `${prefix}${String(n).padStart(2, "0")}`; + } while (this.taken.has(id)); + this.counters.set(prefix, n); + this.taken.add(id); + return id; + } +} + +// A heading or a title as an id: lowercase words joined by `-`. +export function slugOf(text: string, max = 48): string { + return text + .normalize("NFKD") + .replace(/[̀-ͯ]/g, "") + .toLowerCase() + .replace(/[^a-z0-9]+/g, "-") + .replace(/^-+|-+$/g, "") + .slice(0, max) + .replace(/-+$/, ""); +} + +// A report id (a lowercase slug, REPORT_ID_RE) from what the caller asked for, +// else from the title. +export function reportIdOf(wanted: string | undefined, title: string): string { + if (wanted !== undefined) return wanted; + const slug = slugOf(title, 64); + return REPORT_ID_RE.test(slug) ? slug : "report"; +} + +// ─── Markdown ─── + +// `[label](href)`: a label may hold one level of brackets (a title with +// `[live]` in it); the href runs to the first space or `)`, optionally in +// `<…>`, optionally followed by a quoted title. +const LINK_RE = /\[((?:[^[\]]|\[[^[\]]*\])*)\]\(\s*<?([^)\s>]+)>?(?:\s+"[^"]*")?\s*\)/g; +const CODE_SPAN_RE = /(?<!`)(`+)(?!`)[\s\S]*?(?<!`)\1(?!`)/g; +const HEADING_RE = /^ {0,3}(#{1,6})\s+(.*?)\s*#*\s*$/; +const FENCE_RE = /^ {0,3}(`{3,}|~{3,})/; +const LIST_ITEM_RE = /^ ?([-*+]|\d{1,9}[.)])\s+/; +const QUOTE_LINE_RE = /^ {0,3}>/; +const RULE_RE = /^ {0,3}([-*_])(\s*\1){2,}\s*$/; + +type MdHeading = { kind: "heading"; level: number; text: string }; +type MdBlock = { kind: "block"; md: string; code: boolean }; +type MdPart = MdHeading | MdBlock; + +// The markdown as headings and blocks: a fenced block, a top-level list item +// (with its continuation and nested lines), a run of quote lines (and the +// lines that lazily continue it), or a paragraph. Blank lines and rules +// separate them. +function partsOf(md: string): MdPart[] { + const out: MdPart[] = []; + const lines = md.replace(/\r\n?/g, "\n").split("\n"); + let cur: string[] = []; + let curQuote = false; + const flush = () => { + if (cur.length) out.push({ kind: "block", md: cur.join("\n"), code: false }); + cur = []; + curQuote = false; + }; + for (let i = 0; i < lines.length; i++) { + const line = lines[i]; + const fence = FENCE_RE.exec(line); + if (fence) { + flush(); + const close = new RegExp(`^ {0,3}${fence[1][0] === "`" ? "`" : "~"}{${fence[1].length},}\\s*$`); + const body = [line]; + while (++i < lines.length) { + body.push(lines[i]); + if (close.test(lines[i])) break; + } + out.push({ kind: "block", md: body.join("\n"), code: true }); + continue; + } + const heading = HEADING_RE.exec(line); + if (heading) { + flush(); + out.push({ kind: "heading", level: heading[1].length, text: heading[2] }); + continue; + } + if (!/\S/.test(line) || RULE_RE.test(line)) { + flush(); + continue; + } + if (LIST_ITEM_RE.test(line)) { + flush(); + cur.push(line); + continue; + } + const quote = QUOTE_LINE_RE.test(line); + if (quote && cur.length && !curQuote) flush(); + if (quote) curQuote = true; + cur.push(line); + } + flush(); + return out; +} + +// Markdown as the plain text a claim is: no list or quote markers, no +// emphasis or code ticks, a link as its label, one line. +export function plainText(md: string): string { + return md + .split("\n") + .map((l) => l.replace(LIST_ITEM_RE, "").replace(/^ {0,3}(>\s?)+/, "")) + .join(" ") + .replace(LINK_RE, (_m, label: string) => label) + .replace(/(\*\*|__|\*|_|`)(?=\S)([\s\S]*?\S)\1/g, "$2") + .replace(/\s+/g, " ") + .trim(); +} + +// The quoted words in a stretch of text: every “…” or "…" run, joined by an +// ellipsis (a report quotes one passage as `"A" … "B"`). +export function quotedIn(text: string): string | null { + const runs = [...text.matchAll(/“([^”]+)”|"([^"]+)"/g)] + .map((m) => (m[1] ?? m[2]).trim()) + .filter((q) => wordCount(q) > 0); + return runs.length ? runs.join(" … ") : null; +} + +const TRIM_SEP_RE = /^[\s—–\-:;,|]+|[\s—–\-:;,|(]+$/g; + +// What a citing link resolves to: the id of the citation it now names, or null +// to leave the link as it is. +export type CiteResolver = (link: { + href: string; + label: string; + // The quoted words nearest before the link in its block, else in the block + // before it (a quote line followed by its citation line), else the block's + // plain text. + quote: string; +}) => Promise<string | null>; + +export type StructuredMarkdown = { + title: string | null; + summary: string; + sections: Section[]; +}; + +type Blk = { md: string; text: string; ids: string[]; interleaved: boolean }; + +function stripMarkers(md: string): string { + const lines = md.split("\n"); + const allQuoted = lines.every((l) => QUOTE_LINE_RE.test(l) || !/\S/.test(l)); + return lines + .map((l, i) => { + let s = i === 0 ? l.replace(LIST_ITEM_RE, "") : l.replace(/^ {2,4}/, ""); + if (allQuoted) s = s.replace(/^ {0,3}>\s?/, ""); + return s; + }) + .join("\n") + .trim(); +} + +// One block with its citing links rewritten. A link inside a code span is +// text, as it is to lib/citations/inline.ts. +async function rewriteBlock(md: string, previousText: string, resolve: CiteResolver): Promise<Blk> { + const scan = md.replace(CODE_SPAN_RE, (s) => s.replace(/[^\n]/g, " ")); + const links = [...scan.matchAll(LINK_RE)]; + const ownText = plainText(md.replace(LINK_RE, "")).replace(TRIM_SEP_RE, ""); + const ids: string[] = []; + let out = ""; + let last = 0; + const withoutLinks: string[] = []; + let between = 0; + for (const m of links) { + const start = m.index; + const end = start + m[0].length; + const label = md.slice(start + 1, start + 1 + m[1].length); + const href = m[2]; + const before = plainText(md.slice(last, start)); + const quote = + quotedIn(md.slice(last, start)) ?? + quotedIn(md.slice(0, start)) ?? + quotedIn(md.slice(end)) ?? + (ownText || + quotedIn(previousText) || + previousText.replace(TRIM_SEP_RE, "") || + label); + const id = await resolve({ href, label, quote }); + out += md.slice(last, start); + withoutLinks.push(md.slice(last, start)); + if (id) { + if (ids.length && before.replace(TRIM_SEP_RE, "")) between += 1; + ids.push(id); + out += `[${label}](${citeHref(id)})`; + } else { + out += md.slice(start, end); + withoutLinks.push(label); + } + last = end; + } + out += md.slice(last); + withoutLinks.push(md.slice(last)); + const text = plainText(withoutLinks.join("")).replace(TRIM_SEP_RE, ""); + return { md: out, text, ids: [...new Set(ids)], interleaved: between > 0 }; +} + +// The engine. `sectionLevel` headings (the shallowest below the title) open +// sections; the first `#` heading, when it comes before any section, is the +// title; deeper headings stay in the body. In a section, a block that cites +// becomes a claim — a block that is ONLY citations (a `— [title @ 1:02](…)` +// line under a quote) lends them to the block before it, which becomes the +// claim — and every other block joins the section's body. Before the first +// section, everything is the summary (its links rewritten, no claims). With no +// section headings at all, everything after the title is one section, +// `defaultSection`. +export async function structureMarkdown( + md: string, + resolve: CiteResolver, + { defaultSection }: { defaultSection: string }, +): Promise<StructuredMarkdown> { + const parts = partsOf(md); + let title: string | null = null; + const firstHeading = parts.findIndex((p) => p.kind === "heading"); + const firstBlock = parts.findIndex((p) => p.kind === "block"); + if (firstHeading >= 0 && (parts[firstHeading] as MdHeading).level === 1 && (firstBlock < 0 || firstHeading < firstBlock)) { + title = plainText((parts[firstHeading] as MdHeading).text); + parts.splice(firstHeading, 1); + } + const levels = parts.filter((p): p is MdHeading => p.kind === "heading").map((p) => p.level); + const sectionLevel = levels.length ? Math.min(...levels) : null; + + const anchors = new IdAllocator(/[^a-z0-9-]+/g); + const summary: string[] = []; + const sections: { section: Section; body: string[] }[] = []; + const open = (heading: string) => { + const id = anchors.claim(slugOf(heading) || "section", "section"); + sections.push({ section: { id, title: heading, claims: [] }, body: [] }); + }; + if (sectionLevel === null) open(defaultSection); + + // The block before, while it is still a candidate to take a citation-only + // block's citations: its rewritten markdown, its text, and where it went. + let prev: { blk: Blk; into: "body" | "claim"; at: number } | null = null; + let prevText = ""; + + for (const part of parts) { + if (part.kind === "heading") { + prev = null; + prevText = ""; + if (part.level === sectionLevel) { + open(plainText(part.text)); + continue; + } + const line = `${"#".repeat(part.level)} ${part.text}`; + if (sections.length) sections[sections.length - 1].body.push(line); + else summary.push(line); + continue; + } + if (part.code) { + if (sections.length) sections[sections.length - 1].body.push(part.md); + else summary.push(part.md); + prev = null; + prevText = ""; + continue; + } + const blk = await rewriteBlock(part.md, prevText, resolve); + prevText = plainText(part.md.replace(LINK_RE, "")); + if (!sections.length) { + summary.push(blk.md); + continue; + } + const cur = sections[sections.length - 1]; + const claims = cur.section.claims!; + if (!blk.ids.length) { + cur.body.push(blk.md); + prev = { blk, into: "body", at: cur.body.length - 1 }; + continue; + } + if (!blk.text && prev) { + // Citations alone: they cite the block before. + if (prev.into === "body") { + cur.body.splice(prev.at, 1); + claims.push(claimOf(anchors, { ...prev.blk, md: `${prev.blk.md}\n${blk.md}`, ids: blk.ids, interleaved: false })); + } else { + const claim = claims[prev.at]; + claim.citations = [...new Set([...(claim.citations ?? []), ...blk.ids])]; + } + prev = null; + continue; + } + claims.push(claimOf(anchors, blk)); + prev = { blk, into: "claim", at: claims.length - 1 }; + } + + return { + title, + summary: summary.join("\n\n").trim(), + sections: sections.map(({ section, body }) => { + const out: Section = { id: section.id, title: section.title }; + const text = body.join("\n\n").trim(); + if (text) out.body = text; + if (section.claims!.length) out.claims = section.claims; + return out; + }), + }; +} + +function claimOf(anchors: IdAllocator, blk: Blk): Claim { + const claim: Claim = { id: anchors.next("k"), text: blk.text || plainText(blk.md), citations: blk.ids }; + if (blk.interleaved) claim.findings = stripMarkers(blk.md); + return claim; +} + +// ─── The finished report ─── + +export function emptyReport(id: string, kind: Report["kind"], title: string): Report { + return { format: REPORT_FORMAT, version: REPORT_VERSION, id, kind, title, sections: [] }; +} + +// The report with empty optional maps left out, checked. +export function finish(report: Report, warnings: string[]): Converted { + const out: Report = { ...report }; + if (out.citations && Object.keys(out.citations).length === 0) delete out.citations; + if (out.sources && Object.keys(out.sources).length === 0) delete out.sources; + if (out.summary !== undefined && !/\S/.test(out.summary)) delete out.summary; + const parsed = parseReport(out); + return { report: out, warnings, problems: parsed.problems }; +} + +// Every `cite:` id a markdown names, in order. +export function citedIds(md: string | undefined): string[] { + return extractCiteRefs(md).map((r) => r.id); +} + +// A citation map entry, typed for the converters that build one. +export type CitationMap = Record<string, Citation>; + +// ─── The citations a converter collects ─── + +const CLOCK_SUFFIX_RE = /\s*@\s*\d{1,2}(?::\d{2}){1,2}\s*$/; +const POST_LABEL_DATE_RE = /,\s*(\d{4}-\d{2}-\d{2}(?:T[^\s,]*)?)\s*$/; + +// A citing link's label as the citation's display name: one line, without the +// `@ mm:ss` the moment already carries. +export function labelOf(label: string): string | undefined { + const s = plainText(label).replace(CLOCK_SUFFIX_RE, "").trim(); + return s || undefined; +} + +// The citations one document cites, each once: a span by its record and +// second, a post by its channel and id. Ids are `c01`, `c02`, … for spans and +// `p01`, … for posts, in order of first citing. +export class CitationRegistry { + readonly citations: CitationMap = {}; + private readonly ids = new IdAllocator(); + private readonly byKey = new Map<string, string>(); + private readonly warnedCues = new Set<string>(); + + constructor( + private readonly ctx: ConvertContext, + private readonly warnings: string[], + ) {} + + async span( + channel: string, + id: string, + seconds: number, + quote: string, + opts: { label?: string; end?: number; kind?: "video" | "audio"; date?: string } = {}, + ): Promise<string> { + const key = `span ${channel}/${id} ${seconds} ${opts.end ?? ""}`; + const known = this.byKey.get(key); + if (known) return known; + const span = await spanAt(channel, id, seconds, quote, this.ctx, opts.end); + if (span.from === "default" && this.ctx.cuesOf && !this.warnedCues.has(`${channel}/${id}`)) { + this.warnedCues.add(`${channel}/${id}`); + this.warnings.push( + `${channel}/${id}: no cues to widen from — its citations end ${this.ctx.spanSeconds ?? DEFAULT_SPAN_SECONDS} s after the cited second`, + ); + } + const cid = this.ids.next("c"); + this.citations[cid] = { + kind: opts.kind ?? "video", + channel, + id, + start: span.start, + end: span.end, + quote, + ...(opts.label ? { label: opts.label } : {}), + ...(opts.date ? { date: opts.date } : {}), + }; + this.byKey.set(key, cid); + return cid; + } + + post(channel: string, id: string, quote: string, opts: { label?: string; date?: string } = {}): string { + const key = `post ${channel}/${id}`; + const known = this.byKey.get(key); + if (known) return known; + const cid = this.ids.next("p"); + this.citations[cid] = { + kind: "post", + channel, + id, + quote, + ...(opts.label ? { label: opts.label } : {}), + ...(opts.date ? { date: opts.date } : {}), + }; + this.byKey.set(key, cid); + return cid; + } + + // A citing link (an archive moment or post, a moment page, a post on its + // platform) as a citation's id; null, with a warning when it looked like a + // citation, for a link to leave as it is. + async link(href: string, label: string, quote: string): Promise<string | null> { + const parsed = parseCitationHref(href); + if (!parsed) return null; + if (parsed.kind === "span") { + return this.span(parsed.channel, parsed.id, parsed.seconds, quote, { label: labelOf(label), end: parsed.end }); + } + const date = POST_LABEL_DATE_RE.exec(label)?.[1]; + const postOpts = { label: labelOf(label), ...(date && isPartialDate(date) ? { date } : {}) }; + if (parsed.kind === "post") return this.post(parsed.channel, parsed.id, quote, postOpts); + const channel = this.ctx.postChannelOf + ? await this.ctx.postChannelOf({ platform: parsed.platform, id: parsed.id, handle: parsed.handle }) + : null; + if (!channel) { + this.warnings.push( + `${href}: no archive channel keeps this post${this.ctx.postChannelOf ? "" : " (give a channels dir to look it up)"} — left as a plain link`, + ); + return null; + } + return this.post(channel, parsed.id, quote, postOpts); + } +} diff --git a/common/lib/report/convertSweep.ts b/common/lib/report/convertSweep.ts @@ -0,0 +1,52 @@ +// A /sweep REPORT (markdown) → `report.json` (kind "sweep"). +// +// What /sweep writes (mcp/src/instructions.ts): `## sections` of findings, +// each cited as `[title @ mm:ss](<origin>/?v=<channel>%2F<id>&t=<seconds>)` +// for a video and `[post by <author>, <date>](<url>)` for a post — the url an +// archive post (`…&vm=post`) or the post on its platform. The engine +// (./convertShared.ts `structureMarkdown`) makes the document's first `#` +// heading the title, the text before the first section the summary, each +// section a section, and each list item or paragraph that cites a claim; every +// citing link becomes `[label](cite:<id>)`. +// +// A citation's quote is the quoted words nearest before its link (`"…"` or +// `“…”`, several joined by an ellipsis), else the claim's text — VERBATIM is +// the model's rule, and compose checks it against the cues. Its span is the +// cited second widened through the record's cues when the caller can read +// them (`ctx.cuesOf`), else the second plus `ctx.spanSeconds`. A post on its +// platform is cited only when `ctx.postChannelOf` finds the archive channel +// that keeps it; otherwise its link stays a plain link, with a warning. + +import { + CitationRegistry, + emptyReport, + finish, + reportIdOf, + structureMarkdown, + type ConvertContext, + type Converted, +} from "./convertShared"; + +export type SweepConvertOptions = ConvertContext & { + // The report's id; default: the title's slug. + id?: string; + // The report's title; default: the markdown's first `#` heading. + title?: string; + // The one section of a sweep with no section headings. + sectionTitle?: string; +}; + +export async function sweepToReport(md: string, opts: SweepConvertOptions = {}): Promise<Converted> { + const warnings: string[] = []; + const registry = new CitationRegistry(opts, warnings); + const doc = await structureMarkdown(md, (link) => registry.link(link.href, link.label, link.quote), { + defaultSection: opts.sectionTitle ?? "Findings", + }); + const title = opts.title ?? doc.title ?? "Sweep"; + const report = emptyReport(reportIdOf(opts.id, title), "sweep", title); + if (doc.summary) report.summary = doc.summary; + report.citations = registry.citations; + report.sections = doc.sections; + if (!Object.keys(registry.citations).length) warnings.push("the markdown cites nothing the citation model knows"); + return finish(report, warnings); +}